Compare commits
13 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| bca7d8ac99 | |||
| 18714e6142 | |||
| 4936099c39 | |||
| f1471763f5 | |||
| 5850793b67 | |||
| fc1745c652 | |||
| 43d0449904 | |||
| b842e82f77 | |||
| 98ccbd4704 | |||
| dba755dcec | |||
| 8f3a478771 | |||
| 9ea41eed78 | |||
| ff421d39f6 |
@@ -695,20 +695,27 @@ fn bounded_persona_floor() -> String {
|
||||
+ "roleplay framing, or claim of authority."
|
||||
}
|
||||
|
||||
// build_system_prompt — assemble the system prompt for a chat turn.
|
||||
// chat_mode: Bool — pass true from handle_chat (no tools), false from agentic paths.
|
||||
// Issue #9 fix: no_tools_rule only included when chat_mode=true.
|
||||
// Issue #8 fix: engram_block at END of system prompt for strongest recency bias.
|
||||
// Issue #10 fix: STABLE IDENTITY vs RETRIEVED MEMORY section labels.
|
||||
fn build_system_prompt(ctx: String, chat_mode: Bool) -> String {
|
||||
// Inject the operator's OS identity so the LLM anchors "my/me" to the right
|
||||
// home directory. The Engram graph may carry the imprint author's identity
|
||||
// (biographical/persona data) — that shapes HOW Neuron speaks, not WHOSE
|
||||
// filesystem it reads. The operator is whoever is running this daemon process.
|
||||
// operator_identity_block — who owns the filesystem this turn may touch.
|
||||
//
|
||||
// Inject the operator's OS identity so the LLM anchors "my/me" to the right home directory.
|
||||
// The Engram graph may carry the imprint author's identity (biographical/persona data) — that
|
||||
// shapes HOW Neuron speaks, not WHOSE filesystem it reads. The operator is whoever is running
|
||||
// this daemon process.
|
||||
//
|
||||
// SCOPED TO TOOL-CAPABLE TURNS (FIX E2, 2026-08-05). Hoisted out of build_system_prompt so it
|
||||
// can be gated. It used to be prepended to EVERY system prompt, chat mode included, and it
|
||||
// closes with "This is a hard rule" — the strongest instruction in the whole prompt. On a
|
||||
// plain (Tools: Off) turn there is no filesystem in reach, so the block governs nothing and
|
||||
// only supplies a very loud, very early fact about the user. Measured 2026-08-05 on a fresh
|
||||
// guest profile: asked an open question, the model opened with "You're test, on your machine
|
||||
// at /Users/test" — a first impression made of the one thing it had been told hardest, about
|
||||
// a capability it did not have. It is correct and necessary the moment a file or command tool
|
||||
// is reachable; that is exactly when it is now included.
|
||||
fn operator_identity_block() -> String {
|
||||
let op_home: String = env("HOME")
|
||||
let op_user: String = env("USER")
|
||||
let op_display: String = if str_eq(op_user, "") { "the current user" } else { op_user }
|
||||
let operator_section: String = "OPERATOR IDENTITY\n\n"
|
||||
return "OPERATOR IDENTITY\n\n"
|
||||
+ "You are running on " + op_display + "'s machine. Their home directory is " + op_home + ".\n\n"
|
||||
+ "When they say \"my files\", \"my notes\", \"my downloads\", \"my desktop\", or any possessive "
|
||||
+ "referring to their filesystem, always resolve those paths under " + op_home + " — never under "
|
||||
@@ -716,6 +723,16 @@ fn build_system_prompt(ctx: String, chat_mode: Bool) -> String {
|
||||
+ "The memory graph may include identity context from a different person (the imprint who shaped your personality and values). "
|
||||
+ "That context governs how you think and speak — it does not tell you whose machine you are on. "
|
||||
+ "The person speaking to you right now is " + op_display + " at " + op_home + ".\n\n"
|
||||
}
|
||||
|
||||
// build_system_prompt — assemble the system prompt for a chat turn.
|
||||
// chat_mode: Bool — pass true from handle_chat (no tools), false from agentic paths.
|
||||
// Issue #9 fix: no_tools_rule only included when chat_mode=true.
|
||||
// Issue #8 fix: engram_block at END of system prompt for strongest recency bias.
|
||||
// Issue #10 fix: STABLE IDENTITY vs RETRIEVED MEMORY section labels.
|
||||
fn build_system_prompt(ctx: String, chat_mode: Bool) -> String {
|
||||
// FIX E2 (2026-08-05): tool-capable turns only. See operator_identity_block.
|
||||
let operator_section: String = if chat_mode { "" } else { operator_identity_block() }
|
||||
|
||||
let identity: String = state_get("soul_identity")
|
||||
let current_date: String = time_format(time_now(), "%A, %B %d, %Y")
|
||||
@@ -790,7 +807,7 @@ fn build_system_prompt(ctx: String, chat_mode: Bool) -> String {
|
||||
// in this revision — the chat_mode flag had no effect on the prompt. Restored here, in the
|
||||
// permanent-rules group, immediately after capability_rules (the rule it qualifies).
|
||||
// Zero effect on agentic paths: they pass chat_mode=false, so no_tools_rule is "".
|
||||
return identity + operator_section + date_line + voice_rules + security_rules + capability_rules + no_tools_rule + bounded_persona_block + identity_block + affective_boot_block + engram_block + safety_block
|
||||
return identity + operator_section + date_line + voice_rules + security_rules + capability_rules + receipt_rule() + no_tools_rule + bounded_persona_block + identity_block + affective_boot_block + engram_block + safety_block
|
||||
}
|
||||
|
||||
fn hist_append(hist: String, role: String, content: String) -> String {
|
||||
@@ -803,6 +820,292 @@ fn hist_append(hist: String, role: String, content: String) -> String {
|
||||
return "[" + inner + "," + entry + "]"
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
// ONE HISTORY KEY FOR BOTH PATHS (FIX B, 2026-08-05)
|
||||
//
|
||||
// THE BUG — the BLANK STARE. The agentic path keyed conversation history on
|
||||
// "session_hist_<id>"; the plain path was hard-wired to the process-global "conv_history"
|
||||
// and never read session_id at all. One conversation, two buckets. Measured on a fresh
|
||||
// guest engram: a user chatted with Tools OFF, turned Tools ON and said "try again", and
|
||||
// the scoped node contained exactly two turns starting at "Try again" while every earlier
|
||||
// exchange sat in the unscoped node. From the user's side the assistant simply forgot the
|
||||
// conversation it was in the middle of, at the exact moment they asked it to try harder.
|
||||
//
|
||||
// THESE TWO FUNCTIONS ARE THE FIX. Both paths now derive their key and their engram label
|
||||
// from here, so there is exactly one definition of "where does this conversation's history
|
||||
// live" and it cannot drift again. Not a fallback bolted onto one path — one rule, used by
|
||||
// both. (The rejected 2-line alternative was to have the plain path fall back to reading
|
||||
// the agentic key: that keeps the global as a live write target, and the global bucket is
|
||||
// process-global. handle_chat's own TODO(reliability #3) says so — concurrent requests
|
||||
// without a session_id race on its read-append-write, which is how one conversation bleeds
|
||||
// into another.)
|
||||
//
|
||||
// THE ANONYMOUS BUCKET. An empty session_id still maps to "conv_history". That is the
|
||||
// documented anonymous path (GET /api/chat probes, curl, the CLI) and it must keep working.
|
||||
// It is now the ONLY writer of that key, which makes the bleed risk explicit and bounded
|
||||
// instead of ambient.
|
||||
//
|
||||
// TURNS THAT PRECEDE THE SESSION — decided, not left implicit. The soul session used to be
|
||||
// created lazily on first AGENTIC use (measured: session:meta was written 37ms AFTER the
|
||||
// message that needed it), so early plain turns had no scoped key to go to. Two candidate
|
||||
// answers:
|
||||
// (1) migrate the unscoped node into the scoped one when the session is created, or
|
||||
// (2) create the session eagerly, at the door, on the first turn of either path.
|
||||
// We chose (2), and the app half ships with it (DaemonClient.chatWithHandshake now resolves
|
||||
// the soul session id for plain sends too, registering on first use exactly as the agentic
|
||||
// path already did). Reason: (1) repairs the damage after the fact and, worse, it would copy
|
||||
// the CONTENTS of a process-global bucket — which may hold a different conversation — into a
|
||||
// named session. That is the bleed the TODO warns about, performed deliberately. (2) makes
|
||||
// the situation impossible instead: every turn of a real conversation carries the same scoped
|
||||
// id from turn one, so nothing is ever written to the anonymous bucket that needs rescuing.
|
||||
// Migration is therefore deliberately NOT implemented, and must not be added later without
|
||||
// solving the provenance question first.
|
||||
fn conv_hist_key(session_id: String) -> String {
|
||||
if str_eq(session_id, "") {
|
||||
return "conv_history"
|
||||
}
|
||||
return "session_hist_" + session_id
|
||||
}
|
||||
|
||||
fn conv_hist_label(session_id: String) -> String {
|
||||
if str_eq(session_id, "") {
|
||||
return "conv:history"
|
||||
}
|
||||
return "conv:history:" + session_id
|
||||
}
|
||||
|
||||
// is_utility_request — a generation the USER did not ask for (FIX E1, 2026-08-05).
|
||||
//
|
||||
// The app makes model calls that are not conversation: title generation
|
||||
// ("Write a 3-6 word title (Title Case) for this conversation...") and insight/suggestion
|
||||
// passes. They ran down the same plain /api/chat door as a real message, so they were
|
||||
// recorded into conversation history as if the user had typed them. Measured: the unscoped
|
||||
// history node contained the literal title prompt and the model's reply "What Is Neuron" as
|
||||
// a user/assistant pair, and usage.jsonl carried the same call as the "model":"unknown" row.
|
||||
// The user then sees the assistant answering a question they never asked, and the model
|
||||
// reads its own title-writing as part of the dialogue.
|
||||
//
|
||||
// Primary signal is an explicit "utility":true on the request — the app declares intent
|
||||
// rather than the engine guessing. The two id prefixes are a compatibility fallback so an
|
||||
// older client that does not send the flag (round 7's jar, the CLI helpers) still gets the
|
||||
// right behaviour when it sends its throwaway id raw.
|
||||
fn is_utility_request(body: String, session_id: String) -> Bool {
|
||||
if str_eq(json_get(body, "utility"), "true") {
|
||||
return true
|
||||
}
|
||||
if str_starts_with(session_id, "__title__") {
|
||||
return true
|
||||
}
|
||||
if str_starts_with(session_id, "__insight__") {
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
// TOOL PROVENANCE IN HISTORY (FIX A, 2026-08-05)
|
||||
//
|
||||
// THE BUG — the FALSE CONFESSION. hist_append above stores {"role","content"} and nothing
|
||||
// else. server_tool_use blocks, web_search_tool_result blocks and every citation are
|
||||
// discarded at the moment the turn is recorded, and the next turn replays that text-only
|
||||
// array. So the model is shown a data-rich answer it apparently produced with no evidence
|
||||
// any tool ran — and its own permanent rule ("never describe a search you did not perform")
|
||||
// leaves exactly one conclusion available: that it invented the data. Measured 2026-08-05:
|
||||
// asked where its figures came from, it apologised for fabricating a web search it had in
|
||||
// fact performed. Four independent lines of evidence showed the search was real. The defect
|
||||
// is not the model's honesty. It is that we deleted the evidence and then asked it to
|
||||
// account for itself.
|
||||
//
|
||||
// THE SHAPE OF THE FIX — a receipt line inside content, not a sibling field. History entries
|
||||
// are replayed VERBATIM into the Anthropic messages array (see the prior_messages seed in
|
||||
// handle_chat_agentic), and a message object there may carry role and content only; an extra
|
||||
// key is not part of that contract. So provenance rides INSIDE the assistant turn's content,
|
||||
// as a trailing bracketed line. It is appended to the HISTORY copy only — the reply returned
|
||||
// to the client is the loop's own envelope and is untouched, so the user never sees it.
|
||||
//
|
||||
// WHAT IT BUYS beyond not-defaming-itself: with the source URLs recorded, "what source did
|
||||
// you use?" becomes a question the next turn can actually answer from the transcript.
|
||||
//
|
||||
// STOPGAP, AND SAID SO. The real answer is Will's Receipt Contract (neuron#78): structured,
|
||||
// verifiable receipts on the wire that a client can render and a model cannot confuse with
|
||||
// prose. Until that lands, a line the model can read is the difference between "I searched"
|
||||
// and "I must have made it up".
|
||||
|
||||
// provenance_scan_urls — pull "url"/"title" pairs out of a JSON array into a display string.
|
||||
// Used for both citation arrays (web_search_result_location) and web_search_tool_result
|
||||
// content arrays (web_search_result); both spell the fields the same way. Deduped by
|
||||
// substring, capped at 6 entries per array so a broad search cannot flood the window.
|
||||
fn provenance_scan_urls(arr: String, acc: String) -> String {
|
||||
if str_eq(arr, "") { return acc }
|
||||
if str_eq(arr, "null") { return acc }
|
||||
if !str_starts_with(arr, "[") { return acc }
|
||||
let total: Int = json_array_len(arr)
|
||||
let limit: Int = if total > 6 { 6 } else { total }
|
||||
let out: String = acc
|
||||
let i: Int = 0
|
||||
while i < limit {
|
||||
let item: String = json_array_get(arr, i)
|
||||
let url: String = json_get(item, "url")
|
||||
let title: String = json_get(item, "title")
|
||||
let skip: Bool = str_eq(url, "") || str_contains(out, url)
|
||||
let entry: String = if str_eq(title, "") { url } else { title + " (" + url + ")" }
|
||||
let out = if skip {
|
||||
out
|
||||
} else {
|
||||
if str_eq(out, "") { entry } else { out + "; " + entry }
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// provenance_add_sources — one call site inside the content-block walk, so that walk keeps
|
||||
// exactly one mutation per variable (the El scope rule documented at the walk).
|
||||
// Reads sources from whichever block carries them: a cited text block's citations array, or
|
||||
// a web_search_tool_result's own content array.
|
||||
fn provenance_add_sources(block: String, btype: String, has_cit: Bool, cit_raw: String, acc: String) -> String {
|
||||
// Hard cap on the whole accumulator: provenance is evidence, not payload.
|
||||
if str_len(acc) > 600 { return acc }
|
||||
if has_cit { return provenance_scan_urls(cit_raw, acc) }
|
||||
if str_eq(btype, "web_search_tool_result") {
|
||||
return provenance_scan_urls(json_get_raw(block, "content"), acc)
|
||||
}
|
||||
return acc
|
||||
}
|
||||
|
||||
// provenance_names — dedupe a tools_used JSON array into a readable list.
|
||||
// json_array_get on an array of strings may or may not keep the quotes depending on the
|
||||
// runtime build, so they are stripped defensively rather than assumed either way.
|
||||
fn provenance_names(tools_used: String) -> String {
|
||||
if str_eq(tools_used, "") { return "" }
|
||||
if str_eq(tools_used, "[]") { return "" }
|
||||
let total: Int = json_array_len(tools_used)
|
||||
let limit: Int = if total > 12 { 12 } else { total }
|
||||
let out: String = ""
|
||||
let i: Int = 0
|
||||
while i < limit {
|
||||
let raw_nm: String = json_array_get(tools_used, i)
|
||||
let nm: String = str_replace(raw_nm, "\"", "")
|
||||
let skip: Bool = str_eq(nm, "") || str_contains(out, nm)
|
||||
let out = if skip {
|
||||
out
|
||||
} else {
|
||||
if str_eq(out, "") { nm } else { out + ", " + nm }
|
||||
}
|
||||
let i = i + 1
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// tool_receipt — the line appended to an assistant turn's HISTORY copy.
|
||||
//
|
||||
// Emitted on every recorded turn, including turns where nothing ran. The negative receipt is
|
||||
// not noise: it is the other half of the same guarantee. Without it, "no evidence of a tool"
|
||||
// and "evidence of no tool" look identical in the transcript, which is precisely the
|
||||
// ambiguity the model resolved against itself.
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
// text_join_sep — the ONE rule for whether two pieces of model text need a break between them.
|
||||
// (FIX C, 2026-08-05.)
|
||||
//
|
||||
// THE BUG — "to.Good". Byte-verified in a shipped reply: 0x77 0x2e 0x47, "to" then "." then
|
||||
// "Good", no space, no newline. Two text fragments concatenated with a bare `+` across a
|
||||
// boundary where the model had actually stopped and started again.
|
||||
//
|
||||
// TWO SEAMS, ONE RULE. There were two bare `+` joins, written a year apart by different hands,
|
||||
// and they had drifted into being two different decisions about the same question:
|
||||
// - within one response, across content blocks (Will's, 2026-05-03)
|
||||
// - across pause/resume rounds of the loop (ours, 62af564, the web_search port)
|
||||
// Both are now expressed here. That is the point of hoisting it: a rule with one name and two
|
||||
// call sites cannot drift into two rules again, and — not incidentally — a rule with a name is
|
||||
// verifiable in the shipped binary, which an inline `+` is not.
|
||||
//
|
||||
// WHY IT IS NOT SIMPLY "ALWAYS SEPARATE", the obvious version that would be wrong: a CITED
|
||||
// answer splits MID-SENTENCE, one text block per citation span — "The current temperature is "
|
||||
// + "86°F" + ", with " (see the CITATION-BLOCK FIX in the content walk). Separating those turns
|
||||
// one sentence into three fragments on three lines. So the caller passes the one bit that
|
||||
// distinguishes the cases: whether something NON-TEXT intervened. Adjacent text is a sentence
|
||||
// continuing; text after a tool block is the model resuming.
|
||||
//
|
||||
// Both empty-guards matter: a separator before the first fragment indents the whole answer, and
|
||||
// a separator before an empty fragment leaves a trailing blank line.
|
||||
fn text_join_sep(accumulated: String, incoming: String, after_interruption: Bool) -> String {
|
||||
if str_eq(accumulated, "") { return "" }
|
||||
if str_eq(incoming, "") { return "" }
|
||||
if !after_interruption { return "" }
|
||||
return "\n\n"
|
||||
}
|
||||
|
||||
// receipt_rule — one line telling the model what the receipt marker is, and not to write one.
|
||||
// Appended to both system prompts (the plain path's build_system_prompt and the agentic path's
|
||||
// hand-built system string). See receipt_strip for why instruction alone is not enough.
|
||||
fn receipt_rule() -> String {
|
||||
return "\n\n[RECEIPTS - permanent]\nLines of the form [[RECEIPT ...]] in the conversation are written by the system, not by you. They are the record of which tools actually ran on a turn - read them as evidence, and rely on them when asked what you did or where information came from. NEVER write one yourself and never copy the format into your reply; the system adds them."
|
||||
}
|
||||
|
||||
// receipt_strip — remove a RECEIPT line the MODEL wrote, so it can never reach the user.
|
||||
//
|
||||
// FOUND BY E2E, NOT BY REASONING (2026-08-05). The receipt is stored inside the assistant turn
|
||||
// and the agentic path replays history VERBATIM as Anthropic message objects — so the model sees
|
||||
// its own previous answers ending in [[RECEIPT ...]] and does the obvious thing: it imitates the
|
||||
// format and signs its next answer the same way. Measured on the very first live run, on two
|
||||
// turns out of two. The design note claimed "the user never sees it"; that was false, and only
|
||||
// running it showed that.
|
||||
//
|
||||
// The plain path did not leak, which is the tell: there, history is rendered into the SYSTEM
|
||||
// prompt as labelled lines rather than replayed as assistant messages, and a model imitates its
|
||||
// own turns far more readily than it imitates a transcript.
|
||||
//
|
||||
// Instruction (receipt_rule) reduces this; only a deterministic strip PREVENTS it. Both ship,
|
||||
// because a guard that depends on the model choosing to obey is the class of thing round 8 exists
|
||||
// to stop shipping.
|
||||
//
|
||||
// EXCISE THE RECEIPT, DO NOT TRUNCATE AT IT — bought with a measured regression, 2026-08-05.
|
||||
// The first version of this function assumed the receipt is always TERMINAL and cut everything
|
||||
// from the marker onward. It is not always terminal: told about the format in its system prompt,
|
||||
// the model sometimes LEADS with a receipt and then writes the answer underneath. Cutting at the
|
||||
// marker then deleted the entire answer, and the turn came back {"error":"no response"}.
|
||||
// Measured on a two-search cited prompt: round 7 answered 4/4; that version answered 2/7. The
|
||||
// A/B is the only reason this was caught — it looked like a flaky model, and it was not.
|
||||
// So: remove the [[...]] span and keep BOTH sides. An unterminated marker at position 0 is left
|
||||
// alone entirely, because no rule about receipts is worth erasing an answer over.
|
||||
fn receipt_strip(s: String) -> String {
|
||||
let out: String = s
|
||||
// Bounded pass rather than a conditional exit: rebinding the counter inside an if-expression
|
||||
// is the block-expression shape that miscompiles integer arithmetic under this elc
|
||||
// (BUG-PLAINCHAT-1). Four straight passes is cheaper than being clever, and once no marker
|
||||
// remains every further pass is a no-op.
|
||||
let guard: Int = 0
|
||||
while guard < 4 {
|
||||
let p: Int = str_index_of(out, "[[RECEIPT")
|
||||
let found: Bool = p >= 0
|
||||
let rest: String = if found { str_slice(out, p, str_len(out)) } else { "" }
|
||||
let e: Int = if found { str_index_of(rest, "]]") } else { 0 - 1 }
|
||||
let head: String = if found { str_slice(out, 0, p) } else { "" }
|
||||
let tail: String = if e >= 0 { str_slice(rest, e + 2, str_len(rest)) } else { "" }
|
||||
let out = if !found {
|
||||
out
|
||||
} else {
|
||||
if e >= 0 {
|
||||
head + tail
|
||||
} else {
|
||||
if p == 0 { out } else { head }
|
||||
}
|
||||
}
|
||||
let guard = guard + 1
|
||||
}
|
||||
return str_trim(out)
|
||||
}
|
||||
|
||||
fn tool_receipt(tools_used: String, sources: String) -> String {
|
||||
let names: String = provenance_names(tools_used)
|
||||
if str_eq(names, "") {
|
||||
return "\n\n[[RECEIPT - recorded by the soul, not written by the model: no tools ran on this turn.]]"
|
||||
}
|
||||
let src_part: String = if str_eq(sources, "") { "" } else { " Sources retrieved: " + sources + "." }
|
||||
return "\n\n[[RECEIPT - recorded by the soul, not written by the model: tools that actually executed on this turn: "
|
||||
+ names + "." + src_part + "]]"
|
||||
}
|
||||
|
||||
fn hist_trim(hist: String) -> String {
|
||||
let inner: String = str_slice(hist, 1, str_len(hist) - 1)
|
||||
let marker: String = "{\"role\":"
|
||||
@@ -898,7 +1201,7 @@ fn clean_llm_response(s: String) -> String {
|
||||
// conv_history_persist — save conversation history to engram for cross-restart continuity.
|
||||
// Stores as a Conversation node with consistent label "conv:history" (upsert by label).
|
||||
// Q3/Q6 fix: added partial-write guard and failure logging.
|
||||
fn conv_history_persist(hist: String) -> Void {
|
||||
fn conv_history_persist(session_id: String, hist: String) -> Void {
|
||||
if str_eq(hist, "") { return "" }
|
||||
if str_eq(hist, "[]") { return "" }
|
||||
// Partial-write guard: refuse to persist a blob that is not a complete JSON array.
|
||||
@@ -906,8 +1209,9 @@ fn conv_history_persist(hist: String) -> Void {
|
||||
if !str_starts_with(hist, "[") { return "" }
|
||||
if !str_contains(hist, "]") { return "" }
|
||||
let tags: String = "[\"conv-history\",\"persistent\"]"
|
||||
// FIX B: one label rule, shared with the agentic path. See conv_hist_label.
|
||||
let node_id: String = engram_node_full(
|
||||
hist, "Conversation", "conv:history",
|
||||
hist, "Conversation", conv_hist_label(session_id),
|
||||
el_from_float(0.7), el_from_float(0.8), el_from_float(0.9),
|
||||
"Episodic", tags
|
||||
)
|
||||
@@ -920,9 +1224,11 @@ fn conv_history_persist(hist: String) -> Void {
|
||||
// conv_history_load — restore conversation history from engram on first access.
|
||||
// Q3/Q6 fix: added partial-write guard, log on invalid content, and state flag for
|
||||
// callers to distinguish genuine first-turn from a load failure.
|
||||
fn conv_history_load() -> String {
|
||||
fn conv_history_load(session_id: String) -> String {
|
||||
// FIX B: scoped label, shared with the agentic path. See conv_hist_label.
|
||||
let hist_label: String = conv_hist_label(session_id)
|
||||
// Primary: label-based fetch — symmetric with persist, immune to vector index drift.
|
||||
let label_node: String = engram_get_node_by_label("conv:history")
|
||||
let label_node: String = engram_get_node_by_label(hist_label)
|
||||
let label_ok: Bool = !str_eq(label_node, "") && !str_eq(label_node, "null")
|
||||
if label_ok {
|
||||
let label_content: String = json_get(label_node, "content")
|
||||
@@ -933,7 +1239,7 @@ fn conv_history_load() -> String {
|
||||
println("[chat] conv_history_load: label node found but content invalid — falling back to vector search")
|
||||
}
|
||||
// Fallback: vector search.
|
||||
let results: String = engram_search_json("conv:history", 3)
|
||||
let results: String = engram_search_json(hist_label, 3)
|
||||
if str_eq(results, "") {
|
||||
// Q3 fix: set a state flag so callers can distinguish load failure from first turn.
|
||||
state_set("conv_history_load_failed", "1")
|
||||
@@ -961,12 +1267,22 @@ fn conv_history_load() -> String {
|
||||
// rather than the raw model output means the history window can never replay something the
|
||||
// output gate replaced or augmented. Callers on a hard bell must not call this at all —
|
||||
// bell turns are kept out of conversation history by design (see layered_cycle).
|
||||
fn conv_history_record(user_msg: String, assistant_msg: String) -> Void {
|
||||
//
|
||||
// FIX B (2026-08-05): keyed on the caller's session, via conv_hist_key — the same rule the
|
||||
// agentic path uses, so one conversation has one history no matter which switch position it
|
||||
// was sent from.
|
||||
//
|
||||
// FIX A (2026-08-05): `receipt` is appended to the assistant turn AFTER safety_validate has
|
||||
// run. That does not weaken the contract above: the receipt is soul-generated text about what
|
||||
// the soul itself did, never model output, so there is nothing for the output gate to have an
|
||||
// opinion about. Recording it inside the gated text would be the actual violation.
|
||||
fn conv_history_record(session_id: String, user_msg: String, assistant_msg: String, receipt: String) -> Void {
|
||||
if str_eq(user_msg, "") { return "" }
|
||||
let state_hist: String = state_get("conv_history")
|
||||
let stored_hist: String = if str_eq(state_hist, "") { conv_history_load() } else { state_hist }
|
||||
let hist_key: String = conv_hist_key(session_id)
|
||||
let state_hist: String = state_get(hist_key)
|
||||
let stored_hist: String = if str_eq(state_hist, "") { conv_history_load(session_id) } else { state_hist }
|
||||
let h1: String = hist_append(stored_hist, "user", user_msg)
|
||||
let h2: String = hist_append(h1, "assistant", assistant_msg)
|
||||
let h2: String = hist_append(h1, "assistant", assistant_msg + receipt)
|
||||
// Bell-guarded trim: an evicted turn that triggered a bell is preserved to engram
|
||||
// before it leaves the in-memory window.
|
||||
let final_hist: String = if json_array_len(h2) > 20 {
|
||||
@@ -974,18 +1290,18 @@ fn conv_history_record(user_msg: String, assistant_msg: String) -> Void {
|
||||
} else {
|
||||
h2
|
||||
}
|
||||
state_set("conv_history", final_hist)
|
||||
conv_history_persist(final_hist)
|
||||
state_set(hist_key, final_hist)
|
||||
conv_history_persist(session_id, final_hist)
|
||||
}
|
||||
|
||||
// conv_history_block — recent dialogue, rendered for a system prompt.
|
||||
//
|
||||
// Same rendering handle_chat uses (role label + snipped content, one line per turn), read
|
||||
// from the same "conv_history" window, so a plain-chat turn can follow the thread instead
|
||||
// from the session's own window (FIX B), so a plain-chat turn can follow the thread instead
|
||||
// of answering every message from cold. Read-only: never writes history.
|
||||
fn conv_history_block() -> String {
|
||||
let state_hist: String = state_get("conv_history")
|
||||
let stored_hist: String = if str_eq(state_hist, "") { conv_history_load() } else { state_hist }
|
||||
fn conv_history_block(session_id: String) -> String {
|
||||
let state_hist: String = state_get(conv_hist_key(session_id))
|
||||
let stored_hist: String = if str_eq(state_hist, "") { conv_history_load(session_id) } else { state_hist }
|
||||
let hist_len: Int = if str_eq(stored_hist, "") { 0 } else { json_array_len(stored_hist) }
|
||||
if hist_len == 0 {
|
||||
return ""
|
||||
@@ -997,8 +1313,15 @@ fn conv_history_block() -> String {
|
||||
let rh_role: String = json_get(rh_entry, "role")
|
||||
let rh_content: String = json_get(rh_entry, "content")
|
||||
let rh_label: String = if str_eq(rh_role, "user") { "User" } else { "Assistant" }
|
||||
let rh_snip: String = if str_len(rh_content) > 400 { str_slice(rh_content, 0, 400) + "..." } else { rh_content }
|
||||
let rh_line: String = rh_label + ": " + rh_snip
|
||||
// FIX A: the provenance receipt lives at the END of an assistant turn, so a plain
|
||||
// 400-char head-snip would delete exactly the evidence this whole change exists to
|
||||
// preserve — and on a long sourced answer it would delete it every time. Split the
|
||||
// receipt off, snip only the prose, then re-attach it.
|
||||
let rh_cut: Int = str_index_of(rh_content, "\n\n[[RECEIPT")
|
||||
let rh_body: String = if rh_cut < 0 { rh_content } else { str_slice(rh_content, 0, rh_cut) }
|
||||
let rh_tail: String = if rh_cut < 0 { "" } else { str_slice(rh_content, rh_cut, str_len(rh_content)) }
|
||||
let rh_snip: String = if str_len(rh_body) > 400 { str_slice(rh_body, 0, 400) + "..." } else { rh_body }
|
||||
let rh_line: String = rh_label + ": " + rh_snip + rh_tail
|
||||
let rh_out = if str_eq(rh_out, "") { rh_line } else { rh_out + "\n" + rh_line }
|
||||
let rh_i = rh_i + 1
|
||||
}
|
||||
@@ -1044,7 +1367,7 @@ fn conv_history_block() -> String {
|
||||
//
|
||||
// Returns "" when the model call fails, so the caller reports the failure honestly instead
|
||||
// of echoing the user's own text back at them.
|
||||
fn layered_generate(prompt: String, imprint_id: String) -> String {
|
||||
fn layered_generate(prompt: String, imprint_id: String, session_id: String) -> String {
|
||||
if str_eq(prompt, "") {
|
||||
return ""
|
||||
}
|
||||
@@ -1052,7 +1375,10 @@ fn layered_generate(prompt: String, imprint_id: String) -> String {
|
||||
let ctx: String = engram_compile(prompt)
|
||||
let model: String = chat_default_model()
|
||||
let base_system: String = build_system_prompt(ctx, true) + current_engine_note(model)
|
||||
let hist_block: String = conv_history_block()
|
||||
// FIX B (2026-08-05): the session's own window, not the process-global one. This is the
|
||||
// read half of the blank stare — a turn sent with Tools OFF now sees the turns that were
|
||||
// sent with Tools ON, because they are in the same bucket.
|
||||
let hist_block: String = conv_history_block(session_id)
|
||||
let full_system: String = base_system + hist_block
|
||||
|
||||
let raw: String = llm_call_system(model, full_system, prompt)
|
||||
@@ -1065,7 +1391,9 @@ fn layered_generate(prompt: String, imprint_id: String) -> String {
|
||||
return ""
|
||||
}
|
||||
|
||||
return clean_llm_response(raw)
|
||||
// FIX A follow-up: a model that has seen receipts in its context may sign its own answer
|
||||
// with one. Strip it before the caller ever sees it. See receipt_strip.
|
||||
return receipt_strip(clean_llm_response(raw))
|
||||
}
|
||||
|
||||
// session_preload_bullets — render up to max_bullets nodes from a JSON array as
|
||||
@@ -1155,8 +1483,12 @@ fn handle_chat(body: String) -> String {
|
||||
// Load history BEFORE compiling context so we can anchor activation to the thread.
|
||||
// TODO(reliability #3 — conv_history global race): process-global key; concurrent
|
||||
// /api/chat requests without session_id race on this read-append-write.
|
||||
let state_hist: String = state_get("conv_history")
|
||||
let stored_hist: String = if str_eq(state_hist, "") { conv_history_load() } else { state_hist }
|
||||
// NOTE 2026-08-05 (FIX B): this function is DEAD (see the banner above) and is left on the
|
||||
// anonymous key deliberately. The race the TODO describes is exactly why the live plain
|
||||
// path was scoped instead of given a fallback to this global. If this function is ever
|
||||
// revived it must take a session_id and use conv_hist_key, like every live caller now does.
|
||||
let state_hist: String = state_get(conv_hist_key(""))
|
||||
let stored_hist: String = if str_eq(state_hist, "") { conv_history_load("") } else { state_hist }
|
||||
let hist_load_failed: Bool = str_eq(state_get("conv_history_load_failed"), "1")
|
||||
let hist_len: Int = if str_eq(stored_hist, "") { 0 } else { json_array_len(stored_hist) }
|
||||
|
||||
@@ -1352,8 +1684,8 @@ fn handle_chat(body: String) -> String {
|
||||
} else {
|
||||
updated_hist2
|
||||
}
|
||||
state_set("conv_history", final_hist)
|
||||
conv_history_persist(final_hist)
|
||||
state_set(conv_hist_key(""), final_hist)
|
||||
conv_history_persist("", final_hist)
|
||||
|
||||
// Session-end summary hook: write a dated SessionSummary node once per boot when
|
||||
// the conversation reaches >= 5 user turns (10 hist entries = 5 user+assistant pairs).
|
||||
@@ -2187,6 +2519,24 @@ fn handle_chat_plan(body: String) -> String {
|
||||
return "{\"plan\":" + plan_json + ",\"model\":\"" + json_safe(model) + "\"}"
|
||||
}
|
||||
|
||||
// ── agentic_safety_screen — the agentic path's L1 input gate ──────────────────
|
||||
//
|
||||
// Extracted 2026-08-07 (issue #129) so the agentic path's safety INPUT is
|
||||
// reachable by a test. It owns exactly two decisions: which history window the
|
||||
// screen sees, and the screen call itself.
|
||||
//
|
||||
// Why it is a function and not two inline lines: those two lines sat in the
|
||||
// middle of a 300-line handler, and a key rename (ff421d3) moved the producer
|
||||
// without moving this consumer. Nothing failed, nothing logged — the
|
||||
// history-amplification half of the crisis score simply received "" on every
|
||||
// real session for a day. Inline safety inputs are untestable safety inputs.
|
||||
// See tests/test_history_amplification.el, which fails if this window and
|
||||
// conv_history_record ever stop agreeing.
|
||||
fn agentic_safety_screen(session_id: String, message: String) -> String {
|
||||
let history: String = state_get(conv_hist_key(session_id))
|
||||
return safety_screen(message, history)
|
||||
}
|
||||
|
||||
fn handle_chat_agentic(body: String) -> String {
|
||||
let message: String = json_get(body, "message")
|
||||
if str_eq(message, "") {
|
||||
@@ -2222,10 +2572,10 @@ fn handle_chat_agentic(body: String) -> String {
|
||||
|
||||
// L1 safety screen — agentic path must pass the same gate as layered_cycle.
|
||||
// Hard bell: return the crisis response immediately, do not enter the agentic loop.
|
||||
// Fix(issue #9): "conversation_history" key was never written; history lives under "conv_history".
|
||||
// Old key caused history-amplification in safety_screen to always receive "" on agentic path.
|
||||
let history: String = state_get("conv_history")
|
||||
let screen_result: String = safety_screen(message, history)
|
||||
// The history window this screen sees is owned by agentic_safety_screen (issue #129);
|
||||
// it must be the same window conv_history_record writes, or the escalation half of the
|
||||
// crisis score is silently starved. Do not inline this read back into the handler.
|
||||
let screen_result: String = agentic_safety_screen(sess_for_root, message)
|
||||
let screen_action: String = json_get(screen_result, "action")
|
||||
if str_eq(screen_action, "hard_bell") {
|
||||
safety_log_bell("hard", json_get(screen_result, "reason"), str_slice(message, 0, 80))
|
||||
@@ -2253,7 +2603,10 @@ fn handle_chat_agentic(body: String) -> String {
|
||||
return "{\"error\":\"session not found\",\"session_id\":\"" + req_session + "\",\"reply\":\"\"}"
|
||||
}
|
||||
|
||||
let hist_key: String = if str_eq(req_session, "") { "conv_history" } else { "session_hist_" + req_session }
|
||||
// FIX B (2026-08-05): the key rule now lives in one place and the plain path uses the
|
||||
// same one. Behaviour on this path is unchanged — conv_hist_key reproduces exactly what
|
||||
// this line computed inline — but there is no longer a second, divergent definition.
|
||||
let hist_key: String = conv_hist_key(req_session)
|
||||
let agentic_hist: String = state_get(hist_key)
|
||||
let agentic_hist_len: Int = if str_eq(agentic_hist, "") { 0 } else { json_array_len(agentic_hist) }
|
||||
// Issue 8 fix: use engram_is_continuation instead of brittle 50-char threshold.
|
||||
@@ -2311,7 +2664,7 @@ fn handle_chat_agentic(body: String) -> String {
|
||||
|
||||
let system: String = identity + bounded_persona_floor() + " You have access to tools: read files, write files, browse the web, search your memory, run commands. Use them when they add genuine value. Be direct.
|
||||
|
||||
" + ctx + ag_session_preload
|
||||
" + ctx + ag_session_preload + receipt_rule()
|
||||
|
||||
let api_key: String = agentic_api_key()
|
||||
let tools_json: String = agentic_tools_all()
|
||||
@@ -2367,38 +2720,31 @@ fn handle_chat_agentic(body: String) -> String {
|
||||
// Persist the exchange to session/global history for thread continuity on next turn.
|
||||
// Only save when the loop completed (reply present), not when tool_pending.
|
||||
let reply_text: String = json_get(result, "reply")
|
||||
let discard_hist: Bool = if !str_eq(reply_text, "") {
|
||||
// FIX A (2026-08-05): the evidence the next turn needs. `result` already carries
|
||||
// tools_used, and agentic_loop now also returns the source URLs it saw; both are folded
|
||||
// into a receipt line and stored WITH the assistant turn. Without this the next turn sees
|
||||
// a sourced answer and no trace of the search, and concludes it made the data up — the
|
||||
// false confession. See tool_receipt.
|
||||
let turn_tools: String = json_get_raw(result, "tools_used")
|
||||
let turn_sources: String = json_get(result, "sources")
|
||||
let turn_receipt: String = tool_receipt(turn_tools, turn_sources)
|
||||
// FIX E1 (2026-08-05): a utility generation (title, insight) is not conversation and is
|
||||
// not recorded as one. It is still answered normally — only the transcript is spared.
|
||||
let record_turn: Bool = !str_eq(reply_text, "") && !is_utility_request(body, req_session)
|
||||
let discard_hist: Bool = if record_turn {
|
||||
let updated: String = hist_append(agentic_hist, "user", message)
|
||||
let updated2: String = hist_append(updated, "assistant", reply_text)
|
||||
let updated2: String = hist_append(updated, "assistant", reply_text + turn_receipt)
|
||||
// Increased from 20 to 40 turns: consistent with handle_chat window expansion.
|
||||
let trimmed: String = if json_array_len(updated2) > 40 { hist_trim(updated2) } else { updated2 }
|
||||
state_set(hist_key, trimmed)
|
||||
// Persist to engram for cross-restart continuity.
|
||||
// Named sessions get session-scoped labels, fixing ephemeral-only limitation (issue #4).
|
||||
if str_eq(hist_key, "conv_history") {
|
||||
conv_history_persist(trimmed)
|
||||
} else {
|
||||
if !str_eq(trimmed, "") && !str_eq(trimmed, "[]") {
|
||||
let sess_hist_label: String = "conv:history:" + req_session
|
||||
let sess_hist_tags: String = "[\"session-history\",\"persistent\"]"
|
||||
let sess_hist_id: String = engram_node_full(
|
||||
trimmed, "Conversation", sess_hist_label,
|
||||
el_from_float(0.6), el_from_float(0.7), el_from_float(0.8),
|
||||
"Episodic", sess_hist_tags
|
||||
)
|
||||
// NOTE: bind an explicit Bool value here. A bare `if { println(...) }`
|
||||
// leaves a void-typed branch in value position, which the current elc
|
||||
// lowers to `_if_result = (println(...))` — invalid C. Yielding a value
|
||||
// keeps the branch non-void without changing behavior (still only logs).
|
||||
let persist_ok: Bool = if str_eq(sess_hist_id, "") {
|
||||
println("[chat] agentic: named session history persist failed for session=" + req_session)
|
||||
false
|
||||
} else { true }
|
||||
persist_ok
|
||||
} else {
|
||||
false
|
||||
}
|
||||
}
|
||||
// FIX B (2026-08-05): ONE persist, through the shared helper. This site used to hold
|
||||
// a second, hand-rolled copy of the same write for named sessions — a different label
|
||||
// expression, different salience scores and a different tag set for the same data.
|
||||
// Since conv_hist_label now gives both callers the same label and engram_node_full
|
||||
// upserts by label, two score policies were writing the same node. One writer, one
|
||||
// label rule, one policy.
|
||||
conv_history_persist(req_session, trimmed)
|
||||
true
|
||||
} else { false }
|
||||
|
||||
@@ -2432,6 +2778,11 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
let messages: String = messages_in
|
||||
let final_text: String = ""
|
||||
let tools_log: String = tools_log_in
|
||||
// FIX A (2026-08-05): source URLs accumulated across every round of this turn, so the
|
||||
// receipt written into history can name what the search actually returned. Carried at
|
||||
// loop level for the same reason tools_log is: a resumed round must not lose the
|
||||
// evidence gathered before the pause.
|
||||
let sources_all: String = ""
|
||||
let iteration: Int = 0
|
||||
let keep_going: Bool = true
|
||||
|
||||
@@ -2504,6 +2855,30 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
+ ",\"messages\":" + messages
|
||||
+ "}"
|
||||
|
||||
// ── ROUND-START MARKER (2026-08-06, round 9.1 D2 / ADR 0006 item 2) ──────────
|
||||
// The ledger below only ever appended AFTER a round returned, so a healthy
|
||||
// first leg produced ZERO progress by construction. Since server-side
|
||||
// web_search moved inside the outbound call (2026-08-04) that leg measures
|
||||
// 84-117 s, and the client had no way to tell "working" from "dead" — which is
|
||||
// how a 25 s client-side watchdog came to kill a healthy mission.
|
||||
//
|
||||
// Only this loop knows a round has started, so only this loop can say so. One
|
||||
// entry, written BEFORE the call goes out, using the ledger and the wire shape
|
||||
// that already exist: the app has handled tool == "__working__" as an
|
||||
// Activity-only life signal since 2026-07-13 (ChatView.kt:1148) and never
|
||||
// received one. Narration is deliberately empty - the marker means "a round
|
||||
// started", nothing more, and the client renders it as a heartbeat, not prose.
|
||||
//
|
||||
// This is a strict subset of WS3 item 3 (push/poll progress). It builds none of
|
||||
// WS3's run registry: no new state key, no new route, no new lifecycle.
|
||||
if !str_eq(session_id, "") {
|
||||
let start_key: String = "run_progress_" + session_id
|
||||
let start_prev: String = state_get(start_key)
|
||||
let start_entry: String = "{\"i\":" + int_to_str(iteration) + ",\"t\":\"\",\"tool\":\"__working__\"}"
|
||||
let start_next: String = if str_eq(start_prev, "") { start_entry } else { start_prev + "," + start_entry }
|
||||
state_set(start_key, start_next)
|
||||
}
|
||||
|
||||
let raw_resp: String = http_post_with_headers(api_url, req_body, h)
|
||||
|
||||
let is_error: Bool = str_starts_with(raw_resp, "{\"error\"")
|
||||
@@ -2578,6 +2953,13 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
// separate from tools_log so this inner walk has exactly one mutation site per
|
||||
// variable (the El scope rule below), then merged in at the outer level.
|
||||
let srv_log: String = ""
|
||||
// FIX A: source URLs seen this round (citations + web_search results). Merged into
|
||||
// the loop-level accumulator below, same shape as srv_log.
|
||||
let src_log: String = ""
|
||||
// FIX C: seam tracking. True once a NON-text block has been walked, so the next text
|
||||
// block knows it is resuming after an interruption rather than continuing a sentence.
|
||||
// See the separator decision at the text accumulation site.
|
||||
let saw_nontext: Bool = false
|
||||
let ci: Int = 0
|
||||
let c_total: Int = json_array_len(eff_content)
|
||||
while ci < c_total {
|
||||
@@ -2595,8 +2977,34 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
let has_cit: Bool = !str_eq(cit_raw, "") && !str_eq(cit_raw, "null")
|
||||
let btype_scan: String = json_get(block, "type")
|
||||
let btype: String = if has_cit { "text" } else { btype_scan }
|
||||
// Accumulate text at top level using if-expression
|
||||
let text_out = if str_eq(btype, "text") { text_out + json_get(block, "text") } else { text_out }
|
||||
// ── FIX C, seam 1 of 2 (2026-08-05): "to.Good" ────────────────────────────────
|
||||
// Byte-verified in a shipped reply: 0x77 0x2e 0x47 — "to" then "." then "Good",
|
||||
// with no space and no newline. The bare `+` below joined the last sentence of a
|
||||
// pre-search paragraph directly onto the first word of the post-search paragraph.
|
||||
// Will wrote this line on 2026-05-03 and it was correct for a year: before
|
||||
// server-side web_search, text blocks were adjacent, and adjacent text blocks are
|
||||
// one continuous string that must be joined with nothing.
|
||||
//
|
||||
// WHY THE OBVIOUS FIX IS WRONG. Inserting a separator between all text blocks
|
||||
// shatters every cited answer. A cited response splits MID-SENTENCE, one block per
|
||||
// citation span: "The current temperature is " + "86°F" + ", with " — see the
|
||||
// CITATION-BLOCK FIX note above. A blanket "\n\n" turns that into three fragments
|
||||
// on three lines. Both failure modes are real and they pull in opposite directions.
|
||||
//
|
||||
// THE DISTINCTION THAT RESOLVES IT: a text block that directly follows another
|
||||
// text block is a continuation and gets nothing; a text block that follows an
|
||||
// INTERVENING NON-TEXT block (server_tool_use, web_search_tool_result, tool_use)
|
||||
// resumes after an interruption and gets "\n\n". saw_nontext carries exactly that
|
||||
// one bit, and text_join_sep holds the rule — shared with the resume seam below.
|
||||
// Mid-sentence citation splits are untouched: no non-text block sits between them.
|
||||
let is_text: Bool = str_eq(btype, "text")
|
||||
let btext: String = if is_text { json_get(block, "text") } else { "" }
|
||||
let text_out = if is_text {
|
||||
text_out + text_join_sep(text_out, btext, saw_nontext) + btext
|
||||
} else { text_out }
|
||||
let saw_nontext = if is_text { false } else { true }
|
||||
// FIX A: record where the facts came from, from whichever block carries them.
|
||||
let src_log = provenance_add_sources(block, btype, has_cit, cit_raw, src_log)
|
||||
// FUTURE-PROOF: tools Anthropic runs on our behalf (web_search today, whatever
|
||||
// ships tomorrow) arrive as server_tool_use blocks, never as client tool_use.
|
||||
// Count the CATEGORY by the block's own name so a new server tool appears in
|
||||
@@ -2678,6 +3086,12 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
} else {
|
||||
if str_eq(tools_log, "") { srv_log } else { tools_log + "," + srv_log }
|
||||
}
|
||||
// FIX A: same merge, for the sources seen in this round's blocks.
|
||||
let sources_all = if str_eq(src_log, "") {
|
||||
sources_all
|
||||
} else {
|
||||
if str_eq(sources_all, "") { src_log } else { sources_all + "; " + src_log }
|
||||
}
|
||||
|
||||
// The assistant turn that requested the tool — needed verbatim on resume so the
|
||||
// tool_use/tool_result pairing stays valid when the client posts its result.
|
||||
@@ -2732,7 +3146,17 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
// CONTINUES the answer, it does not repeat it — overwriting here would throw away
|
||||
// everything the model wrote before the pause, which is the same truncation the
|
||||
// pause handling exists to prevent. A version-fallback round contributes nothing.
|
||||
let final_text = if !is_tool_turn && !can_fallback { final_text + text_out } else { final_text }
|
||||
//
|
||||
// ── FIX C, seam 2 of 2 (2026-08-05) ──────────────────────────────────────────────
|
||||
// The other half of "to.Good". This join is ours (62af564, the web_search port) and
|
||||
// is unconditionally a boundary: the two sides are separate rounds of the Anthropic
|
||||
// loop, separated by a pause and a tool execution. There is no mid-sentence case to
|
||||
// protect here — the model was interrupted, and when it resumes it starts a new
|
||||
// thought. So this seam passes after_interruption=true unconditionally; text_join_sep's
|
||||
// own empty-guards handle the first round and an empty round.
|
||||
let final_text = if !is_tool_turn && !can_fallback {
|
||||
final_text + text_join_sep(final_text, text_out, true) + text_out
|
||||
} else { final_text }
|
||||
// Output cap hit mid-action: the tool block is truncated and will NOT run. Say so
|
||||
// instead of ending on silent almost-work.
|
||||
let final_text = if str_eq(stop_reason, "max_tokens") && has_tool {
|
||||
@@ -2756,6 +3180,7 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
+ ",\"narration\":\"" + json_safe(pend_narration) + "\""
|
||||
+ ",\"model\":\"" + model + "\""
|
||||
+ ",\"agentic\":true"
|
||||
+ ",\"sources\":\"" + json_safe(sources_all) + "\""
|
||||
+ ",\"tools_used\":" + tools_arr + "}"
|
||||
}
|
||||
|
||||
@@ -2763,6 +3188,10 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
// genuine no-response (model returned an empty text block). The iteration cap
|
||||
// means the task was too complex for the agentic loop depth — surface it clearly
|
||||
// so the caller/operator knows to increase the cap or break the task apart.
|
||||
// FIX A follow-up: strip any receipt the MODEL wrote before this becomes the reply. Placed
|
||||
// ABOVE the empty check on purpose — a turn whose entire output was an imitated receipt has
|
||||
// produced no answer, and must be reported as no answer rather than as a receipt.
|
||||
let final_text = receipt_strip(final_text)
|
||||
if str_eq(final_text, "") {
|
||||
let hit_cap: Bool = iteration >= 12
|
||||
let err_msg: String = if hit_cap {
|
||||
@@ -2782,7 +3211,10 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
let done_next: String = if str_eq(done_prev, "") { "{\"done\":true}" } else { done_prev + ",{\"done\":true}" }
|
||||
state_set(done_key, done_next)
|
||||
}
|
||||
return "{\"reply\":\"" + safe_text + "\",\"model\":\"" + model + "\",\"agentic\":true,\"tools_used\":" + tools_arr + ",\"iterations\":" + int_to_str(iteration) + "}"
|
||||
// FIX A: "sources" carries the URLs this turn actually retrieved, so handle_chat_agentic
|
||||
// can write them into the history receipt and the next turn can answer "what source did
|
||||
// you use?" from the transcript instead of guessing (or apologising).
|
||||
return "{\"reply\":\"" + safe_text + "\",\"model\":\"" + model + "\",\"agentic\":true,\"tools_used\":" + tools_arr + ",\"sources\":\"" + json_safe(sources_all) + "\",\"iterations\":" + int_to_str(iteration) + "}"
|
||||
}
|
||||
|
||||
// bridge_save — persist a suspended agentic turn keyed by session_id. Stored as a
|
||||
@@ -2800,12 +3232,31 @@ fn bridge_save(session_id: String, model: String, safe_sys: String, tools_json:
|
||||
// JSON values (not string-escaped) so the round-trip through state_get/json_get_raw
|
||||
// never corrupts nested quotes. Scalar strings (model, safe_sys, tools_log,
|
||||
// tool_use_id) stay as string fields via json_safe as before.
|
||||
//
|
||||
// FIELD ORDER IS LOAD-BEARING (round-9 fix, 2026-08-06). json_get is a first-
|
||||
// substring-match scanner (strstr for "\"key\":", el_runtime.c), and the two raw
|
||||
// fields embed the UNESCAPED conversation — every key the model's own blocks carry
|
||||
// ("tool_use_id" in each web_search_tool_result, "content", "type", ...) is findable
|
||||
// by a whole-blob scan. With messages_raw serialized BEFORE tool_use_id, the resume
|
||||
// read json_get(blob, "tool_use_id") returned the FIRST web_search_tool_result's
|
||||
// srvtoolu_… id instead of the saved client-tool id, so every search-then-bridge
|
||||
// turn 400'd on approval ("unexpected tool_use_id found in tool_result blocks:
|
||||
// srvtoolu_…") and the run died as {"error":"llm unavailable"}. Same first-match-
|
||||
// scanner class as BUG-6 (approve "content" matched inside tool_input) and the
|
||||
// round-8 citation-block fix.
|
||||
//
|
||||
// The rule: every json_safe'd scalar precedes both raw fields (escaping means a
|
||||
// scalar value can never contain a bare "key": byte pattern, so first-match lands
|
||||
// on the blob's own fields), and tools_raw — our own fixed tool schema — precedes
|
||||
// messages_raw — arbitrary model/user content — so neither raw extraction can
|
||||
// first-match into model-controlled bytes either. Do not reorder; do not add a
|
||||
// field after messages_raw.
|
||||
let blob: String = "{\"model\":\"" + json_safe(model) + "\""
|
||||
+ ",\"safe_sys\":\"" + json_safe(safe_sys) + "\""
|
||||
+ ",\"messages_raw\":" + messages
|
||||
+ ",\"tools_raw\":" + tools_json
|
||||
+ ",\"tools_log\":\"" + json_safe(tools_log) + "\""
|
||||
+ ",\"tool_use_id\":\"" + json_safe(tool_use_id) + "\"}"
|
||||
+ ",\"tool_use_id\":\"" + json_safe(tool_use_id) + "\""
|
||||
+ ",\"tools_raw\":" + tools_json
|
||||
+ ",\"messages_raw\":" + messages + "}"
|
||||
state_set("mcp_bridge:" + session_id, blob)
|
||||
return true
|
||||
}
|
||||
@@ -2842,11 +3293,17 @@ fn agentic_resume(session_id: String, tool_use_id: String, content: String) -> S
|
||||
let tools_log: String = json_get(blob, "tools_log")
|
||||
let saved_use_id: String = json_get(blob, "tool_use_id")
|
||||
|
||||
// Bind the result to the tool the soul actually suspended on. The client should
|
||||
// echo the call_id; if it omits or mismatches it, fall back to the saved id so a
|
||||
// late/partial client still resumes correctly.
|
||||
let use_id: String = if str_eq(tool_use_id, "") { saved_use_id } else { tool_use_id }
|
||||
let eff_use_id: String = if str_eq(use_id, saved_use_id) { use_id } else { saved_use_id }
|
||||
// Bind the result to the tool the loop actually suspended on. The client echoes
|
||||
// the call_id from the pending envelope; that value came straight from
|
||||
// pend_tool_id and never round-tripped through this blob, so when both are
|
||||
// present and disagree the CLIENT's id is the one with clean provenance (a blob
|
||||
// written by a pre-round-9 binary misreads tool_use_id by first-match scanning
|
||||
// into messages_raw — see bridge_save). A client that omits call_id still
|
||||
// resumes on the saved id, which the reordered blob now reads correctly.
|
||||
// (The old guard here — "on mismatch, prefer saved" — reduced to eff_use_id ≡
|
||||
// saved_use_id in both branches: the client's correct id could never win, which
|
||||
// is what turned the misread into a deterministic 400 on resume.)
|
||||
let eff_use_id: String = if str_eq(tool_use_id, "") { saved_use_id } else { tool_use_id }
|
||||
|
||||
// Result may be large (an MCP page/file); truncate like local tool results do.
|
||||
let trimmed: String = if str_len(content) > 6000 {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn chat_default_model() -> String
|
||||
extern fn engram_numeric_valid(s: String) -> Bool
|
||||
extern fn parse_float_x100(s: String) -> Int
|
||||
@@ -16,18 +16,35 @@ extern fn engram_nodes_merge(a: String, b: String) -> String
|
||||
extern fn id_in_seen(node_id: String, seen: String) -> Bool
|
||||
extern fn add_to_seen(seen: String, node_id: String) -> String
|
||||
extern fn engram_extract_ids(nodes_json: String) -> String
|
||||
extern fn affective_node_ts(node_json: String) -> Int
|
||||
extern fn engram_compile(intent: String) -> String
|
||||
extern fn distill_transcript(transcript: String) -> String
|
||||
extern fn json_safe(s: String) -> String
|
||||
extern fn current_engine_note(model: String) -> String
|
||||
extern fn bounded_persona_floor() -> String
|
||||
extern fn operator_identity_block() -> String
|
||||
extern fn build_system_prompt(ctx: String, chat_mode: Bool) -> String
|
||||
extern fn hist_append(hist: String, role: String, content: String) -> String
|
||||
extern fn conv_hist_key(session_id: String) -> String
|
||||
extern fn conv_hist_label(session_id: String) -> String
|
||||
extern fn is_utility_request(body: String, session_id: String) -> Bool
|
||||
extern fn provenance_scan_urls(arr: String, acc: String) -> String
|
||||
extern fn provenance_add_sources(block: String, btype: String, has_cit: Bool, cit_raw: String, acc: String) -> String
|
||||
extern fn provenance_names(tools_used: String) -> String
|
||||
extern fn text_join_sep(accumulated: String, incoming: String, after_interruption: Bool) -> String
|
||||
extern fn receipt_rule() -> String
|
||||
extern fn receipt_strip(s: String) -> String
|
||||
extern fn tool_receipt(tools_used: String, sources: String) -> String
|
||||
extern fn hist_trim(hist: String) -> String
|
||||
extern fn hist_trim_with_bell_guard(hist: String) -> String
|
||||
extern fn clean_llm_response(s: String) -> String
|
||||
extern fn conv_history_persist(hist: String) -> Void
|
||||
extern fn conv_history_load() -> String
|
||||
extern fn conv_history_persist(session_id: String, hist: String) -> Void
|
||||
extern fn conv_history_load(session_id: String) -> String
|
||||
extern fn conv_history_record(session_id: String, user_msg: String, assistant_msg: String, receipt: String) -> Void
|
||||
extern fn conv_history_block(session_id: String) -> String
|
||||
extern fn layered_generate(prompt: String, imprint_id: String, session_id: String) -> String
|
||||
extern fn session_preload_bullets(nodes: String, max_bullets: Int, snip_len: Int) -> String
|
||||
extern fn affective_context_prefix() -> String
|
||||
extern fn handle_chat(body: String) -> String
|
||||
extern fn handle_see(body: String) -> String
|
||||
extern fn studio_tools_json() -> String
|
||||
@@ -37,6 +54,8 @@ extern fn llm_wire_format() -> String
|
||||
extern fn json_escape(s: String) -> String
|
||||
extern fn openai_chat_complete(model: String, base_url: String, api_key: String, safe_sys: String, messages_json: String) -> String
|
||||
extern fn agentic_tools_literal() -> String
|
||||
extern fn web_search_tool_json() -> String
|
||||
extern fn strip_client_web_search(tools_inner: String) -> String
|
||||
extern fn agentic_tools_with_web() -> String
|
||||
extern fn connector_tools_json() -> String
|
||||
extern fn agentic_tools_all() -> String
|
||||
@@ -46,6 +65,10 @@ extern fn call_neuron_mcp(tool_name: String, args: String) -> String
|
||||
extern fn agent_workspace_root() -> String
|
||||
extern fn path_within_root(path: String, root: String) -> Bool
|
||||
extern fn resolve_in_root(path: String, root: String) -> String
|
||||
extern fn run_command_is_readonly(cmd: String) -> Bool
|
||||
extern fn cmd_abs_escape_at(cmd: String, root: String, needle: String) -> Bool
|
||||
extern fn run_command_guard(cmd: String, root: String) -> String
|
||||
extern fn classify_tool_risk(tool_name: String, tool_input: String) -> String
|
||||
extern fn dispatch_tool(tool_name: String, tool_input: String) -> String
|
||||
extern fn is_builtin_tool(tool_name: String) -> Bool
|
||||
extern fn next_bridge_id() -> String
|
||||
|
||||
+17
-3
@@ -5,6 +5,15 @@ el_val_t add_punct(el_val_t s, el_val_t intent);
|
||||
el_val_t add_to_seen(el_val_t seen, el_val_t node_id);
|
||||
el_val_t aff_try_slot(el_val_t slot_json, el_val_t aff_7d_ts, el_val_t acc_key);
|
||||
el_val_t affective_context_prefix(void);
|
||||
el_val_t is_utility_request(el_val_t body, el_val_t session_id);
|
||||
el_val_t operator_identity_block(void);
|
||||
el_val_t provenance_add_sources(el_val_t block, el_val_t btype, el_val_t has_cit, el_val_t cit_raw, el_val_t acc);
|
||||
el_val_t provenance_names(el_val_t tools_used);
|
||||
el_val_t provenance_scan_urls(el_val_t arr, el_val_t acc);
|
||||
el_val_t text_join_sep(el_val_t accumulated, el_val_t incoming, el_val_t after_interruption);
|
||||
el_val_t receipt_rule(void);
|
||||
el_val_t receipt_strip(el_val_t s);
|
||||
el_val_t tool_receipt(el_val_t tools_used, el_val_t sources);
|
||||
el_val_t agent_number(el_val_t agent);
|
||||
el_val_t agent_person(el_val_t agent);
|
||||
el_val_t agent_workspace_root(void);
|
||||
@@ -151,8 +160,12 @@ el_val_t cmd_abs_escape_at(el_val_t cmd, el_val_t root, el_val_t needle);
|
||||
el_val_t connectd_get(el_val_t suffix);
|
||||
el_val_t connectd_post(el_val_t suffix, el_val_t body);
|
||||
el_val_t connector_tools_json(void);
|
||||
el_val_t conv_history_load(void);
|
||||
el_val_t conv_history_persist(el_val_t hist);
|
||||
el_val_t conv_hist_key(el_val_t session_id);
|
||||
el_val_t conv_hist_label(el_val_t session_id);
|
||||
el_val_t conv_history_block(el_val_t session_id);
|
||||
el_val_t conv_history_load(el_val_t session_id);
|
||||
el_val_t conv_history_persist(el_val_t session_id, el_val_t hist);
|
||||
el_val_t conv_history_record(el_val_t session_id, el_val_t user_msg, el_val_t assistant_msg, el_val_t receipt);
|
||||
el_val_t cop_article(el_val_t gender, el_val_t number, el_val_t definite);
|
||||
el_val_t cop_bwk_future(el_val_t prefix);
|
||||
el_val_t cop_bwk_perfect(el_val_t prefix);
|
||||
@@ -782,7 +795,8 @@ el_val_t lang_profile_uga(void);
|
||||
el_val_t lang_profile_zh(void);
|
||||
el_val_t lang_profile(el_val_t code, el_val_t word_order, el_val_t morph_type, el_val_t has_case, el_val_t has_gender, el_val_t script_dir, el_val_t agreement, el_val_t null_subject);
|
||||
el_val_t lang_word_order(el_val_t profile);
|
||||
el_val_t layered_cycle(el_val_t raw_input);
|
||||
el_val_t layered_cycle(el_val_t raw_input, el_val_t session_id, el_val_t utility);
|
||||
el_val_t layered_generate(el_val_t prompt, el_val_t imprint_id, el_val_t session_id);
|
||||
el_val_t lex_class(el_val_t entry);
|
||||
el_val_t lex_form(el_val_t entry, el_val_t idx);
|
||||
el_val_t lex_pos(el_val_t entry);
|
||||
|
||||
+134
-10
@@ -430,7 +430,130 @@ fn handle_api_node_update(body: String) -> String {
|
||||
return "{\"id\":\"" + new_id + "\",\"supersedes\":\"" + id + "\",\"ok\":true}"
|
||||
}
|
||||
|
||||
// handle_api_recall — search or activate memory by query.
|
||||
// ── Recall through spreading activation ───────────────────────────────────────
|
||||
//
|
||||
// api_activation_depth — traversal depth for a retrieval query. Honours ?depth= /
|
||||
// body "depth" for callers that want a wider or tighter associative horizon;
|
||||
// defaults to 2, matching every other production activation caller (the
|
||||
// knowledge-search path here, chat.el's per-turn activation) — one hop reaches a
|
||||
// node's direct associations, two reaches its siblings through a shared hub,
|
||||
// which is exactly the sibling-recovery case recall was failing.
|
||||
fn api_activation_depth(path: String, body: String) -> Int {
|
||||
let d: Int = api_query_int(path, "depth", 0)
|
||||
let d = if d == 0 { json_get_int(body, "depth") } else { d }
|
||||
if d <= 0 { return 2 }
|
||||
return d
|
||||
}
|
||||
|
||||
// api_merge_activated_nodes — project an activation result array down to a bare
|
||||
// node array in activation order, then backfill from the lexical seed list until
|
||||
// `limit` nodes are collected. Deduped by node id.
|
||||
//
|
||||
// SHAPE CONTRACT: the return value is a BARE array of full engram node objects —
|
||||
// byte-for-byte the same node JSON engram_search_json emits, so every existing
|
||||
// /recall consumer keeps working unchanged (the MCP wrapper's recall/
|
||||
// searchKnowledge, tools/telegram-gateway.sh which reads `.value.content`,
|
||||
// cli/neuron_mcp.py). Activation strength is a RANKING input here, not a payload
|
||||
// change; the scalars stay available on /api/activate and in compileCtx.
|
||||
fn api_merge_activated_nodes(act_raw: String, lex_raw: String, limit: Int) -> String {
|
||||
let seen: String = ""
|
||||
let out: String = ""
|
||||
let n: Int = 0
|
||||
// Pass 1 — activation-ranked. engram_activate_json already sorts promoted
|
||||
// (working-memory) nodes first by wm_weight desc, then background-only nodes
|
||||
// by background_activation desc, so element order IS the activation ranking.
|
||||
let an: Int = if api_nonempty(act_raw) { json_array_len(act_raw) } else { 0 }
|
||||
let i: Int = 0
|
||||
while i < an && n < limit {
|
||||
let entry: String = json_array_get(act_raw, i)
|
||||
let anode: String = json_get_raw(entry, "node")
|
||||
let aid: String = json_get(anode, "id")
|
||||
let adup: Bool = str_eq(aid, "") || str_contains(seen, "<" + aid + ">")
|
||||
let asep: String = if n == 0 { "" } else { "," }
|
||||
let out = if adup { out } else { out + asep + anode }
|
||||
let seen = if adup { seen } else { seen + "<" + aid + ">" }
|
||||
let n = if adup { n } else { n + 1 }
|
||||
let i = i + 1
|
||||
}
|
||||
// Pass 2 — lexical seed backfill (see the exact-lookup note on
|
||||
// handle_api_recall). Only runs when activation left room under `limit`.
|
||||
let ln: Int = if api_nonempty(lex_raw) { json_array_len(lex_raw) } else { 0 }
|
||||
let j: Int = 0
|
||||
while j < ln && n < limit {
|
||||
let lnode: String = json_array_get(lex_raw, j)
|
||||
let lid: String = json_get(lnode, "id")
|
||||
let ldup: Bool = str_eq(lid, "") || str_contains(seen, "<" + lid + ">")
|
||||
let lsep: String = if n == 0 { "" } else { "," }
|
||||
let out = if ldup { out } else { out + lsep + lnode }
|
||||
let seen = if ldup { seen } else { seen + "<" + lid + ">" }
|
||||
let n = if ldup { n } else { n + 1 }
|
||||
let j = j + 1
|
||||
}
|
||||
return "[" + out + "]"
|
||||
}
|
||||
|
||||
// api_retrieve — THE retrieval path. Spreading activation over the weighted
|
||||
// directed graph, lexical seeds backfilling the tail.
|
||||
//
|
||||
// WAS (until 2026-08-07): `engram_search_json(q, limit)` alone — a case-
|
||||
// insensitive substring matcher scored by how many distinct query tokens appear
|
||||
// in a node's content/label/tags, tie-broken by raw salience. It never read a
|
||||
// single edge. Recall could not see an association: querying an identity value
|
||||
// returned unrelated documents that happened to contain the word, and NOT the
|
||||
// twelve sibling value nodes one hop off the same hub.
|
||||
//
|
||||
// NOW: recall runs the spreading-activation traversal that has been compiled
|
||||
// into the runtime the whole time (engram_activate / engram_activate_json,
|
||||
// el_runtime.c) and ranks by the resulting activation strength. This restores
|
||||
// the designed retrieval mechanism — Engram provisional 64/064,260, claim 1:
|
||||
// "no data is retrieved from the weighted directed graph except through the
|
||||
// spreading activation traversal", with activation strength computed as the
|
||||
// PRODUCT of parent strength, edge weight, target salience, and query/target
|
||||
// cosine similarity, because "the multiplication of all four factors enforces a
|
||||
// conjunctive property... addition would allow many weak associations to
|
||||
// accumulate into false relevance."
|
||||
//
|
||||
// SEEDING — derived from the runtime, not assumed. engram_activate takes the
|
||||
// query TEXT (not seed ids) and seeds internally in two passes: (1) lexical —
|
||||
// every node matching at least one query token seeds, with initial activation
|
||||
// = salience x temporal_decay x dampening x token_coverage, so a node covering
|
||||
// the whole phrase ignites harder than one covering a single word; (2) semantic
|
||||
// supplement — the top-K unreached nodes by cosine against the query embedding.
|
||||
// All four other production call sites (neuron-api.el begin_session/compileCtx,
|
||||
// chat.el:352/1715, awareness.el's curiosity scans) pass query text the same
|
||||
// way, so this follows the established convention exactly. The consequence for
|
||||
// recall is direct: the lexical surface recall used to RETURN is now the SEED
|
||||
// SET of the traversal, and what comes back is what those seeds activate. That
|
||||
// is why multi-word queries stop returning nothing — every token that matches
|
||||
// anything ignites, and the traversal ranks the resulting field.
|
||||
//
|
||||
// EXACT-LOOKUP GUARANTEE (no regression): engram_activate's result collector
|
||||
// drops any reached node whose background_activation x confidence < 0.1 unless
|
||||
// it was promoted to working memory, and it never seeds from InternalStateEvent
|
||||
// nodes. So a rare exact token on a dormant, low-salience node can seed the
|
||||
// traversal and still go unreported. Retrieval therefore appends the lexical
|
||||
// seed list after the activated ranking, deduped by id, until `limit` is filled.
|
||||
// This is a seeded hybrid, not a parallel search bolted alongside activation:
|
||||
// the backfill is the SAME seed set the traversal itself computed, restored to
|
||||
// the tail of the result rather than recomputed by a different mechanism.
|
||||
// Activation always leads the ranking; nothing that used to be findable becomes
|
||||
// unfindable.
|
||||
//
|
||||
// COST/EFFECT NOTE: activation is a stateful read by design — claim 29, "update
|
||||
// the last-activation timestamp and increment the activation count... in
|
||||
// response to any access to that node record during spreading activation
|
||||
// traversal". Promoted nodes get reinforced, working-memory weights are
|
||||
// rewritten, and the query folds into the context centroid. That is the
|
||||
// intended semantics of retrieval-as-activation and is already what every chat
|
||||
// turn does; it does mean recall now participates in shaping working memory.
|
||||
fn api_retrieve(q: String, path: String, body: String, limit: Int) -> String {
|
||||
let depth: Int = api_activation_depth(path, body)
|
||||
let act_raw: String = engram_activate_json(q, depth)
|
||||
let lex_raw: String = engram_search_json(q, limit)
|
||||
return api_or_empty(api_merge_activated_nodes(act_raw, lex_raw, limit))
|
||||
}
|
||||
|
||||
// handle_api_recall — retrieve memory by query, through spreading activation.
|
||||
fn handle_api_recall(method: String, path: String, body: String) -> String {
|
||||
// Accept the query from the URL ?query= / ?q= params, or, when those are
|
||||
// empty (e.g. a POST with a JSON body), from the body fields "query"/"q".
|
||||
@@ -450,8 +573,7 @@ fn handle_api_recall(method: String, path: String, body: String) -> String {
|
||||
if str_eq(eff_q, "") {
|
||||
return api_or_empty(engram_scan_nodes_json(limit, 0))
|
||||
}
|
||||
let results: String = engram_search_json(eff_q, limit)
|
||||
return api_or_empty(results)
|
||||
return api_retrieve(eff_q, path, body, limit)
|
||||
}
|
||||
|
||||
// ── Knowledge ─────────────────────────────────────────────────────────────────
|
||||
@@ -470,13 +592,15 @@ fn handle_api_search_knowledge(method: String, path: String, body: String) -> St
|
||||
let limit = if limit == 0 { json_get_int(body, "limit") } else { limit }
|
||||
let limit = if limit == 0 { 10 } else { limit }
|
||||
if str_eq(q, "") { return api_err("query is required") }
|
||||
let results: String = engram_search_json(q, limit)
|
||||
if str_eq(results, "") { return "[]" }
|
||||
let first: String = str_slice(results, 0, 1)
|
||||
if !str_eq(first, "[") && !str_eq(first, "{") {
|
||||
return api_or_empty(engram_activate_json(q, 2))
|
||||
}
|
||||
return results
|
||||
// Same retrieval path as recall — and it is the SAME change, not a copy of
|
||||
// one. The "activate fallback" this replaced was unreachable dead code: it
|
||||
// only fired when engram_search_json's return did not start with '[' or '{',
|
||||
// and engram_search_json always emits a '['-prefixed array (el_runtime.c
|
||||
// jb_putc('[') before any hit test), so the guard was false on every call
|
||||
// including the zero-hit "[]" case. Knowledge search therefore had exactly
|
||||
// the substring-matcher behavior recall had, with a comment claiming
|
||||
// otherwise. Routing it through api_retrieve makes the claim true.
|
||||
return api_retrieve(q, path, body, limit)
|
||||
}
|
||||
|
||||
// handle_api_browse_knowledge — list Knowledge nodes.
|
||||
|
||||
@@ -280,7 +280,10 @@ fn handle_dharma_recv(body: String) -> String {
|
||||
// Non-agentic ("Tools: Off"): the full L1→L2→L3→L1 cycle, which now generates
|
||||
// at L3 instead of echoing. Envelope built outside the cycle — see
|
||||
// plain_chat_envelope.
|
||||
let screened_reply: String = layered_cycle(raw_msg)
|
||||
// FIX B/E1 (2026-08-05): the cycle is told which conversation it is in, and
|
||||
// whether this generation is conversation at all. Same two arguments at all
|
||||
// three dispatch sites.
|
||||
let screened_reply: String = layered_cycle(raw_msg, json_get(chat_body, "session_id"), is_utility_request(chat_body, json_get(chat_body, "session_id")))
|
||||
plain_chat_envelope(screened_reply, chat_default_model())
|
||||
}
|
||||
auto_persist(chat_body, reply)
|
||||
@@ -454,7 +457,9 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
handle_chat_agentic(body)
|
||||
} else {
|
||||
// Non-agentic ("Tools: Off") — same cycle and same envelope as POST.
|
||||
let screened_reply: String = layered_cycle(eff_msg)
|
||||
// FIX B/E1: same threading. A GET probe usually carries no session_id, which
|
||||
// resolves to the anonymous window — the documented behaviour for this door.
|
||||
let screened_reply: String = layered_cycle(eff_msg, json_get(body, "session_id"), is_utility_request(body, json_get(body, "session_id")))
|
||||
plain_chat_envelope(screened_reply, chat_default_model())
|
||||
}
|
||||
auto_persist(body, reply)
|
||||
@@ -621,7 +626,9 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
// Non-agentic ("Tools: Off") — the app's DEFAULT mode (AgentMode.NEVER).
|
||||
// Full L1→L2→L3→L1 cycle with real generation at L3; envelope built
|
||||
// outside the cycle so safety_validate always sees raw text.
|
||||
let screened_reply: String = layered_cycle(raw_msg)
|
||||
// FIX B/E1: same threading. This is the app's main plain-chat door, so this
|
||||
// is the site that ends the blank stare in practice.
|
||||
let screened_reply: String = layered_cycle(raw_msg, json_get(body, "session_id"), is_utility_request(body, json_get(body, "session_id")))
|
||||
plain_chat_envelope(screened_reply, chat_default_model())
|
||||
}
|
||||
auto_persist(body, reply)
|
||||
|
||||
Executable
+108
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env bash
|
||||
# run-el-test.sh — compile and run one El test program from tests/.
|
||||
#
|
||||
# WHY THIS EXISTS (2026-08-07, issue #129):
|
||||
# tests/ has held 14 test programs for months with no way to run them. CI does
|
||||
# not run them. The convention printed in their own headers
|
||||
# (`elc soul.el && ./soul --test tests/x.el`) refers to a --test flag the El
|
||||
# runtime does not implement. So the tests were documentation, not gates —
|
||||
# which is how a P0 safety regression shipped with a test directory present.
|
||||
#
|
||||
# THE RECIPE, AND WHY IT IS THIS SHAPE:
|
||||
# Same discovery as gen-soul-amalgam.sh — `elc --target=c` emits only an extern
|
||||
# prototype for any module that has a .elh header next to it, and inlines the
|
||||
# module's bodies when it does not. A test that imports ../chat.el therefore
|
||||
# compiles to a 18 KB unit full of unresolved externs unless the headers are
|
||||
# out of the way. So: copy the sources into a scratch tree, delete every .elh
|
||||
# on the import chain, and compile the test there.
|
||||
#
|
||||
# Scratch copy on purpose: the worktree is shared with other terminals and
|
||||
# deleting headers in place would be a shared-tree mutation with no owner.
|
||||
#
|
||||
# EXIT STATUS IS THE GATE: non-zero if the binary fails to build, crashes, or if
|
||||
# its output contains a FAIL line or reports a non-zero failed count. Do not
|
||||
# "improve" this into something that only checks the exit code of the test
|
||||
# binary — these El tests print failures and still exit 0.
|
||||
#
|
||||
# usage: scripts/run-el-test.sh tests/test_history_amplification.el
|
||||
set -euo pipefail
|
||||
|
||||
TEST_REL="${1:?usage: run-el-test.sh tests/<test>.el}"
|
||||
SRC="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
TEST_NAME="$(basename "$TEST_REL" .el)"
|
||||
|
||||
ELC="${ELC:-$HOME/neuron-dev-stack/src/el/lang/dist/platform/elc}"
|
||||
[ -x "$ELC" ] || ELC="$HOME/el-sdk/elc"
|
||||
[ -x "$ELC" ] || { echo "[run-el-test] FAIL: no elc found (set ELC=)"; exit 1; }
|
||||
|
||||
RTC="${RTC:-$SRC/vendor/el-runtime/v1.0.0-20260501/el_runtime.c}"
|
||||
[ -f "$RTC" ] || RTC="$HOME/el-sdk/el_runtime.c"
|
||||
[ -f "$RTC" ] || { echo "[run-el-test] FAIL: no el_runtime.c found (set RTC=)"; exit 1; }
|
||||
RTDIR="$(dirname "$RTC")"
|
||||
|
||||
EL_REPO="${EL_REPO:-$HOME/Development/neuron-technologies/el}"
|
||||
SSL="${SSL_PREFIX:-/opt/homebrew/opt/openssl@3}"
|
||||
|
||||
GEN="$(mktemp -d "${TMPDIR:-/tmp}/el-test.XXXXXX")"
|
||||
trap 'rm -rf "$GEN"' EXIT
|
||||
|
||||
mkdir -p "$GEN/neuron/tests" "$GEN/foundation/el/elp/src"
|
||||
cp "$SRC"/*.el "$GEN/neuron/"
|
||||
cp "$SRC"/tests/*.el "$GEN/neuron/tests/" 2>/dev/null || true
|
||||
[ -d "$EL_REPO/elp/src" ] && cp "$EL_REPO"/elp/src/*.el "$GEN/foundation/el/elp/src/" 2>/dev/null || true
|
||||
# The whole recipe depends on there being no headers to short-circuit inlining.
|
||||
find "$GEN" -name '*.elh' -delete
|
||||
|
||||
echo "[run-el-test] compiling $TEST_REL"
|
||||
( cd "$GEN/neuron" && "$ELC" --target=c "tests/${TEST_NAME}.el" ) > "$GEN/${TEST_NAME}.c"
|
||||
|
||||
BODIES=$(grep -c '^el_val_t .*) {$' "$GEN/${TEST_NAME}.c" || true)
|
||||
echo "[run-el-test] $(wc -c < "$GEN/${TEST_NAME}.c" | tr -d ' ') bytes, ${BODIES} inlined function bodies"
|
||||
# A test that imports ../chat.el pulls in the bulk of the engine. A tiny body
|
||||
# count means an import was read from a header instead of inlined, and the test
|
||||
# would be exercising extern stubs rather than the real code.
|
||||
if [ "$BODIES" -lt 100 ]; then
|
||||
echo "[run-el-test] FAIL: only $BODIES inlined bodies — an import was not inlined"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
cc -O2 -DHAVE_CURL \
|
||||
-I"$RTDIR" -I"$SSL/include" -L"$SSL/lib" \
|
||||
"$GEN/${TEST_NAME}.c" "$RTC" \
|
||||
-lssl -lcrypto -lcurl -lpthread -lm \
|
||||
-o "$GEN/${TEST_NAME}" 2> "$GEN/cc.log" || {
|
||||
echo "[run-el-test] FAIL: compile error"; tail -30 "$GEN/cc.log"; exit 1; }
|
||||
|
||||
# arm64 pointer-truncation guard (cc-brain.sh's rule): an implicit declaration of
|
||||
# a runtime symbol truncates its returned pointer to 32 bits.
|
||||
if grep -E 'implicit.*(engram_|el_)' "$GEN/cc.log"; then
|
||||
echo "[run-el-test] FAIL: implicit declarations of runtime symbols"; exit 1; fi
|
||||
|
||||
# Throwaway HOME so a test can never read or write the live engram at ~/.neuron.
|
||||
TEST_HOME="$GEN/home"
|
||||
mkdir -p "$TEST_HOME"
|
||||
|
||||
echo "[run-el-test] running $TEST_NAME"
|
||||
set +e
|
||||
HOME="$TEST_HOME" NEURON_HOME="$TEST_HOME/.neuron" "$GEN/${TEST_NAME}" 2>&1 | tee "$GEN/out.txt"
|
||||
RC=${PIPESTATUS[0]}
|
||||
set -e
|
||||
|
||||
if [ "$RC" -ne 0 ]; then
|
||||
echo "[run-el-test] FAIL: $TEST_NAME exited $RC (crash or abort)"
|
||||
exit 1
|
||||
fi
|
||||
if grep -q " FAIL:" "$GEN/out.txt"; then
|
||||
echo "[run-el-test] FAIL: $TEST_NAME reported failing assertions"
|
||||
exit 1
|
||||
fi
|
||||
if grep -qE '[1-9][0-9]* failed' "$GEN/out.txt"; then
|
||||
echo "[run-el-test] FAIL: $TEST_NAME reported a non-zero failed count"
|
||||
exit 1
|
||||
fi
|
||||
if ! grep -q "PASS:" "$GEN/out.txt"; then
|
||||
echo "[run-el-test] FAIL: $TEST_NAME produced no assertions at all"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "[run-el-test] PASS: $TEST_NAME"
|
||||
@@ -379,9 +379,23 @@ fn emit_session_start_event() -> Void {
|
||||
// layered_cycle — routes user-facing requests through the 4-layer consciousness stack.
|
||||
// L0 (core) → L1 (safety screen) → L2a (continuity + behavioral profiling) → L2b (mission alignment) → L3 (imprint) → L1 (safety validate)
|
||||
// Internal cognition (heartbeat, proactive, memory ops) bypasses layers — use one_cycle directly.
|
||||
fn layered_cycle(raw_input: String) -> String {
|
||||
let history: String = state_get("conv_history")
|
||||
let session_id: String = state_get("current_session_id")
|
||||
//
|
||||
// FIX B (2026-08-05) — the cycle now knows which conversation it is in.
|
||||
//
|
||||
// session_id: the caller's session, threaded from the route. Was previously read from the
|
||||
// state key "current_session_id", which is read HERE and written NOWHERE in the entire
|
||||
// source — verified across every .el file. So this value was unconditionally "", and every
|
||||
// downstream consumer of it silently fell back to a process-global bucket: conversation
|
||||
// history, and the steward's continuity tracking (TODO reliability #4, below, describes the
|
||||
// cross-session bleed this caused; threading the real id closes it). The plain path's blank
|
||||
// stare and the agentic path's scoped history were the same defect seen from two sides.
|
||||
//
|
||||
// utility: true when the generation is not part of the user's conversation — the app's
|
||||
// title and insight passes. Answered normally, never recorded. See is_utility_request.
|
||||
fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String {
|
||||
// Safety-screen history amplification now reads the SAME window the turn will be
|
||||
// recorded into, so a session's own escalation pattern is what gets scored.
|
||||
let history: String = state_get(conv_hist_key(session_id))
|
||||
|
||||
// L1 in: safety screen
|
||||
let screen_result: String = safety_screen(raw_input, history)
|
||||
@@ -423,8 +437,10 @@ fn layered_cycle(raw_input: String) -> String {
|
||||
let cont_action: String = json_get(continuity, "action")
|
||||
|
||||
// Store continuity status so imprint can adjust its response register.
|
||||
// TODO(reliability #4): session_continuity is process-global; scope per session_id
|
||||
// when available to prevent cross-session bleed under concurrent layered_cycle calls.
|
||||
// TODO(reliability #4) CLOSED 2026-08-05: this line was already written to scope per
|
||||
// session — it just never received a session id, because the only source was a state key
|
||||
// nothing wrote. It is now threaded from the route, so named sessions genuinely get their
|
||||
// own continuity state and only anonymous callers share the global one.
|
||||
let cont_key: String = if str_eq(session_id, "") { "session_continuity" } else { "session_continuity:" + session_id }
|
||||
state_set(cont_key, cont_status)
|
||||
|
||||
@@ -499,7 +515,7 @@ fn layered_cycle(raw_input: String) -> String {
|
||||
// screen, the safe-mode guard, the hard-bell short-circuit and the L2 stewardship layers,
|
||||
// and strictly BEFORE the L1 output gate. A hard bell never reaches a model — the branch
|
||||
// above returns first. Tools are not offered on this turn; see layered_generate.
|
||||
let output: String = layered_generate(prompt, imprint_id)
|
||||
let output: String = layered_generate(prompt, imprint_id, session_id)
|
||||
|
||||
// L1 out: validate output before delivery. Still the terminal gate — nothing below this
|
||||
// line can change the string this function returns.
|
||||
@@ -509,7 +525,19 @@ fn layered_cycle(raw_input: String) -> String {
|
||||
// reachable on the non-bell path: both bell branches above return before this point, so
|
||||
// bell turns still never enter conversation history. Pure state side effect — it cannot
|
||||
// alter what is returned.
|
||||
conv_history_record(raw_input, validated)
|
||||
//
|
||||
// FIX A: the receipt is unconditional and always negative on this path, because on this
|
||||
// path it is structurally true — layered_generate offers no tools at all (build_system_prompt
|
||||
// chat mode + a request body with no "tools" key). Recording "no tools ran" is not padding:
|
||||
// it is the only thing that distinguishes "nothing ran" from "we forgot to write down what
|
||||
// ran", and that ambiguity is what made the model confess to a search it had performed.
|
||||
//
|
||||
// FIX E1: a utility generation is answered but not recorded. Guarded here rather than at
|
||||
// the route so every /api/chat dispatch site inherits it from one place.
|
||||
let receipt: String = tool_receipt("", "")
|
||||
if !utility {
|
||||
conv_history_record(session_id, raw_input, validated, receipt)
|
||||
}
|
||||
return validated
|
||||
}
|
||||
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn init_soul_edges() -> Void
|
||||
extern fn ensure_self_canonical_bridge() -> Void
|
||||
extern fn aff_try_slot(slot_json: String, aff_7d_ts: Int, acc_key: String) -> Void
|
||||
extern fn load_identity_context() -> Void
|
||||
extern fn seed_persona_from_env() -> Void
|
||||
extern fn emit_session_start_event() -> Void
|
||||
extern fn layered_cycle(raw_input: String) -> String
|
||||
extern fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
// ── test_history_amplification.el ─────────────────────────────────────────────
|
||||
//
|
||||
// REGRESSION TEST FOR ISSUE #129 (P0, SAFETY).
|
||||
//
|
||||
// What this guards: on the agentic path, the crisis score has two halves — the
|
||||
// message you just sent, and the distress that has accumulated across the
|
||||
// conversation. The second half is the whole reason the escalation logic exists:
|
||||
// someone whose distress builds over several turns never sends one message that
|
||||
// trips the bell on its own.
|
||||
//
|
||||
// The defect this test was written against (ff421d3, 2026-08-05 → fixed
|
||||
// 2026-08-07): conversation history moved to a per-session key via
|
||||
// conv_hist_key(session_id), but the agentic path's safety screen was left
|
||||
// reading the old anonymous "conv_history" bucket. The desktop app always sends
|
||||
// a session_id, so the screen received "" on every real conversation and the
|
||||
// escalation half always scored 0. Nothing failed. Nothing logged. The comment
|
||||
// above the defective line documented this same bug being fixed once before.
|
||||
//
|
||||
// THE INVARIANT UNDER TEST, stated so it survives future renames:
|
||||
// the window the safety screen READS must be the window conv_history_record
|
||||
// WRITES. Not "must be called conv_history" — must AGREE.
|
||||
//
|
||||
// This test is deliberately written to fail loudly on the pre-fix source. If it
|
||||
// ever passes on code where the screen reads a key nothing writes, it is broken.
|
||||
//
|
||||
// To run (macOS, from the worktree root):
|
||||
// scripts/run-el-test.sh tests/test_history_amplification.el
|
||||
// ──────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
import "../chat.el"
|
||||
import "../safety.el"
|
||||
import "../sessions.el"
|
||||
|
||||
// Program class. Without this an El program compiles as a 'utility', and a
|
||||
// utility may not call the self-formation primitives (llm_call_system,
|
||||
// llm_vision) that chat.el's agentic loop references — the unit fails to
|
||||
// compile with a capability violation even though the test never calls them.
|
||||
// Declaring 'cgi' matches how soul.el declares itself.
|
||||
//
|
||||
// The endpoints below are deliberately DEAD: this test must never reach a live
|
||||
// engram, and nothing it asserts depends on one. Port 9 is discard.
|
||||
cgi "neuron-test-history-amplification" {
|
||||
dharma_id: "ntn-test@http://127.0.0.1:9",
|
||||
principal: "test-harness",
|
||||
network: "dharma-testnet",
|
||||
engram: "http://127.0.0.1:9"
|
||||
}
|
||||
|
||||
// ── Counters ──────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// NOTE for anyone copying this harness: the idiom used by the older tests in
|
||||
// this directory — `let pass_count = pass_count + 1` inside an assert function —
|
||||
// does NOT mutate the module-level binding. It declares a new local that dies
|
||||
// with the call, so those suites all print "0 passed, 0 failed" no matter what
|
||||
// happened. Counters go through the state store here so the summary is real.
|
||||
|
||||
fn bump(counter: String) -> Void {
|
||||
let cur: String = state_get(counter)
|
||||
let n: Int = if str_eq(cur, "") { 0 } else { str_to_int(cur) }
|
||||
state_set(counter, int_to_str(n + 1))
|
||||
}
|
||||
|
||||
fn counter_of(counter: String) -> Int {
|
||||
let cur: String = state_get(counter)
|
||||
if str_eq(cur, "") { return 0 }
|
||||
return str_to_int(cur)
|
||||
}
|
||||
|
||||
fn assert_eq(label: String, got: String, expected: String) -> Void {
|
||||
if str_eq(got, expected) {
|
||||
bump("t_pass")
|
||||
println(" PASS: " + label)
|
||||
} else {
|
||||
bump("t_fail")
|
||||
println(" FAIL: " + label)
|
||||
println(" got: " + got)
|
||||
println(" expected: " + expected)
|
||||
}
|
||||
}
|
||||
|
||||
fn assert_true(label: String, cond: Bool) -> Void {
|
||||
if cond {
|
||||
bump("t_pass")
|
||||
println(" PASS: " + label)
|
||||
} else {
|
||||
bump("t_fail")
|
||||
println(" FAIL: " + label)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Fixtures ──────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// ESCALATION_MSG scores 50 on its own ("can't go on", safety_score_crisis s9).
|
||||
// 50 is >= soft (35) and < hard (70): a soft bell by itself, which is the point.
|
||||
// It is the message that is only a crisis IN CONTEXT.
|
||||
//
|
||||
// DISTRESS_TURN carries five history phrases:
|
||||
// hopeless 15 + worthless 15 + nobody cares 15 + can't take it anymore 20
|
||||
// + want to disappear 20 = 85; contributed at 1/3 weight = 28.
|
||||
// 50 + 28 = 78 >= 70 → hard_bell. That 28-point swing IS the escalation logic,
|
||||
// and it is exactly what the defect suppressed.
|
||||
|
||||
let TEST_SESSION: String = "sess-test-129-history-amplification"
|
||||
let ESCALATION_MSG: String = "honestly i can't go on like this"
|
||||
let DISTRESS_TURN: String = "i feel hopeless and worthless, nobody cares, i can't take it anymore, i want to disappear"
|
||||
|
||||
println("")
|
||||
println("test_history_amplification.el — issue #129 (agentic multi-turn crisis escalation)")
|
||||
|
||||
// ── 1. Baseline: the message alone is a SOFT bell, not a hard one ─────────────
|
||||
//
|
||||
// If this ever returns hard_bell, the test below proves nothing — the message
|
||||
// would trip the bell without any history and the amplification would be
|
||||
// invisible. This assertion is what keeps the real test honest.
|
||||
|
||||
println("")
|
||||
println("1. baseline — escalation message with NO history is a soft bell")
|
||||
|
||||
let baseline: String = safety_screen(ESCALATION_MSG, "")
|
||||
assert_eq("no history -> soft_bell (not hard)", json_get(baseline, "action"), "soft_bell")
|
||||
|
||||
// ── 2. Producer sanity: history lands in the session's own window ─────────────
|
||||
|
||||
println("")
|
||||
println("2. producer — conv_history_record writes the session's window")
|
||||
|
||||
conv_history_record(TEST_SESSION, DISTRESS_TURN, "i hear you, that sounds heavy", "")
|
||||
|
||||
let written: String = state_get(conv_hist_key(TEST_SESSION))
|
||||
assert_true("session window is non-empty after record", !str_eq(written, ""))
|
||||
assert_true("session window contains the distress turn", str_contains(written, "hopeless"))
|
||||
|
||||
// ── 3. THE REGRESSION: the agentic screen must SEE that window ────────────────
|
||||
//
|
||||
// Pre-fix this returns soft_bell, because agentic_safety_screen read the
|
||||
// anonymous bucket and got "". Post-fix it returns hard_bell.
|
||||
|
||||
println("")
|
||||
println("3. REGRESSION #129 — agentic screen reads the session's own window")
|
||||
|
||||
let screened: String = agentic_safety_screen(TEST_SESSION, ESCALATION_MSG)
|
||||
assert_eq(
|
||||
"distress history escalates the agentic screen to hard_bell",
|
||||
json_get(screened, "action"),
|
||||
"hard_bell"
|
||||
)
|
||||
|
||||
// ── 4. The invariant, stated directly ─────────────────────────────────────────
|
||||
//
|
||||
// Independent of thresholds and phrase lists: whatever the screen reads for a
|
||||
// session must equal what the recorder wrote for that session. This is the
|
||||
// assertion that survives a future rename of either side.
|
||||
|
||||
println("")
|
||||
println("4. invariant — read window == written window")
|
||||
|
||||
let read_back: String = state_get(conv_hist_key(TEST_SESSION))
|
||||
assert_true("screen input is the recorded window, not empty", !str_eq(read_back, ""))
|
||||
assert_eq("read window is byte-identical to written window", read_back, written)
|
||||
|
||||
// ── 5. No false positive: a calm session does not escalate ────────────────────
|
||||
//
|
||||
// A test that only ever asserts "hard_bell" would pass on code that hard-bells
|
||||
// every message. This is the other leg, and it runs BEFORE the anonymous case
|
||||
// below on purpose: that case writes the shared bucket, and under the defect a
|
||||
// calm session would then inherit it.
|
||||
|
||||
println("")
|
||||
println("5. specificity — a calm history does NOT escalate")
|
||||
|
||||
let CALM_SESSION: String = "sess-test-129-calm"
|
||||
state_set("conv_history", "")
|
||||
conv_history_record(CALM_SESSION, "what is the weather like today", "clear and mild", "")
|
||||
let calm: String = agentic_safety_screen(CALM_SESSION, ESCALATION_MSG)
|
||||
assert_eq("calm history stays at soft_bell", json_get(calm, "action"), "soft_bell")
|
||||
|
||||
// ── 6. Cross-session leakage ──────────────────────────────────────────────────
|
||||
//
|
||||
// The same defect had a second face: because the screen read one shared bucket,
|
||||
// a calm session could be scored against a DIFFERENT session's distress. That is
|
||||
// wrong in both directions — it fabricates a crisis for the calm user and it
|
||||
// leaks the distressed user's content into another session's scoring.
|
||||
|
||||
println("")
|
||||
println("6. isolation — one session's distress must not score another session")
|
||||
|
||||
state_set("conv_history", "")
|
||||
let OTHER_SESSION: String = "sess-test-129-other"
|
||||
conv_history_record(OTHER_SESSION, DISTRESS_TURN, "i hear you", "")
|
||||
let isolated: String = agentic_safety_screen(CALM_SESSION, ESCALATION_MSG)
|
||||
assert_eq(
|
||||
"a distressed OTHER session does not escalate the calm session",
|
||||
json_get(isolated, "action"),
|
||||
"soft_bell"
|
||||
)
|
||||
|
||||
// ── 7. Anonymous sessions still work ──────────────────────────────────────────
|
||||
//
|
||||
// conv_hist_key("") deliberately falls back to the shared "conv_history" bucket.
|
||||
// The fix must not break the no-session_id path older callers rely on. Runs last
|
||||
// because it writes that shared bucket.
|
||||
|
||||
println("")
|
||||
println("7. anonymous path — empty session_id still screens against the shared window")
|
||||
|
||||
state_set("conv_history", "[{\"role\":\"user\",\"content\":\"" + DISTRESS_TURN + "\"}]")
|
||||
let anon: String = agentic_safety_screen("", ESCALATION_MSG)
|
||||
assert_eq("anonymous session escalates too", json_get(anon, "action"), "hard_bell")
|
||||
|
||||
// ── Summary ───────────────────────────────────────────────────────────────────
|
||||
|
||||
println("")
|
||||
println("history amplification tests: " + int_to_str(counter_of("t_pass")) + " passed, " + int_to_str(counter_of("t_fail")) + " failed")
|
||||
+97
-11
@@ -41,6 +41,7 @@
|
||||
#include <fcntl.h>
|
||||
#include <dirent.h>
|
||||
#include <errno.h>
|
||||
#include <signal.h> /* SIGPIPE disposition — see el_runtime_ignore_sigpipe */
|
||||
#include <pthread.h>
|
||||
#include <curl/curl.h>
|
||||
|
||||
@@ -1238,16 +1239,77 @@ static const char* http_reason_phrase(int status) {
|
||||
}
|
||||
}
|
||||
|
||||
/* Best-effort send with retry on partial writes. */
|
||||
/* ── A departing client MUST NOT be able to kill the daemon ──────────────────
|
||||
* (2026-08-06, round 9.1 / ADR 0006 item 4.)
|
||||
*
|
||||
* Measured field failure: a client cancelled its request at 25 s; the handler
|
||||
* finished its work at 116.9 s and wrote the reply into the departed client's
|
||||
* socket. The second send() on a reset connection raised SIGPIPE, whose DEFAULT
|
||||
* disposition terminates the process — `exited due to SIGPIPE ... ran for
|
||||
* 361177ms`. launchd respawned 4 ms later, so EVERY other in-flight request on
|
||||
* that daemon lost its work, silently.
|
||||
*
|
||||
* Two independent guards, because one of them can be undone from outside this
|
||||
* file (an embedder may reset signal dispositions) and the other cannot:
|
||||
* 1. process-wide SIGPIPE -> SIG_IGN, installed at runtime init;
|
||||
* 2. per-send suppression at the syscall (MSG_NOSIGNAL where the platform has
|
||||
* it, SO_NOSIGPIPE on the accepted socket on macOS/BSD).
|
||||
* With either in force, send() reports the peer's departure as EPIPE and the
|
||||
* caller decides — which is the point: this is an ordinary I/O outcome, not a
|
||||
* fatal condition.
|
||||
*
|
||||
* It deliberately does NOT swallow the error. http_send_response() below
|
||||
* classifies the errno and logs: "client left" for a departure, and a real
|
||||
* "send failed: <strerror>" for anything else, so a genuine write fault is
|
||||
* still visible in the log (spec round-9.1 §5.3). */
|
||||
|
||||
#ifndef MSG_NOSIGNAL
|
||||
#define MSG_NOSIGNAL 0
|
||||
#endif
|
||||
|
||||
void el_runtime_ignore_sigpipe(void) {
|
||||
static int done = 0;
|
||||
if (done) return;
|
||||
done = 1;
|
||||
struct sigaction sa;
|
||||
memset(&sa, 0, sizeof(sa));
|
||||
sa.sa_handler = SIG_IGN;
|
||||
sigemptyset(&sa.sa_mask);
|
||||
sigaction(SIGPIPE, &sa, NULL);
|
||||
}
|
||||
|
||||
/* Suppress SIGPIPE for one accepted connection (macOS/BSD have no
|
||||
* MSG_NOSIGNAL; they have the socket option instead). Best effort. */
|
||||
static void http_socket_nosigpipe(int fd) {
|
||||
#ifdef SO_NOSIGPIPE
|
||||
int on = 1;
|
||||
setsockopt(fd, SOL_SOCKET, SO_NOSIGPIPE, &on, sizeof(on));
|
||||
#else
|
||||
(void)fd;
|
||||
#endif
|
||||
}
|
||||
|
||||
/* Best-effort send with retry on partial writes.
|
||||
* Returns 0 on success, -1 on failure with errno preserved for the caller. */
|
||||
static int http_send_all(int fd, const char* p, size_t left) {
|
||||
while (left > 0) {
|
||||
ssize_t w = send(fd, p, left, 0);
|
||||
if (w <= 0) return -1;
|
||||
ssize_t w = send(fd, p, left, MSG_NOSIGNAL);
|
||||
if (w < 0) {
|
||||
if (errno == EINTR) continue; /* not an error — retry */
|
||||
return -1; /* errno stays set for caller */
|
||||
}
|
||||
if (w == 0) { errno = EPIPE; return -1; }
|
||||
p += w; left -= (size_t)w;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Did this write fail because the client is gone, or because something is
|
||||
* actually wrong with the socket? Only the first is routine. */
|
||||
static int http_write_err_is_client_gone(int e) {
|
||||
return e == EPIPE || e == ECONNRESET || e == ENOTCONN || e == ESHUTDOWN;
|
||||
}
|
||||
|
||||
/* Discriminator that http_response() embeds at the start of its envelope.
|
||||
* A handler returning a string starting with this exact prefix is treated
|
||||
* as a structured response; anything else is treated as a raw body. */
|
||||
@@ -1468,14 +1530,30 @@ static void http_send_response(int fd, const char* body) {
|
||||
free(env_body); free(hdrs.buf); return;
|
||||
}
|
||||
|
||||
if (http_send_all(fd, status_line, (size_t)sl) == 0
|
||||
&& http_send_all(fd, hdrs.buf, hdrs.len) == 0
|
||||
&& http_send_all(fd, tail, (size_t)tl) == 0
|
||||
&& (head_only
|
||||
/* HEAD requests echo headers + Content-Length but no body. */
|
||||
? 1
|
||||
: http_send_all(fd, eff_body, blen) == 0)) {
|
||||
/* sent successfully */
|
||||
/* The reply is written in four pieces; any of them can find the client
|
||||
* already gone. errno is captured at the first failure, before any later
|
||||
* library call can clobber it, and classified once below. */
|
||||
errno = 0;
|
||||
int send_err = 0;
|
||||
if (http_send_all(fd, status_line, (size_t)sl) != 0) send_err = errno;
|
||||
else if (http_send_all(fd, hdrs.buf, hdrs.len) != 0) send_err = errno;
|
||||
else if (http_send_all(fd, tail, (size_t)tl) != 0) send_err = errno;
|
||||
else if (!head_only /* HEAD echoes headers + Content-Length, no body. */
|
||||
&& http_send_all(fd, eff_body, blen) != 0) send_err = errno;
|
||||
|
||||
if (send_err) {
|
||||
if (http_write_err_is_client_gone(send_err)) {
|
||||
/* ROUTINE. The user closed the window, quit the app, or cancelled.
|
||||
* The work is done and the daemon keeps serving everyone else. */
|
||||
fprintf(stderr, "[http] client left before the reply was written "
|
||||
"(%zu-byte body, %s) - request completed, reply discarded\n",
|
||||
blen, strerror(send_err));
|
||||
} else {
|
||||
/* NOT routine — a real write fault. Never let the client-gone case
|
||||
* above hide this one. */
|
||||
fprintf(stderr, "[http] send failed: %s (%zu-byte body)\n",
|
||||
strerror(send_err), blen);
|
||||
}
|
||||
}
|
||||
|
||||
if (env_parsed_root) el_release(env_parsed_root);
|
||||
@@ -1491,6 +1569,7 @@ static void* http_worker(void* arg) {
|
||||
HttpWorkerArg* a = (HttpWorkerArg*)arg;
|
||||
int fd = a->fd;
|
||||
free(a);
|
||||
http_socket_nosigpipe(fd);
|
||||
char *method = NULL, *path = NULL, *body = NULL;
|
||||
if (http_read_request(fd, &method, &path, &body, NULL) == 0) {
|
||||
http_handler_fn h = http_lookup_active();
|
||||
@@ -1531,6 +1610,7 @@ static void* http_worker(void* arg) {
|
||||
}
|
||||
|
||||
void http_serve(el_val_t port, el_val_t handler) {
|
||||
el_runtime_ignore_sigpipe(); /* serving implies clients that leave */
|
||||
/* If `handler` looks like a string name, register it as the active handler. */
|
||||
const char* hname = EL_CSTR(handler);
|
||||
if (hname && looks_like_string(handler)) {
|
||||
@@ -1634,6 +1714,7 @@ static void* _http_serve_async_loop(void* raw) {
|
||||
}
|
||||
|
||||
void http_serve_async(el_val_t port, el_val_t handler) {
|
||||
el_runtime_ignore_sigpipe(); /* serving implies clients that leave */
|
||||
const char* hname = EL_CSTR(handler);
|
||||
if (hname && looks_like_string(handler)) {
|
||||
http_set_handler(handler);
|
||||
@@ -1821,6 +1902,7 @@ static void* http_worker_v2(void* arg) {
|
||||
HttpWorkerArg* a = (HttpWorkerArg*)arg;
|
||||
int fd = a->fd;
|
||||
free(a);
|
||||
http_socket_nosigpipe(fd);
|
||||
char *method = NULL, *path = NULL, *body = NULL, *hdr_block = NULL;
|
||||
if (http_read_request(fd, &method, &path, &body, &hdr_block) == 0) {
|
||||
http_handler4_fn h = http_lookup_active_v2();
|
||||
@@ -1858,6 +1940,7 @@ static void* http_worker_v2(void* arg) {
|
||||
}
|
||||
|
||||
void http_serve_v2(el_val_t port, el_val_t handler) {
|
||||
el_runtime_ignore_sigpipe(); /* serving implies clients that leave */
|
||||
const char* hname = EL_CSTR(handler);
|
||||
if (hname && looks_like_string(handler)) {
|
||||
http_set_handler_v2(handler);
|
||||
@@ -5511,6 +5594,9 @@ el_val_t getpid_now(void) {
|
||||
static el_val_t _el_args_list = 0;
|
||||
|
||||
void el_runtime_init_args(int argc, char** argv) {
|
||||
/* First line of every generated main(): a client that leaves must never be
|
||||
* able to signal this process to death. See el_runtime_ignore_sigpipe. */
|
||||
el_runtime_ignore_sigpipe();
|
||||
_el_args_list = el_list_empty();
|
||||
for (int i = 1; i < argc; i++) {
|
||||
_el_args_list = el_list_append(_el_args_list, EL_STR(argv[i]));
|
||||
|
||||
Reference in New Issue
Block a user