|
|
|
@@ -416,6 +416,46 @@ fn engram_extract_ids(nodes_json: String) -> String {
|
|
|
|
|
// A proper cache/circuit-breaker requires C runtime support (e.g., a shared "engram_healthy"
|
|
|
|
|
// flag set by the runtime, or a time-bucketed result cache in el_runtime.c). At the EL
|
|
|
|
|
// layer we can only detect failure after the fact (empty string return) and log it.
|
|
|
|
|
// affective_node_ts — unix timestamp of an affective node (BellEvent / PositiveEvent).
|
|
|
|
|
//
|
|
|
|
|
// Prefers the " | ts:<epoch>" marker auto_persist writes into the node content; falls back
|
|
|
|
|
// to created_at / updated_at. Returns 0 when there is no usable timestamp, which every
|
|
|
|
|
// caller already treats as "too old to surface".
|
|
|
|
|
//
|
|
|
|
|
// ─── ELC CODEGEN NOTE — THIS MUST STAY A TOP-LEVEL FUNCTION ───────────────────────────
|
|
|
|
|
// Do not inline this back into a block-expression initializer. Written inline as
|
|
|
|
|
// let start: Int = pos + str_len(marker)
|
|
|
|
|
// inside a `let x: String = if cond { ... }` initializer, elc loses the declared Int type
|
|
|
|
|
// and emits el_str_concat() for the `+`. el_str_concat takes the C string of each operand,
|
|
|
|
|
// so two integers become a wild pointer and the daemon SEGFAULTS (EXC_BAD_ACCESS in
|
|
|
|
|
// strlen). The identical expression in a plain function body compiles to integer addition
|
|
|
|
|
// — verified in the generated C both ways. This is the same defect family Will hit on
|
|
|
|
|
// 2026-06-23 and solved the same way (aff_try_slot, soul.el).
|
|
|
|
|
//
|
|
|
|
|
// Six inline copies of this parser existed before this function: two in layered_cycle's
|
|
|
|
|
// L2c, two in engram_compile below, two in affective_context_prefix. All six emitted the
|
|
|
|
|
// bad concat and all six now call here. Full write-up: BUG-PLAINCHAT-1 in
|
|
|
|
|
// _engine-plainchat-20260805/README.md and the PR that introduced this function.
|
|
|
|
|
// ──────────────────────────────────────────────────────────────────────────────────────
|
|
|
|
|
fn affective_node_ts(node_json: String) -> Int {
|
|
|
|
|
if str_eq(node_json, "") { return 0 }
|
|
|
|
|
let content: String = json_get(node_json, "content")
|
|
|
|
|
let marker: String = " | ts:"
|
|
|
|
|
let mpos: Int = str_index_of(content, marker)
|
|
|
|
|
if mpos < 0 {
|
|
|
|
|
let ca: String = json_get(node_json, "created_at")
|
|
|
|
|
let alt: String = if str_eq(ca, "") { json_get(node_json, "updated_at") } else { ca }
|
|
|
|
|
if !engram_numeric_valid(alt) { return 0 }
|
|
|
|
|
return str_to_int(alt)
|
|
|
|
|
}
|
|
|
|
|
let start: Int = mpos + str_len(marker)
|
|
|
|
|
let rest: String = str_slice(content, start, str_len(content))
|
|
|
|
|
let nxt: Int = str_index_of(rest, " | ")
|
|
|
|
|
let raw: String = if nxt < 0 { rest } else { str_slice(rest, 0, nxt) }
|
|
|
|
|
if !engram_numeric_valid(raw) { return 0 }
|
|
|
|
|
return str_to_int(raw)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
fn engram_compile(intent: String) -> String {
|
|
|
|
|
// Issue 1: decompose multi-topic messages into sub-queries.
|
|
|
|
|
let topics: String = engram_split_topics(intent)
|
|
|
|
@@ -519,20 +559,9 @@ fn engram_compile(intent: String) -> String {
|
|
|
|
|
let cutoff_ts: Int = now_ts - 1209600
|
|
|
|
|
let recent_bell: String = if bell_ok {
|
|
|
|
|
let bn0: String = json_array_get(bell_nodes, 0)
|
|
|
|
|
let bn_content: String = json_get(bn0, "content")
|
|
|
|
|
let ts_marker: String = " | ts:"
|
|
|
|
|
let ts_pos: Int = str_index_of(bn_content, ts_marker)
|
|
|
|
|
let bn_ts_raw: String = if ts_pos >= 0 {
|
|
|
|
|
let ts_start: Int = ts_pos + str_len(ts_marker)
|
|
|
|
|
let rest: String = str_slice(bn_content, ts_start, str_len(bn_content))
|
|
|
|
|
let next_sep: Int = str_index_of(rest, " | ")
|
|
|
|
|
if next_sep < 0 { rest } else { str_slice(rest, 0, next_sep) }
|
|
|
|
|
} else {
|
|
|
|
|
let ca: String = json_get(bn0, "created_at")
|
|
|
|
|
if str_eq(ca, "") { json_get(bn0, "updated_at") } else { ca }
|
|
|
|
|
}
|
|
|
|
|
// Q1 fix: validate bell timestamp before str_to_int.
|
|
|
|
|
let bn_ts: Int = if !engram_numeric_valid(bn_ts_raw) { 0 } else { str_to_int(bn_ts_raw) }
|
|
|
|
|
// Q1 fix (validate before str_to_int) now lives inside affective_node_ts, which
|
|
|
|
|
// also replaces the inline " | ts:" parser that miscompiled to el_str_concat here.
|
|
|
|
|
let bn_ts: Int = affective_node_ts(bn0)
|
|
|
|
|
if bn_ts > cutoff_ts { bn0 } else { "" }
|
|
|
|
|
} else { "" }
|
|
|
|
|
// Positive emotion context: check for recent joy/success moments within 72h.
|
|
|
|
@@ -540,19 +569,7 @@ fn engram_compile(intent: String) -> String {
|
|
|
|
|
let pos_ec_ok: Bool = !str_eq(pos_ec_nodes, "") && !str_eq(pos_ec_nodes, "[]")
|
|
|
|
|
let recent_positive_ec: String = if pos_ec_ok {
|
|
|
|
|
let pec0: String = json_array_get(pos_ec_nodes, 0)
|
|
|
|
|
let pec_content: String = json_get(pec0, "content")
|
|
|
|
|
let pec_ts_marker: String = " | ts:"
|
|
|
|
|
let pec_ts_pos: Int = str_index_of(pec_content, pec_ts_marker)
|
|
|
|
|
let pec_ts_raw: String = if pec_ts_pos >= 0 {
|
|
|
|
|
let pec_ts_start: Int = pec_ts_pos + str_len(pec_ts_marker)
|
|
|
|
|
let pec_rest: String = str_slice(pec_content, pec_ts_start, str_len(pec_content))
|
|
|
|
|
let pec_next: Int = str_index_of(pec_rest, " | ")
|
|
|
|
|
if pec_next < 0 { pec_rest } else { str_slice(pec_rest, 0, pec_next) }
|
|
|
|
|
} else {
|
|
|
|
|
let pec_ca: String = json_get(pec0, "created_at")
|
|
|
|
|
if str_eq(pec_ca, "") { json_get(pec0, "updated_at") } else { pec_ca }
|
|
|
|
|
}
|
|
|
|
|
let pec_ts: Int = if str_eq(pec_ts_raw, "") { 0 } else { str_to_int(pec_ts_raw) }
|
|
|
|
|
let pec_ts: Int = affective_node_ts(pec0)
|
|
|
|
|
if pec_ts > cutoff_ts { pec0 } else { "" }
|
|
|
|
|
} else { "" }
|
|
|
|
|
let affective_part: String = if !str_eq(recent_bell, "") {
|
|
|
|
@@ -768,7 +785,12 @@ fn build_system_prompt(ctx: String, chat_mode: Bool) -> String {
|
|
|
|
|
safety_addendum
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return identity + operator_section + date_line + voice_rules + security_rules + capability_rules + bounded_persona_block + identity_block + affective_boot_block + engram_block + safety_block
|
|
|
|
|
// BUG FIX 2026-08-05: no_tools_rule was computed above and then never concatenated into
|
|
|
|
|
// this return, so the "[NO TOOLS THIS TURN]" instruction has not actually reached a model
|
|
|
|
|
// in this revision — the chat_mode flag had no effect on the prompt. Restored here, in the
|
|
|
|
|
// permanent-rules group, immediately after capability_rules (the rule it qualifies).
|
|
|
|
|
// Zero effect on agentic paths: they pass chat_mode=false, so no_tools_rule is "".
|
|
|
|
|
return identity + operator_section + date_line + voice_rules + security_rules + capability_rules + no_tools_rule + bounded_persona_block + identity_block + affective_boot_block + engram_block + safety_block
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
fn hist_append(hist: String, role: String, content: String) -> String {
|
|
|
|
@@ -929,6 +951,123 @@ fn conv_history_load() -> String {
|
|
|
|
|
return content
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// conv_history_record — append one completed turn to the conversation window.
|
|
|
|
|
//
|
|
|
|
|
// Same window, same append, same bell-guarded eviction handle_chat uses inline. It exists
|
|
|
|
|
// as a function so the layered_cycle path records turns through exactly this code instead
|
|
|
|
|
// of growing a second copy that can drift away from the bell guard.
|
|
|
|
|
//
|
|
|
|
|
// CONTRACT: assistant_msg MUST be post-safety_validate text. Recording the validated text
|
|
|
|
|
// rather than the raw model output means the history window can never replay something the
|
|
|
|
|
// output gate replaced or augmented. Callers on a hard bell must not call this at all —
|
|
|
|
|
// bell turns are kept out of conversation history by design (see layered_cycle).
|
|
|
|
|
fn conv_history_record(user_msg: String, assistant_msg: String) -> Void {
|
|
|
|
|
if str_eq(user_msg, "") { return "" }
|
|
|
|
|
let state_hist: String = state_get("conv_history")
|
|
|
|
|
let stored_hist: String = if str_eq(state_hist, "") { conv_history_load() } else { state_hist }
|
|
|
|
|
let h1: String = hist_append(stored_hist, "user", user_msg)
|
|
|
|
|
let h2: String = hist_append(h1, "assistant", assistant_msg)
|
|
|
|
|
// Bell-guarded trim: an evicted turn that triggered a bell is preserved to engram
|
|
|
|
|
// before it leaves the in-memory window.
|
|
|
|
|
let final_hist: String = if json_array_len(h2) > 20 {
|
|
|
|
|
hist_trim_with_bell_guard(h2)
|
|
|
|
|
} else {
|
|
|
|
|
h2
|
|
|
|
|
}
|
|
|
|
|
state_set("conv_history", final_hist)
|
|
|
|
|
conv_history_persist(final_hist)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// conv_history_block — recent dialogue, rendered for a system prompt.
|
|
|
|
|
//
|
|
|
|
|
// Same rendering handle_chat uses (role label + snipped content, one line per turn), read
|
|
|
|
|
// from the same "conv_history" window, so a plain-chat turn can follow the thread instead
|
|
|
|
|
// of answering every message from cold. Read-only: never writes history.
|
|
|
|
|
fn conv_history_block() -> String {
|
|
|
|
|
let state_hist: String = state_get("conv_history")
|
|
|
|
|
let stored_hist: String = if str_eq(state_hist, "") { conv_history_load() } else { state_hist }
|
|
|
|
|
let hist_len: Int = if str_eq(stored_hist, "") { 0 } else { json_array_len(stored_hist) }
|
|
|
|
|
if hist_len == 0 {
|
|
|
|
|
return ""
|
|
|
|
|
}
|
|
|
|
|
let rh_out: String = ""
|
|
|
|
|
let rh_i: Int = 0
|
|
|
|
|
while rh_i < hist_len {
|
|
|
|
|
let rh_entry: String = json_array_get(stored_hist, rh_i)
|
|
|
|
|
let rh_role: String = json_get(rh_entry, "role")
|
|
|
|
|
let rh_content: String = json_get(rh_entry, "content")
|
|
|
|
|
let rh_label: String = if str_eq(rh_role, "user") { "User" } else { "Assistant" }
|
|
|
|
|
let rh_snip: String = if str_len(rh_content) > 400 { str_slice(rh_content, 0, 400) + "..." } else { rh_content }
|
|
|
|
|
let rh_line: String = rh_label + ": " + rh_snip
|
|
|
|
|
let rh_out = if str_eq(rh_out, "") { rh_line } else { rh_out + "\n" + rh_line }
|
|
|
|
|
let rh_i = rh_i + 1
|
|
|
|
|
}
|
|
|
|
|
return "\n\n[RECENT CONVERSATION — last " + int_to_str(hist_len) + " turns]\n" + rh_out
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// layered_generate — the L3 generation step of layered_cycle. This is where the imprint
|
|
|
|
|
// SPEAKS. (Added 2026-08-05.)
|
|
|
|
|
//
|
|
|
|
|
// layered_cycle calls this immediately after imprint_respond(), which has already applied
|
|
|
|
|
// the active imprint's voice/domain annotation to the steward-aligned input. Before this
|
|
|
|
|
// existed, L3 ended at that annotation and layered_cycle handed the user's own text back
|
|
|
|
|
// as the "reply" — every gate ran, but nothing ever generated.
|
|
|
|
|
//
|
|
|
|
|
// It lives in chat.el, not imprint.el, on purpose: imprint.el is L3 and declares that the
|
|
|
|
|
// lower layers are structurally inaccessible from it. It has zero imports, and making it
|
|
|
|
|
// reach chat.el would transitively pull in safety.el (L1), inverting the layering and
|
|
|
|
|
// breaking the tests that link imprint.el on its own. The layer ORDER is enforced by
|
|
|
|
|
// layered_cycle, which is the only caller; this function holds no layer authority.
|
|
|
|
|
//
|
|
|
|
|
// SAFETY CONTRACT — this function is NOT a gate and must never become one:
|
|
|
|
|
// - Everything upstream has already run inside layered_cycle: L1 safety_screen, the
|
|
|
|
|
// safe-mode guard, the hard-bell short-circuit, L2a continuity/profiling, L2b mission
|
|
|
|
|
// alignment, L2c affective context. A hard bell can never reach this function —
|
|
|
|
|
// layered_cycle returns the fixed crisis message before L3 is entered.
|
|
|
|
|
// - Everything downstream is safety_validate(), which layered_cycle applies to this
|
|
|
|
|
// function's return value. Nothing here may bypass it, so this always returns plain
|
|
|
|
|
// text: no JSON envelope, no escaping, nothing for the output gate to have to unwrap.
|
|
|
|
|
// - The bell directive layered_cycle computed with safety_augment_system() is parked in
|
|
|
|
|
// the state key "layered_cycle_safety_system_addendum", and build_system_prompt() is
|
|
|
|
|
// its designated consumer. That is why the system prompt is assembled through
|
|
|
|
|
// build_system_prompt() here rather than hand-rolled: routing around it would drop the
|
|
|
|
|
// soft-bell / crisis directive on the floor. safety_augment_system() is NOT called
|
|
|
|
|
// again here — one evaluation per turn, one InternalStateEvent per bell.
|
|
|
|
|
// - No affective_context_prefix() call either: L2c already folded the affective note
|
|
|
|
|
// into that same addendum, and injecting it twice would double the cue.
|
|
|
|
|
//
|
|
|
|
|
// TOOLS: none, structurally. build_system_prompt(ctx, true) is chat mode, which injects
|
|
|
|
|
// the permanent "NO TOOLS THIS TURN" rule, and llm_call_system() is a plain /v1/messages
|
|
|
|
|
// call whose request body is built in el_runtime.c (llm_provider_request) with no "tools"
|
|
|
|
|
// and no "tool_choice" key at all. Tools:Off means no tool is offered to the model, not
|
|
|
|
|
// merely that none is used.
|
|
|
|
|
//
|
|
|
|
|
// Returns "" when the model call fails, so the caller reports the failure honestly instead
|
|
|
|
|
// of echoing the user's own text back at them.
|
|
|
|
|
fn layered_generate(prompt: String, imprint_id: String) -> String {
|
|
|
|
|
if str_eq(prompt, "") {
|
|
|
|
|
return ""
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
let ctx: String = engram_compile(prompt)
|
|
|
|
|
let model: String = chat_default_model()
|
|
|
|
|
let base_system: String = build_system_prompt(ctx, true) + current_engine_note(model)
|
|
|
|
|
let hist_block: String = conv_history_block()
|
|
|
|
|
let full_system: String = base_system + hist_block
|
|
|
|
|
|
|
|
|
|
let raw: String = llm_call_system(model, full_system, prompt)
|
|
|
|
|
|
|
|
|
|
let is_error: Bool = str_starts_with(raw, "{\"error\"")
|
|
|
|
|
|| str_starts_with(raw, "{\"type\":\"error\"")
|
|
|
|
|
|| str_contains(raw, "authentication_error")
|
|
|
|
|
if is_error {
|
|
|
|
|
println("[chat] layered_generate: model call failed — returning empty so the caller can report it honestly")
|
|
|
|
|
return ""
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return clean_llm_response(raw)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// session_preload_bullets — render up to max_bullets nodes from a JSON array as
|
|
|
|
|
// bullet lines, truncating content at snip_len chars each.
|
|
|
|
|
fn session_preload_bullets(nodes: String, max_bullets: Int, snip_len: Int) -> String {
|
|
|
|
@@ -969,19 +1108,7 @@ fn affective_context_prefix() -> String {
|
|
|
|
|
} else {
|
|
|
|
|
if has_dist_aff {
|
|
|
|
|
let dn0: String = json_array_get(dist_nodes_aff, 0)
|
|
|
|
|
let dn_content: String = json_get(dn0, "content")
|
|
|
|
|
let daff_marker: String = " | ts:"
|
|
|
|
|
let daff_pos: Int = str_index_of(dn_content, daff_marker)
|
|
|
|
|
let daff_ts_str: String = if daff_pos >= 0 {
|
|
|
|
|
let daff_start: Int = daff_pos + str_len(daff_marker)
|
|
|
|
|
let daff_rest: String = str_slice(dn_content, daff_start, str_len(dn_content))
|
|
|
|
|
let daff_next: Int = str_index_of(daff_rest, " | ")
|
|
|
|
|
if daff_next < 0 { daff_rest } else { str_slice(daff_rest, 0, daff_next) }
|
|
|
|
|
} else {
|
|
|
|
|
let daff_ca: String = json_get(dn0, "created_at")
|
|
|
|
|
if str_eq(daff_ca, "") { json_get(dn0, "updated_at") } else { daff_ca }
|
|
|
|
|
}
|
|
|
|
|
let daff_ts: Int = if str_eq(daff_ts_str, "") { 0 } else { str_to_int(daff_ts_str) }
|
|
|
|
|
let daff_ts: Int = affective_node_ts(dn0)
|
|
|
|
|
daff_ts > aff_cutoff
|
|
|
|
|
} else { false }
|
|
|
|
|
}
|
|
|
|
@@ -989,19 +1116,7 @@ fn affective_context_prefix() -> String {
|
|
|
|
|
let has_pos_aff: Bool = !str_eq(pos_nodes_aff, "") && !str_eq(pos_nodes_aff, "[]")
|
|
|
|
|
let found_recent_pos: Bool = if has_pos_aff && !found_recent_dist {
|
|
|
|
|
let pn0: String = json_array_get(pos_nodes_aff, 0)
|
|
|
|
|
let pn_content: String = json_get(pn0, "content")
|
|
|
|
|
let paff_marker: String = " | ts:"
|
|
|
|
|
let paff_pos: Int = str_index_of(pn_content, paff_marker)
|
|
|
|
|
let paff_ts_str: String = if paff_pos >= 0 {
|
|
|
|
|
let paff_start: Int = paff_pos + str_len(paff_marker)
|
|
|
|
|
let paff_rest: String = str_slice(pn_content, paff_start, str_len(pn_content))
|
|
|
|
|
let paff_next: Int = str_index_of(paff_rest, " | ")
|
|
|
|
|
if paff_next < 0 { paff_rest } else { str_slice(paff_rest, 0, paff_next) }
|
|
|
|
|
} else {
|
|
|
|
|
let paff_ca: String = json_get(pn0, "created_at")
|
|
|
|
|
if str_eq(paff_ca, "") { json_get(pn0, "updated_at") } else { paff_ca }
|
|
|
|
|
}
|
|
|
|
|
let paff_ts: Int = if str_eq(paff_ts_str, "") { 0 } else { str_to_int(paff_ts_str) }
|
|
|
|
|
let paff_ts: Int = affective_node_ts(pn0)
|
|
|
|
|
paff_ts > aff_cutoff
|
|
|
|
|
} else { false }
|
|
|
|
|
let affective_out: String = if found_recent_dist {
|
|
|
|
@@ -1014,6 +1129,23 @@ fn affective_context_prefix() -> String {
|
|
|
|
|
return affective_out
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
|
|
|
// handle_chat — UNWIRED. DO NOT ROUTE /api/chat HERE. (annotated 2026-08-05)
|
|
|
|
|
//
|
|
|
|
|
// This was the non-agentic chat handler until 2026-06-11 (f52d5bd, "wire consciousness
|
|
|
|
|
// layers"), when Will moved /api/chat onto layered_cycle. It has had zero call sites since.
|
|
|
|
|
//
|
|
|
|
|
// It must stay unwired: it has NO enforcing input gate and NO enforcing output gate.
|
|
|
|
|
// It never calls safety_screen, so a hard bell would reach the model instead of being
|
|
|
|
|
// refused; it never calls safety_validate, the only enforcing output gate in the codebase;
|
|
|
|
|
// and it runs no stewardship layer. Its one safety touch, safety_augment_system at the
|
|
|
|
|
// llm_call_system site below, is an advisory system-prompt string — it cannot refuse,
|
|
|
|
|
// replace, or block anything.
|
|
|
|
|
//
|
|
|
|
|
// Plain chat generates through layered_cycle's L3 (layered_generate) instead, which keeps
|
|
|
|
|
// the screen, the stewardship layers and the output gate wrapped around the model call.
|
|
|
|
|
// If this function is ever revived, it must be gated first, not wired first.
|
|
|
|
|
// ─────────────────────────────────────────────────────────────────────────────
|
|
|
|
|
fn handle_chat(body: String) -> String {
|
|
|
|
|
let message: String = json_get(body, "message")
|
|
|
|
|
if str_eq(message, "") {
|
|
|
|
@@ -1410,6 +1542,71 @@ fn agentic_tools_literal() -> String {
|
|
|
|
|
"]"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// web_search_tool_json — the ONE place the server-side web_search tool block is built.
|
|
|
|
|
//
|
|
|
|
|
// Anthropic executes this tool on their side, inside the same /v1/messages call: no
|
|
|
|
|
// third-party search key, no new account, and no anthropic-beta header (the plain
|
|
|
|
|
// "anthropic-version: 2023-06-01" this soul already sends is sufficient).
|
|
|
|
|
//
|
|
|
|
|
// FUTURE-PROOF (design carried from soul-webfix-20260711.patch, Tim-approved): the tool
|
|
|
|
|
// VERSION lives in exactly one place — state key "web_search_tool_version" — so a future
|
|
|
|
|
// bump is a config write, not a recompile.
|
|
|
|
|
//
|
|
|
|
|
// DEFAULT IS THE BASIC VARIANT, AND THAT IS A DELIBERATE, MEASURED CHOICE.
|
|
|
|
|
// The July patch defaulted to web_search_20260209 (dynamic filtering). That variant is
|
|
|
|
|
// HARD-INCOMPATIBLE with the parallel-tool stopgap this engine currently depends on
|
|
|
|
|
// (ADR 0005). Dynamic filtering is implemented with server-side programmatic tool
|
|
|
|
|
// calling, and the API rejects the pair outright — verified live 2026-08-04, verbatim:
|
|
|
|
|
//
|
|
|
|
|
// HTTP 400 invalid_request_error
|
|
|
|
|
// "tool_choice.disable_parallel_tool_use: true cannot be used with programmatic
|
|
|
|
|
// tool calling"
|
|
|
|
|
//
|
|
|
|
|
// So the two cannot both ship today. Dropping the stopgap would resurrect neuron#78
|
|
|
|
|
// bug b, which killed agentic runs 3/3 in the ADR's own A/B — a functional regression.
|
|
|
|
|
// Dropping web search entirely would leave the product's promise unmet. The basic
|
|
|
|
|
// variant is compatible with the stopgap AND returns real, current results (proven
|
|
|
|
|
// E2E), so it is what we default to. What we give up is dynamic filtering — a
|
|
|
|
|
// token-efficiency and result-quality nicety, not the capability itself.
|
|
|
|
|
//
|
|
|
|
|
// REMOVAL TRIGGER: this default should move to web_search_20260209 the moment ADR
|
|
|
|
|
// 0005's stopgap is retired (i.e. when the soul calls el_runtime's llm_call_agentic
|
|
|
|
|
// and no longer needs disable_parallel_tool_use). At that point it is a one-line
|
|
|
|
|
// config write — state_set("web_search_tool_version", "web_search_20260209") — with
|
|
|
|
|
// no recompile. Flagged for Will's ratification; see the PR body.
|
|
|
|
|
fn web_search_tool_json() -> String {
|
|
|
|
|
let ver: String = state_get("web_search_tool_version")
|
|
|
|
|
let eff: String = if str_eq(ver, "") { "web_search_20250305" } else { ver }
|
|
|
|
|
return "{\"type\":\"" + eff + "\",\"name\":\"web_search\",\"max_uses\":5}"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// strip_client_web_search — remove any CLIENT-side tool literally named "web_search"
|
|
|
|
|
// from a tools-array interior, so attaching Anthropic's server-side tool cannot produce
|
|
|
|
|
// two tools sharing one name (the API rejects that with "Tool names must be unique",
|
|
|
|
|
// and before it did, the model picked between them nondeterministically).
|
|
|
|
|
//
|
|
|
|
|
// The soul's own literal set carries "web_get", not "web_search", so today this is a
|
|
|
|
|
// no-op there — it exists for the MCP connector tools merged in by agentic_tools_all(),
|
|
|
|
|
// which are third-party and may well ship a "web_search".
|
|
|
|
|
//
|
|
|
|
|
// LIMITATION (inherited from the July patch, stated rather than hidden): the scan keys
|
|
|
|
|
// on the object terminator "]}}," so it only strips a matching tool that is followed by
|
|
|
|
|
// another tool. A client web_search in the LAST array position is left in place.
|
|
|
|
|
fn strip_client_web_search(tools_inner: String) -> String {
|
|
|
|
|
let ws_start: Int = str_index_of(tools_inner, "{\"name\":\"web_search\"")
|
|
|
|
|
if ws_start < 0 {
|
|
|
|
|
return tools_inner
|
|
|
|
|
}
|
|
|
|
|
let ws_rest: String = str_slice(tools_inner, ws_start, str_len(tools_inner))
|
|
|
|
|
let ws_end: Int = str_index_of(ws_rest, "]}},")
|
|
|
|
|
if ws_end <= 0 {
|
|
|
|
|
return tools_inner
|
|
|
|
|
}
|
|
|
|
|
let head: String = str_slice(tools_inner, 0, ws_start)
|
|
|
|
|
let tail: String = str_slice(ws_rest, ws_end + 4, str_len(ws_rest))
|
|
|
|
|
return head + tail
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// agentic_tools_with_web — the standard tool set, always plus Anthropic's NATIVE
|
|
|
|
|
// server-side web_search tool. Web search is BUILT IN: the model invokes it only when a
|
|
|
|
|
// query needs fresh info (max_uses caps it), so there is no user-facing toggle. The native
|
|
|
|
@@ -1417,8 +1614,8 @@ fn agentic_tools_literal() -> String {
|
|
|
|
|
// and needs no local runtime — it sidesteps the soul's lack of executable tools entirely.
|
|
|
|
|
fn agentic_tools_with_web() -> String {
|
|
|
|
|
let base: String = agentic_tools_literal()
|
|
|
|
|
let inner: String = str_slice(base, 1, str_len(base) - 1)
|
|
|
|
|
return "[" + inner + ",{\"type\":\"web_search_20250305\",\"name\":\"web_search\",\"max_uses\":5}]"
|
|
|
|
|
let inner: String = strip_client_web_search(str_slice(base, 1, str_len(base) - 1))
|
|
|
|
|
return "[" + inner + "," + web_search_tool_json() + "]"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ---------------------------------------------------------------------------
|
|
|
|
@@ -1443,20 +1640,35 @@ fn connector_tools_json() -> String {
|
|
|
|
|
return arr
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Built-in tools + every connector tool, as one tools array.
|
|
|
|
|
// Uses agentic_tools_literal (not agentic_tools_with_web) to avoid a duplicate
|
|
|
|
|
// "web_search" name — the literal already includes a custom web_search handler,
|
|
|
|
|
// and adding the Anthropic server-side web_search_20250305 (same name) causes
|
|
|
|
|
// Anthropic to reject with "Tool names must be unique."
|
|
|
|
|
// agentic_tools_all — built-in tools + every connector tool + Anthropic's NATIVE
|
|
|
|
|
// server-side web_search, as one tools array. This is the tool set EVERY agentic route
|
|
|
|
|
// uses (handle_chat_agentic and handle_dharma_room_turn_agentic both call it; a
|
|
|
|
|
// bridge-suspended run carries the same array through agentic_resume), so attaching the
|
|
|
|
|
// web tool here is what actually turns web search on for the product.
|
|
|
|
|
//
|
|
|
|
|
// ACTIVATION — restoring an existing design, not inventing a mechanism:
|
|
|
|
|
// * Commit 8eea1d9 (2026-06-09, Tim-approved) made native web_search BUILT IN with no
|
|
|
|
|
// user-facing toggle — "the model invokes it only when a query needs fresh info
|
|
|
|
|
// (max_uses:5 caps it)" — and the desktop app removed its web-search toggle to pair
|
|
|
|
|
// with that. The call site was lost when agentic_tools_all() (connector tools, PR #19)
|
|
|
|
|
// replaced agentic_tools_with_web() in handle_chat_agentic.
|
|
|
|
|
// * tests/test_agentic_tools.el §2 still asserts agentic_tools_all() contains the
|
|
|
|
|
// native web_search tool. main currently FAILS that assertion; this restores it.
|
|
|
|
|
// The old comment here claimed "the literal already includes a custom web_search handler"
|
|
|
|
|
// — that is stale: agentic_tools_literal() ships web_get, and is_builtin_tool() has no
|
|
|
|
|
// web_search. The duplicate-name hazard now lives only in third-party connector tools,
|
|
|
|
|
// which is exactly what strip_client_web_search() handles.
|
|
|
|
|
fn agentic_tools_all() -> String {
|
|
|
|
|
let base: String = agentic_tools_literal()
|
|
|
|
|
let conn: String = connector_tools_json()
|
|
|
|
|
let base_inner: String = str_slice(base, 1, str_len(base) - 1)
|
|
|
|
|
let conn_inner: String = str_slice(conn, 1, str_len(conn) - 1)
|
|
|
|
|
if str_eq(conn_inner, "") {
|
|
|
|
|
return base
|
|
|
|
|
let merged: String = if str_eq(conn_inner, "") {
|
|
|
|
|
base_inner
|
|
|
|
|
} else {
|
|
|
|
|
base_inner + "," + conn_inner
|
|
|
|
|
}
|
|
|
|
|
let base_open: String = str_slice(base, 0, str_len(base) - 1)
|
|
|
|
|
return base_open + "," + conn_inner + "]"
|
|
|
|
|
return "[" + strip_client_web_search(merged) + "," + web_search_tool_json() + "]"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Proxy one tool call to the bridge. The model-supplied input is written to a
|
|
|
|
@@ -2223,6 +2435,20 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
let iteration: Int = 0
|
|
|
|
|
let keep_going: Bool = true
|
|
|
|
|
|
|
|
|
|
// Server-side web_search state, all three carried across iterations.
|
|
|
|
|
//
|
|
|
|
|
// tools_eff: a local copy of the caller's tools array, because the DRIFT branch below
|
|
|
|
|
// may rewrite the web_search version in place after an API rejection. (Parameters are
|
|
|
|
|
// not rebindable across loop iterations; a top-level local is.)
|
|
|
|
|
// container_id: web_search's dynamic filtering runs server-side code execution on
|
|
|
|
|
// Anthropic's side, which hands back a container id that MUST be echoed on every
|
|
|
|
|
// follow-up request in the same turn chain or the API 400s with "container_id is
|
|
|
|
|
// required". Captured from each response, replayed on the next.
|
|
|
|
|
// ws_drift: set when we fell back, so we only ever fall back once per run.
|
|
|
|
|
let tools_eff: String = tools_json
|
|
|
|
|
let container_id: String = ""
|
|
|
|
|
let ws_drift: Bool = false
|
|
|
|
|
|
|
|
|
|
// Suspension state — captured at top level so it escapes the while body.
|
|
|
|
|
let pending: Bool = false
|
|
|
|
|
let pend_tool_id: String = ""
|
|
|
|
@@ -2242,11 +2468,39 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
state_set("run_progress_" + session_id, "")
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
while keep_going && iteration < 8 {
|
|
|
|
|
// Cap raised 8 -> 12: a server-side web_search turn can pause and resume several
|
|
|
|
|
// times (see is_pause below), and each resume consumes an iteration. At 8 a
|
|
|
|
|
// multi-search answer could exhaust the loop before the model finished writing.
|
|
|
|
|
while keep_going && iteration < 12 {
|
|
|
|
|
// Carry the server-side code-execution container forward when we have one.
|
|
|
|
|
let cont_frag: String = if str_eq(container_id, "") {
|
|
|
|
|
""
|
|
|
|
|
} else {
|
|
|
|
|
",\"container\":\"" + container_id + "\""
|
|
|
|
|
}
|
|
|
|
|
// STOPGAP (2026-08-03, neuron#78 bug b / ADR 0005): the block walk below keeps
|
|
|
|
|
// only the FIRST tool_use block per round, so a PARALLEL tool_use response leaves
|
|
|
|
|
// the other ids without tool_result and the next turn 400s with
|
|
|
|
|
// "tool_use ids found without tool_result" — killing the run. Sending Anthropic's
|
|
|
|
|
// documented tool_choice.disable_parallel_tool_use makes the wire contract match
|
|
|
|
|
// what this loop can actually assemble. The real fix — a multi-tool_result loop
|
|
|
|
|
// that answers every block in a round — is Will's; see neuron#78.
|
|
|
|
|
//
|
|
|
|
|
// The stopgap and server-side web_search coexist: disable_parallel_tool_use
|
|
|
|
|
// constrains how many CLIENT tool_use blocks the model may emit per response, and
|
|
|
|
|
// Anthropic runs web_search itself (it comes back as server_tool_use +
|
|
|
|
|
// web_search_tool_result, never as a client tool_use the loop has to answer).
|
|
|
|
|
// Verified live, not assumed — see the README's probe transcript.
|
|
|
|
|
//
|
|
|
|
|
// max_tokens raised 4096 -> 16384: web_search injects whole result documents into
|
|
|
|
|
// the response, and a multi-search answer at 4096 truncated mid-sentence. This is
|
|
|
|
|
// the second half of the anti-truncation fix; pause_turn handling is the first.
|
|
|
|
|
let req_body: String = "{\"model\":\"" + model + "\""
|
|
|
|
|
+ ",\"max_tokens\":4096"
|
|
|
|
|
+ cont_frag
|
|
|
|
|
+ ",\"max_tokens\":16384"
|
|
|
|
|
+ ",\"tool_choice\":{\"type\":\"auto\",\"disable_parallel_tool_use\":true}"
|
|
|
|
|
+ ",\"system\":\"" + safe_sys + "\""
|
|
|
|
|
+ ",\"tools\":" + tools_json
|
|
|
|
|
+ ",\"tools\":" + tools_eff
|
|
|
|
|
+ ",\"messages\":" + messages
|
|
|
|
|
+ "}"
|
|
|
|
|
|
|
|
|
@@ -2255,11 +2509,60 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
let is_error: Bool = str_starts_with(raw_resp, "{\"error\"")
|
|
|
|
|
|| str_starts_with(raw_resp, "{\"type\":\"error\"")
|
|
|
|
|
|| str_contains(raw_resp, "authentication_error")
|
|
|
|
|
if is_error {
|
|
|
|
|
|
|
|
|
|
// DRIFT / FALLBACK: if the API rejected our configured web_search variant (the
|
|
|
|
|
// usual cause is a model too old for it — chat_default_model() is sonnet-4-5,
|
|
|
|
|
// which predates web_search_20260209), swap to the known-old basic variant,
|
|
|
|
|
// record the drift for the drift-watch, and retry this iteration.
|
|
|
|
|
//
|
|
|
|
|
// Everything here is computed as top-level if-expressions rather than inside an
|
|
|
|
|
// if-BLOCK, because in El a mutation inside an if-block does not escape its scope
|
|
|
|
|
// (see the block-walk note below). There is deliberately no `continue`: the
|
|
|
|
|
// fallback simply leaves `messages` untouched and keeps `keep_going` true, so the
|
|
|
|
|
// next iteration re-sends the same turn with the corrected tools array.
|
|
|
|
|
let ws_pos: Int = if is_error { str_index_of(tools_eff, "\"type\":\"web_search_") } else { 0 - 1 }
|
|
|
|
|
let ws_rest: String = if ws_pos >= 0 { str_slice(tools_eff, ws_pos + 8, str_len(tools_eff)) } else { "" }
|
|
|
|
|
// '"type":"' is 8 chars, so the version starts at ws_pos+8 and ends at the next quote.
|
|
|
|
|
let ws_vend: Int = if ws_pos >= 0 { str_index_of(ws_rest, "\"") } else { 0 - 1 }
|
|
|
|
|
let ws_cur: String = if ws_vend > 0 { str_slice(ws_rest, 0, ws_vend) } else { "" }
|
|
|
|
|
// Fall back on either of two signals, and ONLY these two — a loose prefix guess
|
|
|
|
|
// here once matched errors that had nothing to do with the tool:
|
|
|
|
|
// 1. the API names OUR configured version in its complaint (version drift, e.g.
|
|
|
|
|
// a model too old for the configured variant); or
|
|
|
|
|
// 2. the API rejects the request for "programmatic tool calling" — the verified
|
|
|
|
|
// signature of the dynamic-filtering variant colliding with ADR 0005's
|
|
|
|
|
// disable_parallel_tool_use. An operator who sets web_search_tool_version to
|
|
|
|
|
// web_search_20260209 while the stopgap still stands would otherwise get a
|
|
|
|
|
// dead chat; this downgrades them automatically and says so in the log.
|
|
|
|
|
let ws_conflict: Bool = str_contains(raw_resp, "programmatic tool calling")
|
|
|
|
|
let can_fallback: Bool = is_error && !ws_drift && ws_pos >= 0 && ws_vend > 0
|
|
|
|
|
&& !str_eq(ws_cur, "") && !str_eq(ws_cur, "web_search_20250305")
|
|
|
|
|
&& (str_contains(raw_resp, ws_cur) || ws_conflict)
|
|
|
|
|
let tools_eff = if can_fallback {
|
|
|
|
|
str_slice(tools_eff, 0, ws_pos + 8) + "web_search_20250305" + str_slice(ws_rest, ws_vend, str_len(ws_rest))
|
|
|
|
|
} else { tools_eff }
|
|
|
|
|
let ws_drift = if can_fallback { true } else { ws_drift }
|
|
|
|
|
if can_fallback {
|
|
|
|
|
println("[soul] DRIFT: web_search variant '" + ws_cur + "' rejected by API - fell back to web_search_20250305")
|
|
|
|
|
state_set("web_search_version_drift", ws_cur)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if is_error && !can_fallback {
|
|
|
|
|
// Log the actual API error head — the old code swallowed it entirely, which
|
|
|
|
|
// made every upstream failure look identical from the outside.
|
|
|
|
|
let err_head: String = if str_len(raw_resp) > 220 { str_slice(raw_resp, 0, 220) } else { raw_resp }
|
|
|
|
|
println("[soul] llm error: " + err_head)
|
|
|
|
|
return "{\"error\":\"llm unavailable\",\"reply\":\"\"}"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
let stop_reason: String = json_get(raw_resp, "stop_reason")
|
|
|
|
|
|
|
|
|
|
// Capture/refresh the server-side container id when the response carries one.
|
|
|
|
|
let cont_raw: String = json_get_raw(raw_resp, "container")
|
|
|
|
|
let cont_id_new: String = if !str_eq(cont_raw, "") && !str_eq(cont_raw, "null") {
|
|
|
|
|
json_get(cont_raw, "id")
|
|
|
|
|
} else { "" }
|
|
|
|
|
let container_id = if !str_eq(cont_id_new, "") { cont_id_new } else { container_id }
|
|
|
|
|
// json_get_raw needed — content is an array, json_get returns "" for non-strings
|
|
|
|
|
let content_arr: String = json_get_raw(raw_resp, "content")
|
|
|
|
|
let eff_content: String = if str_eq(content_arr, "") { "[]" } else { content_arr }
|
|
|
|
@@ -2271,13 +2574,39 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
let tool_id: String = ""
|
|
|
|
|
let tool_name: String = ""
|
|
|
|
|
let tool_input: String = ""
|
|
|
|
|
// Server-executed tool names seen this round, quoted and comma-joined. Kept
|
|
|
|
|
// separate from tools_log so this inner walk has exactly one mutation site per
|
|
|
|
|
// variable (the El scope rule below), then merged in at the outer level.
|
|
|
|
|
let srv_log: String = ""
|
|
|
|
|
let ci: Int = 0
|
|
|
|
|
let c_total: Int = json_array_len(eff_content)
|
|
|
|
|
while ci < c_total {
|
|
|
|
|
let block: String = json_array_get(eff_content, ci)
|
|
|
|
|
let btype: String = json_get(block, "type")
|
|
|
|
|
// CITATION-BLOCK FIX (2026-08-04, found by E2E once web_search was live).
|
|
|
|
|
// json_get is a first-match scanner, and a cited text block serialises as
|
|
|
|
|
// {"citations":[{"type":"web_search_result_location",...}],"type":"text",...}
|
|
|
|
|
// — citations FIRST. So json_get(block,"type") returns the nested citation's
|
|
|
|
|
// type, "text" never matches, and the block is silently dropped. Those are
|
|
|
|
|
// exactly the blocks carrying the searched facts, so the user got an answer
|
|
|
|
|
// with the data punched out of it ("The current temperature is , with .").
|
|
|
|
|
// Only text blocks carry a citations array, so its presence identifies the
|
|
|
|
|
// block type unambiguously without needing a real JSON parser.
|
|
|
|
|
let cit_raw: String = json_get_raw(block, "citations")
|
|
|
|
|
let has_cit: Bool = !str_eq(cit_raw, "") && !str_eq(cit_raw, "null")
|
|
|
|
|
let btype_scan: String = json_get(block, "type")
|
|
|
|
|
let btype: String = if has_cit { "text" } else { btype_scan }
|
|
|
|
|
// Accumulate text at top level using if-expression
|
|
|
|
|
let text_out = if str_eq(btype, "text") { text_out + json_get(block, "text") } else { text_out }
|
|
|
|
|
// FUTURE-PROOF: tools Anthropic runs on our behalf (web_search today, whatever
|
|
|
|
|
// ships tomorrow) arrive as server_tool_use blocks, never as client tool_use.
|
|
|
|
|
// Count the CATEGORY by the block's own name so a new server tool appears in
|
|
|
|
|
// tools_used with zero code changes — honest accounting for work we didn't run.
|
|
|
|
|
let is_srv: Bool = str_eq(btype, "server_tool_use")
|
|
|
|
|
let srv_name_raw: String = if is_srv { json_get(block, "name") } else { "" }
|
|
|
|
|
let srv_name: String = if is_srv && str_eq(srv_name_raw, "") { "server_tool" } else { srv_name_raw }
|
|
|
|
|
let srv_log = if is_srv {
|
|
|
|
|
if str_eq(srv_log, "") { "\"" + srv_name + "\"" } else { srv_log + ",\"" + srv_name + "\"" }
|
|
|
|
|
} else { srv_log }
|
|
|
|
|
// Capture first tool_use block only
|
|
|
|
|
let is_new_tool: Bool = str_eq(btype, "tool_use") && !has_tool
|
|
|
|
|
let has_tool = if is_new_tool { true } else { has_tool }
|
|
|
|
@@ -2291,6 +2620,24 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
// A real tool turn that targets a tool the soul cannot run in-process is a
|
|
|
|
|
// CLIENT bridge: suspend the loop and hand the tool to the client.
|
|
|
|
|
let is_tool_turn: Bool = str_eq(stop_reason, "tool_use") && has_tool
|
|
|
|
|
|
|
|
|
|
// pause_turn — REQUIRED for server-side web_search, not optional.
|
|
|
|
|
// Anthropic's server-side tool loop has its own iteration limit. When it hits it
|
|
|
|
|
// mid-answer the turn comes back with stop_reason "pause_turn" and only the text
|
|
|
|
|
// written SO FAR. Per Anthropic's contract the caller must re-send with the
|
|
|
|
|
// assistant content appended and NO tool_result, and the server resumes where it
|
|
|
|
|
// left off. Skip this and the user silently gets a truncated answer — no error,
|
|
|
|
|
// no warning, just a sentence that stops. (The bug that ate Tim's report,
|
|
|
|
|
// 2026-07-11.) Nothing else in this engine handles it: "pause_turn" appears
|
|
|
|
|
// nowhere else in the .el sources or in the shipped binary.
|
|
|
|
|
let is_pause: Bool = str_eq(stop_reason, "pause_turn")
|
|
|
|
|
// Unknown stop reasons (future API drift): finish honestly and log loudly rather
|
|
|
|
|
// than treating an unrecognised terminal state as a completed answer.
|
|
|
|
|
if !str_eq(stop_reason, "end_turn") && !str_eq(stop_reason, "tool_use") && !is_pause
|
|
|
|
|
&& !str_eq(stop_reason, "max_tokens") && !str_eq(stop_reason, "refusal")
|
|
|
|
|
&& !str_eq(stop_reason, "") {
|
|
|
|
|
println("[soul] DRIFT: unknown stop_reason from API: " + stop_reason)
|
|
|
|
|
}
|
|
|
|
|
// If the user previously chose "always allow" for this tool in this session,
|
|
|
|
|
// treat it like a builtin — run server-side via dispatch_tool and skip the
|
|
|
|
|
// bridge suspension entirely so the approval UI is never shown again.
|
|
|
|
@@ -2318,10 +2665,19 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
let tool_msg: String = "{\"type\":\"tool_result\",\"tool_use_id\":\"" + tool_id + "\",\"content\":\"" + tool_result + "\"}"
|
|
|
|
|
|
|
|
|
|
// Accumulate tool names for the tools_used log surfaced in the response.
|
|
|
|
|
// Gated on is_tool_turn, not has_tool: a round truncated by max_tokens can carry a
|
|
|
|
|
// half-written tool_use block that will never run, and logging it would claim work
|
|
|
|
|
// the soul did not do.
|
|
|
|
|
let tool_quoted: String = "\"" + tool_name + "\""
|
|
|
|
|
let tools_log = if has_tool {
|
|
|
|
|
let tools_log = if is_tool_turn {
|
|
|
|
|
if str_eq(tools_log, "") { tool_quoted } else { tools_log + "," + tool_quoted }
|
|
|
|
|
} else { tools_log }
|
|
|
|
|
// Merge in the server-executed tools (web_search) seen in this round's blocks.
|
|
|
|
|
let tools_log = if str_eq(srv_log, "") {
|
|
|
|
|
tools_log
|
|
|
|
|
} else {
|
|
|
|
|
if str_eq(tools_log, "") { srv_log } else { tools_log + "," + srv_log }
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// The assistant turn that requested the tool — needed verbatim on resume so the
|
|
|
|
|
// tool_use/tool_result pairing stays valid when the client posts its result.
|
|
|
|
@@ -2332,15 +2688,22 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
|
|
|
|
|
// Local built-in tool turn: append assistant + tool_result and keep looping.
|
|
|
|
|
let local_continue: Bool = is_tool_turn && !needs_bridge
|
|
|
|
|
// Pause resume: the assistant content goes back VERBATIM with NO tool_result —
|
|
|
|
|
// that is the whole contract. Anthropic's server resumes its own tool loop from
|
|
|
|
|
// there; sending a tool_result (or a "continue" user turn) breaks it.
|
|
|
|
|
let is_pause_resume: Bool = is_pause && !needs_bridge
|
|
|
|
|
let messages = if local_continue {
|
|
|
|
|
let inner2: String = str_slice(messages_with_assistant, 1, str_len(messages_with_assistant) - 1)
|
|
|
|
|
"[" + inner2 + ",{\"role\":\"user\",\"content\":[" + tool_msg + "]}]"
|
|
|
|
|
} else if is_pause_resume {
|
|
|
|
|
messages_with_assistant
|
|
|
|
|
} else { messages }
|
|
|
|
|
|
|
|
|
|
// Live progress ledger: one entry per round — the model's own narration
|
|
|
|
|
// (its pre-tool prose, previously discarded here) plus the tool it reached
|
|
|
|
|
// for. Clients poll /api/run-progress/<sid> to render these live.
|
|
|
|
|
if !str_eq(session_id, "") {
|
|
|
|
|
// Skipped on a version-fallback round: nothing happened that a poller should see.
|
|
|
|
|
if !str_eq(session_id, "") && !can_fallback {
|
|
|
|
|
let prog_key: String = "run_progress_" + session_id
|
|
|
|
|
let prog_prev: String = state_get(prog_key)
|
|
|
|
|
let prog_snip: String = if str_len(text_out) > 280 { str_slice(text_out, 0, 280) } else { text_out }
|
|
|
|
@@ -2365,8 +2728,19 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
bridge_save(session_id, model, safe_sys, tools_json, messages_with_assistant, tools_log, pend_tool_id)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
let final_text = if !is_tool_turn { text_out } else { final_text }
|
|
|
|
|
let keep_going = if local_continue { keep_going } else { false }
|
|
|
|
|
// ACCUMULATE across pause/resume cycles instead of overwriting. A resumed turn
|
|
|
|
|
// CONTINUES the answer, it does not repeat it — overwriting here would throw away
|
|
|
|
|
// everything the model wrote before the pause, which is the same truncation the
|
|
|
|
|
// pause handling exists to prevent. A version-fallback round contributes nothing.
|
|
|
|
|
let final_text = if !is_tool_turn && !can_fallback { final_text + text_out } else { final_text }
|
|
|
|
|
// Output cap hit mid-action: the tool block is truncated and will NOT run. Say so
|
|
|
|
|
// instead of ending on silent almost-work.
|
|
|
|
|
let final_text = if str_eq(stop_reason, "max_tokens") && has_tool {
|
|
|
|
|
final_text + "\n\n[Output limit reached mid-action - the last planned action did not run. Ask me to continue to finish it.]"
|
|
|
|
|
} else { final_text }
|
|
|
|
|
// Keep looping for a local tool round, a server-side pause resume, or a one-shot
|
|
|
|
|
// web_search version fallback.
|
|
|
|
|
let keep_going = if local_continue || is_pause_resume || can_fallback { keep_going } else { false }
|
|
|
|
|
let iteration = iteration + 1
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
@@ -2390,9 +2764,9 @@ fn agentic_loop(session_id: String, model: String, safe_sys: String, tools_json:
|
|
|
|
|
// means the task was too complex for the agentic loop depth — surface it clearly
|
|
|
|
|
// so the caller/operator knows to increase the cap or break the task apart.
|
|
|
|
|
if str_eq(final_text, "") {
|
|
|
|
|
let hit_cap: Bool = iteration >= 8
|
|
|
|
|
let hit_cap: Bool = iteration >= 12
|
|
|
|
|
let err_msg: String = if hit_cap {
|
|
|
|
|
"agentic loop hit the 8-iteration cap without producing a final reply - task may be too complex or a tool call is looping"
|
|
|
|
|
"agentic loop hit the 12-iteration cap without producing a final reply - task may be too complex or a tool call is looping"
|
|
|
|
|
} else {
|
|
|
|
|
"no response"
|
|
|
|
|
}
|
|
|
|
|