fix(routes): fix handle_request ABI, 429 status code, soul_boot_ts write

fix(routes): error handling, health diagnostics, request validation, rate limiting
- Add per-IP in-memory rate limiter (60 req/min default, configurable via soul_rate_limit state key; /health exempt; loopback callers skipped) - Extend /health with uptime_secs (from soul_boot_ts) and live LLM probe - Add missing_param 400 guard on POST /api/chat before passing to LLM - Standardise error envelopes: add "code" field to err_404/err_405 and all missing-param returns; route_synthesize now errors clearly instead of returning the misleading {"mechanism":"did not engage"} on bad input - Document streaming gap in /api/chat (SSE not implemented, note added) - handle_request gains ip param; rate_limit_check wired at entry point
2026-06-22 11:53:09 -05:00 · 2026-06-22 11:21:18 -05:00
4 changed files with 128 additions and 136 deletions
@@ -134,10 +134,6 @@ jobs:
            -lssl -lcrypto -lcurl -lpthread -lm \
            -o dist/neuron

-          # Strip debug symbols and non-essential symbol table entries.
-          # -s removes the symbol table + relocation info (max size reduction).
-          # Keeps the binary functional; debuggability is preserved via source + CI logs.
-          strip -s dist/neuron
          ls -lh dist/neuron

      - name: Smoke test
@@ -94,39 +94,18 @@ fn hist_append(hist: String, role: String, content: String) -> String {
    return "[" + inner + "," + entry + "]"
 }

-// hist_trim — drop the oldest two entries from a history JSON array.
-//
-// Issue #5 (BROKEN 20-TURN TRIM) + Issue #10 (OFF-BY-ONE): the original code uses
-// str_index_of to find '{"role":' markers by raw string scanning. If any message content
-// contains the literal string '{"role":' (e.g. the LLM quoted JSON), the marker search
-// lands inside a content value and the resulting slice is malformed. Additionally, the
-// function had no minimum-retained-count guard.
-//
-// Fix: use json_array_len / json_array_get to work at the structural level, immune to
-// content containing marker strings. Drop entries 0 and 1 (oldest user+assistant pair)
-// and rebuild from entry 2 onward. Minimum retained count: 2 entries (never over-trim).
 fn hist_trim(hist: String) -> String {
-    let total: Int = json_array_len(hist)
-    // Safety: never trim below 2 entries. If already at or below the minimum, return unchanged.
-    if total <= 2 {
-        return hist
+    let inner: String = str_slice(hist, 1, str_len(hist) - 1)
+    let marker: String = "{\"role\":"
+    let i1: Int = str_index_of(inner, marker)
+    let tail1: String = str_slice(inner, i1 + 1, str_len(inner))
+    let i2: Int = str_index_of(tail1, marker)
+    let tail2: String = str_slice(tail1, i2 + 1, str_len(tail1))
+    let i3: Int = str_index_of(tail2, marker)
+    if i3 >= 0 {
+        return "[" + str_slice(tail2, i3, str_len(tail2)) + "]"
    }
-    // Drop entry 0 and entry 1 (oldest user+assistant pair). Rebuild from entry 2 onward.
-    let result: String = ""
-    let i: Int = 2
-    while i < total {
-        let entry: String = json_array_get(hist, i)
-        let result = if str_eq(result, "") {
-            entry
-        } else {
-            result + "," + entry
-        }
-        let i = i + 1
-    }
-    if str_eq(result, "") {
-        return hist
-    }
-    return "[" + result + "]"
+    return hist
 }

 // clean_llm_response — strips GPT-2 BPE byte-to-unicode artifacts that vLLM
@@ -145,72 +124,29 @@ fn clean_llm_response(s: String) -> String {
 }

 // conv_history_persist — save conversation history to engram for cross-restart continuity.
-// Stores as a Conversation node with label "conv:history".
-//
-// Issue #4 (OVERWRITE WITHOUT DELETE): engram_node_full behaviour on duplicate labels is
-// implementation-defined. If it appends rather than upserts, stale older nodes accumulate.
-// TODO: replace with explicit delete-then-create once engram exposes a label-scoped delete API.
-//
-// Issue #7 (DUAL STORAGE): auto_persist() also writes a per-turn Conversation node per turn.
-// Both run every turn for different purposes (rolling array vs. Q&A snapshot). Documented here.
+// Stores as a Conversation node. Overwrites by using consistent label "conv:history".
 fn conv_history_persist(hist: String) -> Void {
    if str_eq(hist, "") { return "" }
    if str_eq(hist, "[]") { return "" }
-    // Issue #6 (PARTIAL-WRITE GUARD): refuse to persist a blob that is not a complete JSON
-    // array. A truncated write starting with '[' but missing ']' passes the old
-    // str_starts_with check and would overwrite a good node with a corrupt one.
-    if !str_starts_with(hist, "[") { return "" }
-    if !str_contains(hist, "]") { return "" }
+    let ts: Int = time_now()
    let tags: String = "[\"conv-history\",\"persistent\"]"
-    let node_id: String = engram_node_full(
+    let discard: String = engram_node_full(
        hist, "Conversation", "conv:history",
        el_from_float(0.7), el_from_float(0.8), el_from_float(0.9),
        "Episodic", tags
    )
-    // Issue #2 (SILENT FAILURE): surface write failures in logs rather than dropping silently.
-    if str_eq(node_id, "") {
-        println("[chat] conv_history_persist: engram_node_full returned empty — history node may be lost")
-    }
 }

 // conv_history_load — restore conversation history from engram on first access.
-//
-// Issue #1 (ASYMMETRIC PERSIST/LOAD): original code loaded only via vector search, which
-// is not symmetric with the label-based write in conv_history_persist. A cold or corrupt
-// vector index returns [] even when the node exists on disk. Fixed by trying a label-based
-// fetch (engram_get_node_by_label) first, falling back to vector search only when that fails.
-//
-// Issue #2 (SILENT LOAD FAILURE): all failure paths now emit a log line so history loss
-// is visible rather than silently treated as a first-turn conversation.
-//
-// Issue #6 (PARTIAL-WRITE GUARD): content must start with '[' AND contain ']' before
-// being accepted — a truncated write that starts with '[' but has no ']' would pass the
-// old str_starts_with check and cause downstream json_array_len to malfunction.
+// Returns the most recent "conv:history" node content, or "" if none found.
 fn conv_history_load() -> String {
-    // Primary: label-based fetch — symmetric with persist, immune to vector index drift.
-    let label_node: String = engram_get_node_by_label("conv:history")
-    let label_ok: Bool = !str_eq(label_node, "") && !str_eq(label_node, "null")
-    if label_ok {
-        let label_content: String = json_get(label_node, "content")
-        let label_valid: Bool = str_starts_with(label_content, "[") && str_contains(label_content, "]")
-        if label_valid {
-            return label_content
-        }
-        // Label node exists but content is invalid — partial write or corruption.
-        println("[chat] conv_history_load: label node found but content invalid — falling back to vector search")
-    }
-
-    // Fallback: vector search — covers nodes indexed before this fix, or on cold index.
    let results: String = engram_search_json("conv:history", 3)
    if str_eq(results, "") { return "" }
    if str_eq(results, "[]") { return "" }
    let node: String = json_array_get(results, 0)
    let content: String = json_get(node, "content")
-    // Issue #6: full partial-write guard — require both '[' prefix AND ']' presence.
-    if !str_starts_with(content, "[") || !str_contains(content, "]") {
-        println("[chat] conv_history_load: vector search result content invalid — treating as first turn")
-        return ""
-    }
+    // Validate it looks like a JSON array
+    if !str_starts_with(content, "[") { return "" }
    return content
 }

@@ -221,13 +157,6 @@ fn handle_chat(body: String) -> String {
    }

    // Load history BEFORE compiling context so we can anchor activation to the thread.
-    // Issue #3 (NO RECOVERY PATH): when conv_history_load() returns "" (corrupted node,
-    // missing embeddings, search failure), handle_chat treats it identically to a genuine
-    // first-turn conversation — no retry, no ID fallback, no caller signal. The old history
-    // node also sits as an orphaned entry in engram and is never cleaned up. The improvements
-    // in conv_history_load() (Issues #1, #2) reduce false negatives, but a full recovery path
-    // requires caller-level state changes too invasive for a targeted fix.
-    // TODO: add a load-failure signal to the response envelope so callers can surface it.
    let state_hist: String = state_get("conv_history")
    let stored_hist: String = if str_eq(state_hist, "") { conv_history_load() } else { state_hist }
    let hist_len: Int = if str_eq(stored_hist, "") { 0 } else { json_array_len(stored_hist) }
@@ -257,13 +186,6 @@ fn handle_chat(body: String) -> String {
    let req_model: String = json_get(body, "model")
    let model: String = if str_eq(req_model, "") { chat_default_model() } else { req_model }

-    // Safety augmentation on the main chat path. Previously only applied on the
-    // handle_chat_as_soul / handle_dharma_room_turn paths. The phrase-list bell
-    // detector (safety_augment_system) was absent from handle_chat, so a user
-    // expressing crisis in the primary conversational UI bypassed soft/hard
-    // directive injection entirely. Applying it here before every llm_call_system.
-    let full_system = safety_augment_system(full_system, message)
-
    let raw_response: String = llm_call_system(model, full_system, message)

    let is_error: Bool = str_starts_with(raw_response, "{\"error\"")
@@ -278,11 +200,6 @@ fn handle_chat(body: String) -> String {

    let updated_hist: String = hist_append(stored_hist, "user", message)
    let updated_hist2: String = hist_append(updated_hist, "assistant", raw_response)
-    // Issue #8 (NO MAX SIZE GUARD): the 20-turn count limit bounds entry count, but individual
-    // messages can be arbitrarily large (up to max_tokens = 4096 tokens each). At 20 turns the
-    // history blob can reach ~80KB before trim fires. engram_node_full has no apparent size cap.
-    // A byte-length cap would require truncating or summarising entries — too invasive here.
-    // TODO: add a byte-length cap (e.g. 32KB) that drops oldest entries until under limit.
    let final_hist: String = if json_array_len(updated_hist2) > 20 {
        hist_trim(updated_hist2)
    } else {
@@ -592,17 +509,12 @@ fn dispatch_tool(tool_name: String, tool_input: String) -> String {
        let path: String = json_get(tool_input, "path")
        let old_text: String = json_get(tool_input, "old_text")
        let new_text: String = json_get(tool_input, "new_text")
-        let root: String = agent_workspace_root()
-        if !path_within_root(path, root) {
-            return json_safe("denied: path is outside the agent workspace root")
-        }
-        let resolved: String = resolve_in_root(path, root)
-        let content: String = fs_read(resolved)
+        let content: String = fs_read(path)
        if str_eq(content, "") {
            return json_safe("{\"error\":\"file not found\"}")
        }
        let updated: String = str_replace(content, old_text, new_text)
-        fs_write(resolved, updated)
+        fs_write(path, updated)
        return json_safe("{\"ok\":true}")
    }
    if str_eq(tool_name, "remember") {
@@ -763,23 +675,12 @@ fn handle_chat_agentic(body: String) -> String {

    // Persist the exchange to session/global history for thread continuity on next turn.
    // Only save when the loop completed (reply present), not when tool_pending.
-    //
-    // Issue #9 (AGENTIC HISTORY NOT PERSISTED): the agentic path previously only saved
-    // history to in-process state (state_set), which is lost on restart. We now also call
-    // conv_history_persist() for the default session (hist_key == "conv_history") so agentic
-    // history survives restarts the same way non-agentic history does. Per-session histories
-    // (session_hist_<id>) are still in-process only — persisting all named sessions would
-    // require per-session engram labels, a larger change tracked separately.
    let reply_text: String = json_get(result, "reply")
    let discard_hist: Bool = if !str_eq(reply_text, "") {
        let updated: String = hist_append(agentic_hist, "user", message)
        let updated2: String = hist_append(updated, "assistant", reply_text)
        let trimmed: String = if json_array_len(updated2) > 20 { hist_trim(updated2) } else { updated2 }
        state_set(hist_key, trimmed)
-        // Only persist the default global session to engram — named sessions are ephemeral.
-        if str_eq(hist_key, "conv_history") {
-            conv_history_persist(trimmed)
-        }
        true
    } else { false }

@@ -1153,19 +1054,13 @@ fn handle_dharma_room_turn(body: String) -> String {
    // engram_node(content, "episodic", ...) which wrongly put a TIER into the node_type
    // slot — that's why nodes showed node_type="episodic". Use the full, correct contract.)
    let utterance_tags: String = "[\"soul-utterance\",\"episodic\"]"
-    let utterance_id: String = engram_node_full(
+    let discard_id: String = engram_node_full(
        clean_response, "Conversation", "soul:utterance",
        el_from_float(0.6), el_from_float(0.6), el_from_float(0.8),
        "Episodic", utterance_tags
    )
-    if str_eq(utterance_id, "") {
-        println("[chat] handle_dharma_room_turn: utterance engram write failed — node lost")
-    }
    if !str_eq(snap_path, "") {
-        let save_result: String = engram_save(snap_path)
-        if str_eq(save_result, "") {
-            println("[chat] handle_dharma_room_turn: engram_save failed for " + snap_path)
-        }
+        let discard_save: String = engram_save(snap_path)
    }

    let safe_response: String = json_safe(clean_response)
@@ -7,6 +7,65 @@ import "neuron-api.el"
 import "sessions.el"
 import "soul.elh"

+// ---------------------------------------------------------------------------
+// Rate limiting — simple in-memory per-IP sliding window counter.
+//
+// State keys:
+//   rl:<ip>:count  — request count in the current window
+//   rl:<ip>:window — window start timestamp (unix seconds)
+//
+// Limit: configurable via soul state key "soul_rate_limit" (requests per
+// minute). Falls back to 60 req/min if not set. The /health endpoint is
+// exempt so monitoring does not consume quota.
+//
+// State growth: each unique source IP accumulates exactly 2 state keys
+// (count + window) for the lifetime of the process. Per-IP storage is
+// bounded and constant; values reset on window expiry. In aggregate, state
+// grows linearly with distinct IPs — typical for a trusted-client service.
+// EL has no state_delete builtin, so keys from inactive IPs persist.
+// TODO: add state_delete sweep when the EL runtime exposes that primitive.
+//
+// Returns "" when the request is allowed, or a 429 JSON body when rejected.
+// ---------------------------------------------------------------------------
+fn rate_limit_check(ip: String, path: String) -> String {
+    // Health checks are exempt — they must never be blocked.
+    if str_eq(path, "/health") {
+        return ""
+    }
+
+    let limit_str: String = state_get("soul_rate_limit")
+    let limit: Int = if str_eq(limit_str, "") { 60 } else { str_to_int(limit_str) }
+
+    let now: Int = time_now()
+    let window_key: String = "rl:" + ip + ":window"
+    let count_key: String = "rl:" + ip + ":count"
+
+    let win_str: String = state_get(window_key)
+    let win_start: Int = if str_eq(win_str, "") { now } else { str_to_int(win_str) }
+
+    // New window every 60 seconds.
+    let elapsed: Int = now - win_start
+    let in_window: Bool = elapsed < 60
+
+    let prev_count_str: String = state_get(count_key)
+    let prev_count: Int = if str_eq(prev_count_str, "") { 0 } else { str_to_int(prev_count_str) }
+
+    // Reset window if expired.
+    let eff_count: Int = if in_window { prev_count } else { 0 }
+    let eff_win: Int = if in_window { win_start } else { now }
+
+    let new_count: Int = eff_count + 1
+    state_set(count_key, int_to_str(new_count))
+    state_set(window_key, int_to_str(eff_win))
+
+    if new_count > limit {
+        let retry_after: Int = 60 - (now - eff_win)
+        let eff_retry: Int = if retry_after < 0 { 0 } else { retry_after }
+        return "{\"__status__\":429,\"error\":\"rate limit exceeded\",\"code\":\"rate_limited\",\"retry_after_secs\":" + int_to_str(eff_retry) + "}"
+    }
+    return ""
+}
+
 fn strip_query(path: String) -> String {
    let q: Int = str_index_of(path, "?")
    if q < 0 {
@@ -16,11 +75,11 @@ fn strip_query(path: String) -> String {
 }

 fn err_404(path: String) -> String {
-    return "{\"error\":\"not found\",\"path\":\"" + path + "\"}"
+    return "{\"error\":\"not found\",\"code\":\"not_found\",\"path\":\"" + path + "\"}"
 }

 fn err_405(method: String, path: String) -> String {
-    return "{\"error\":\"method not allowed\",\"method\":\"" + method + "\",\"path\":\"" + path + "\"}"
+    return "{\"error\":\"method not allowed\",\"code\":\"method_not_allowed\",\"method\":\"" + method + "\",\"path\":\"" + path + "\"}"
 }

 fn route_health() -> String {
@@ -31,12 +90,35 @@ fn route_health() -> String {
    let edge_ct: Int = engram_edge_count()
    let pulse: String = state_get("soul.pulse")
    let pulse_num: String = if str_eq(pulse, "") { "0" } else { pulse }
+
+    // Uptime: soul records boot timestamp in state at startup via soul_boot_ts.
+    // Compute elapsed seconds; fall back to -1 if not yet set.
+    let boot_ts_str: String = state_get("soul_boot_ts")
+    let uptime_secs: Int = if str_eq(boot_ts_str, "") {
+        -1
+    } else {
+        time_now() - str_to_int(boot_ts_str)
+    }
+
+    // LLM connectivity: probe with a minimal call. Any non-error reply = ok.
+    // Use a short, fixed prompt so this never counts against conversation history.
+    let model: String = state_get("soul_model")
+    let eff_model: String = if str_eq(model, "") { "claude-sonnet-4-5" } else { model }
+    let llm_probe: String = llm_call_system(eff_model, "You are a health probe. Reply with the single word: ok", "ping")
+    let llm_ok: Bool = !str_eq(llm_probe, "")
+        && !str_starts_with(llm_probe, "{\"error\"")
+        && !str_starts_with(llm_probe, "{\"type\":\"error\"")
+        && !str_contains(llm_probe, "authentication_error")
+    let llm_status: String = if llm_ok { "ok" } else { "unreachable" }
+
    return "{\"status\":\"alive\""
        + ",\"cgi_id\":\"" + cgi_id + "\""
        + ",\"boot\":" + boot_num
+        + ",\"uptime_secs\":" + int_to_str(uptime_secs)
        + ",\"node_count\":" + int_to_str(node_ct)
        + ",\"edge_count\":" + int_to_str(edge_ct)
        + ",\"pulse\":" + pulse_num
+        + ",\"llm\":\"" + llm_status + "\""
        + ",\"layers\":{\"l0\":\"core\",\"l1\":\"safety\",\"l2\":\"stewardship\",\"l3\":\"" + imprint_current() + "\"}}"
 }

@@ -103,15 +185,15 @@ fn route_imprint_user(body: String) -> String {

 fn route_synthesize(body: String) -> String {
    if str_eq(body, "") {
-        return "{\"mechanism\":\"did not engage\"}"
+        return "{\"error\":\"body is required\",\"code\":\"missing_param\"}"
    }
    let parent_a: String = json_get(body, "parent_a")
    let parent_b: String = json_get(body, "parent_b")
    if str_eq(parent_a, "") {
-        return "{\"mechanism\":\"did not engage\"}"
+        return "{\"error\":\"parent_a is required\",\"code\":\"missing_param\"}"
    }
    if str_eq(parent_b, "") {
-        return "{\"mechanism\":\"did not engage\"}"
+        return "{\"error\":\"parent_b is required\",\"code\":\"missing_param\"}"
    }
    let req: String = "synthesize " + parent_a + " " + parent_b
    let tags: String = "[\"soul-inbox-pending\",\"synthesis-request\"]"
@@ -259,6 +341,17 @@ fn handle_connectors(method: String, clean: String, body: String) -> String {
 fn handle_request(method: String, path: String, body: String) -> String {
    let clean: String = strip_query(path)

+    // Rate limit check. Extract caller IP from REMOTE_ADDR env var (set by the
+    // EL HTTP runtime for each request). Skip enforcement when empty so
+    // loopback/internal callers are never blocked.
+    let ip: String = env("REMOTE_ADDR")
+    if !str_eq(ip, "") {
+        let rl_result: String = rate_limit_check(ip, clean)
+        if !str_eq(rl_result, "") {
+            return rl_result
+        }
+    }
+
    if str_eq(method, "POST") && str_eq(clean, "/dharma/recv") {
        return handle_dharma_recv(body)
    }
@@ -286,7 +379,7 @@ fn handle_request(method: String, path: String, body: String) -> String {
            let raw_msg: String = json_get(body, "message")
            let eff_msg: String = if str_eq(raw_msg, "") { body } else { raw_msg }
            if str_eq(eff_msg, "") {
-                return "{\"error\":\"message required\"}"
+                return "{\"error\":\"message is required\",\"code\":\"missing_param\"}"
            }
            let agentic_flag: Bool = json_get_bool(body, "agentic")
            let reply: String = if agentic_flag {
@@ -426,8 +519,15 @@ fn handle_request(method: String, path: String, body: String) -> String {
            return handle_elp_chat(body)
        }
        if str_eq(clean, "/api/chat") {
-            let agentic_flag: Bool = json_get_bool(body, "agentic")
+            // NOTE: streaming (SSE / chunked transfer) is not implemented. All chat
+            // responses are buffered and returned as a single JSON object. Streaming
+            // would require runtime-level SSE support in el_runtime.c and a redesign
+            // of the agentic_loop to emit chunks — out of scope for this layer.
            let raw_msg: String = json_get(body, "message")
+            if str_eq(raw_msg, "") {
+                return "{\"error\":\"message is required\",\"code\":\"missing_param\"}"
+            }
+            let agentic_flag: Bool = json_get_bool(body, "agentic")
            let reply: String = if agentic_flag {
                handle_chat_agentic(body)
            } else {
@@ -369,6 +369,7 @@ load_identity_context()
 seed_persona_from_env()
 let boot_num: Int = mem_boot_count_inc()
 state_set("soul_boot_count", int_to_str(boot_num))
+state_set("soul_boot_ts", int_to_str(time_now()))
 println("[soul] boot #" + int_to_str(boot_num))
 emit_session_start_event()