Compare commits
12 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 027a573d89 | |||
| 43d0449904 | |||
| b842e82f77 | |||
| 98ccbd4704 | |||
| dba755dcec | |||
| 8f3a478771 | |||
| 9ea41eed78 | |||
| ff421d39f6 | |||
| 635f6febe4 | |||
| 62af5649fe | |||
| 710761e2d5 | |||
| 74520b8333 |
@@ -1,4 +1,4 @@
|
||||
// auto-generated by elc --emit-header — do not edit
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn chat_default_model() -> String
|
||||
extern fn engram_numeric_valid(s: String) -> Bool
|
||||
extern fn parse_float_x100(s: String) -> Int
|
||||
@@ -16,18 +16,35 @@ extern fn engram_nodes_merge(a: String, b: String) -> String
|
||||
extern fn id_in_seen(node_id: String, seen: String) -> Bool
|
||||
extern fn add_to_seen(seen: String, node_id: String) -> String
|
||||
extern fn engram_extract_ids(nodes_json: String) -> String
|
||||
extern fn affective_node_ts(node_json: String) -> Int
|
||||
extern fn engram_compile(intent: String) -> String
|
||||
extern fn distill_transcript(transcript: String) -> String
|
||||
extern fn json_safe(s: String) -> String
|
||||
extern fn current_engine_note(model: String) -> String
|
||||
extern fn bounded_persona_floor() -> String
|
||||
extern fn operator_identity_block() -> String
|
||||
extern fn build_system_prompt(ctx: String, chat_mode: Bool) -> String
|
||||
extern fn hist_append(hist: String, role: String, content: String) -> String
|
||||
extern fn conv_hist_key(session_id: String) -> String
|
||||
extern fn conv_hist_label(session_id: String) -> String
|
||||
extern fn is_utility_request(body: String, session_id: String) -> Bool
|
||||
extern fn provenance_scan_urls(arr: String, acc: String) -> String
|
||||
extern fn provenance_add_sources(block: String, btype: String, has_cit: Bool, cit_raw: String, acc: String) -> String
|
||||
extern fn provenance_names(tools_used: String) -> String
|
||||
extern fn text_join_sep(accumulated: String, incoming: String, after_interruption: Bool) -> String
|
||||
extern fn receipt_rule() -> String
|
||||
extern fn receipt_strip(s: String) -> String
|
||||
extern fn tool_receipt(tools_used: String, sources: String) -> String
|
||||
extern fn hist_trim(hist: String) -> String
|
||||
extern fn hist_trim_with_bell_guard(hist: String) -> String
|
||||
extern fn clean_llm_response(s: String) -> String
|
||||
extern fn conv_history_persist(hist: String) -> Void
|
||||
extern fn conv_history_load() -> String
|
||||
extern fn conv_history_persist(session_id: String, hist: String) -> Void
|
||||
extern fn conv_history_load(session_id: String) -> String
|
||||
extern fn conv_history_record(session_id: String, user_msg: String, assistant_msg: String, receipt: String) -> Void
|
||||
extern fn conv_history_block(session_id: String) -> String
|
||||
extern fn layered_generate(prompt: String, imprint_id: String, session_id: String) -> String
|
||||
extern fn session_preload_bullets(nodes: String, max_bullets: Int, snip_len: Int) -> String
|
||||
extern fn affective_context_prefix() -> String
|
||||
extern fn handle_chat(body: String) -> String
|
||||
extern fn handle_see(body: String) -> String
|
||||
extern fn studio_tools_json() -> String
|
||||
@@ -37,6 +54,8 @@ extern fn llm_wire_format() -> String
|
||||
extern fn json_escape(s: String) -> String
|
||||
extern fn openai_chat_complete(model: String, base_url: String, api_key: String, safe_sys: String, messages_json: String) -> String
|
||||
extern fn agentic_tools_literal() -> String
|
||||
extern fn web_search_tool_json() -> String
|
||||
extern fn strip_client_web_search(tools_inner: String) -> String
|
||||
extern fn agentic_tools_with_web() -> String
|
||||
extern fn connector_tools_json() -> String
|
||||
extern fn agentic_tools_all() -> String
|
||||
@@ -46,6 +65,10 @@ extern fn call_neuron_mcp(tool_name: String, args: String) -> String
|
||||
extern fn agent_workspace_root() -> String
|
||||
extern fn path_within_root(path: String, root: String) -> Bool
|
||||
extern fn resolve_in_root(path: String, root: String) -> String
|
||||
extern fn run_command_is_readonly(cmd: String) -> Bool
|
||||
extern fn cmd_abs_escape_at(cmd: String, root: String, needle: String) -> Bool
|
||||
extern fn run_command_guard(cmd: String, root: String) -> String
|
||||
extern fn classify_tool_risk(tool_name: String, tool_input: String) -> String
|
||||
extern fn dispatch_tool(tool_name: String, tool_input: String) -> String
|
||||
extern fn is_builtin_tool(tool_name: String) -> Bool
|
||||
extern fn next_bridge_id() -> String
|
||||
|
||||
+17
-3
@@ -5,6 +5,15 @@ el_val_t add_punct(el_val_t s, el_val_t intent);
|
||||
el_val_t add_to_seen(el_val_t seen, el_val_t node_id);
|
||||
el_val_t aff_try_slot(el_val_t slot_json, el_val_t aff_7d_ts, el_val_t acc_key);
|
||||
el_val_t affective_context_prefix(void);
|
||||
el_val_t is_utility_request(el_val_t body, el_val_t session_id);
|
||||
el_val_t operator_identity_block(void);
|
||||
el_val_t provenance_add_sources(el_val_t block, el_val_t btype, el_val_t has_cit, el_val_t cit_raw, el_val_t acc);
|
||||
el_val_t provenance_names(el_val_t tools_used);
|
||||
el_val_t provenance_scan_urls(el_val_t arr, el_val_t acc);
|
||||
el_val_t text_join_sep(el_val_t accumulated, el_val_t incoming, el_val_t after_interruption);
|
||||
el_val_t receipt_rule(void);
|
||||
el_val_t receipt_strip(el_val_t s);
|
||||
el_val_t tool_receipt(el_val_t tools_used, el_val_t sources);
|
||||
el_val_t agent_number(el_val_t agent);
|
||||
el_val_t agent_person(el_val_t agent);
|
||||
el_val_t agent_workspace_root(void);
|
||||
@@ -151,8 +160,12 @@ el_val_t cmd_abs_escape_at(el_val_t cmd, el_val_t root, el_val_t needle);
|
||||
el_val_t connectd_get(el_val_t suffix);
|
||||
el_val_t connectd_post(el_val_t suffix, el_val_t body);
|
||||
el_val_t connector_tools_json(void);
|
||||
el_val_t conv_history_load(void);
|
||||
el_val_t conv_history_persist(el_val_t hist);
|
||||
el_val_t conv_hist_key(el_val_t session_id);
|
||||
el_val_t conv_hist_label(el_val_t session_id);
|
||||
el_val_t conv_history_block(el_val_t session_id);
|
||||
el_val_t conv_history_load(el_val_t session_id);
|
||||
el_val_t conv_history_persist(el_val_t session_id, el_val_t hist);
|
||||
el_val_t conv_history_record(el_val_t session_id, el_val_t user_msg, el_val_t assistant_msg, el_val_t receipt);
|
||||
el_val_t cop_article(el_val_t gender, el_val_t number, el_val_t definite);
|
||||
el_val_t cop_bwk_future(el_val_t prefix);
|
||||
el_val_t cop_bwk_perfect(el_val_t prefix);
|
||||
@@ -782,7 +795,8 @@ el_val_t lang_profile_uga(void);
|
||||
el_val_t lang_profile_zh(void);
|
||||
el_val_t lang_profile(el_val_t code, el_val_t word_order, el_val_t morph_type, el_val_t has_case, el_val_t has_gender, el_val_t script_dir, el_val_t agreement, el_val_t null_subject);
|
||||
el_val_t lang_word_order(el_val_t profile);
|
||||
el_val_t layered_cycle(el_val_t raw_input);
|
||||
el_val_t layered_cycle(el_val_t raw_input, el_val_t session_id, el_val_t utility);
|
||||
el_val_t layered_generate(el_val_t prompt, el_val_t imprint_id, el_val_t session_id);
|
||||
el_val_t lex_class(el_val_t entry);
|
||||
el_val_t lex_form(el_val_t entry, el_val_t idx);
|
||||
el_val_t lex_pos(el_val_t entry);
|
||||
|
||||
+15
@@ -1,3 +1,18 @@
|
||||
// ╔══════════════════════════════════════════════════════════════════════════╗
|
||||
// ║ STALE BUNDLE — DO NOT BUILD. UNSAFE CHAT PATH. ║
|
||||
// ╚══════════════════════════════════════════════════════════════════════════╝
|
||||
// This concatenated bundle is a snapshot, not a source of truth, and it is stale in
|
||||
// a way that matters for safety: it wires /api/chat straight to handle_chat and
|
||||
// contains NO layered_cycle at all (verified: zero occurrences in the bundled code —
|
||||
// the only textual hit in this file is this banner). A binary built from
|
||||
// this file would run chat with no enforcing input gate (no safety_screen, no
|
||||
// hard-bell short-circuit) and no enforcing output gate (no safety_validate).
|
||||
//
|
||||
// Build from the .el sources via manifest.el (entry soul.el), or from dist/soul.c.
|
||||
// Nothing in the repo references this file. It is kept only as a historical artifact
|
||||
// and should be deleted once Will confirms nothing external depends on it.
|
||||
// (Flagged 2026-08-04 in _engine-websearch-20260804/SAFETY-STOP.md; banner added
|
||||
// 2026-08-05 with the plain-chat generation fix.)
|
||||
// language-profile.el - Language profile data and accessors.
|
||||
//
|
||||
// A language profile is a slot map ([String] key-value list) describing the
|
||||
|
||||
+1
-1
@@ -28170,7 +28170,7 @@ el_val_t agentic_loop(el_val_t session_id, el_val_t model, el_val_t safe_sys, el
|
||||
state_set(el_str_concat(EL_STR("run_progress_"), session_id), EL_STR(""));
|
||||
}
|
||||
while (keep_going && (iteration < 8)) {
|
||||
el_val_t req_body = el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"model\":\""), model), EL_STR("\"")), EL_STR(",\"max_tokens\":4096")), EL_STR(",\"system\":\"")), safe_sys), EL_STR("\"")), EL_STR(",\"tools\":")), tools_json), EL_STR(",\"messages\":")), messages), EL_STR("}"));
|
||||
el_val_t req_body = el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(el_str_concat(EL_STR("{\"model\":\""), model), EL_STR("\"")), EL_STR(",\"max_tokens\":4096,\"tool_choice\":{\"type\":\"auto\",\"disable_parallel_tool_use\":true}")), EL_STR(",\"system\":\"")), safe_sys), EL_STR("\"")), EL_STR(",\"tools\":")), tools_json), EL_STR(",\"messages\":")), messages), EL_STR("}"));
|
||||
el_val_t raw_resp = http_post_with_headers(api_url, req_body, h);
|
||||
el_val_t is_error = ((str_starts_with(raw_resp, EL_STR("{\"error\"")) || str_starts_with(raw_resp, EL_STR("{\"type\":\"error\""))) || str_contains(raw_resp, EL_STR("authentication_error")));
|
||||
if (is_error) {
|
||||
|
||||
@@ -15,6 +15,40 @@ fn flag_true(body: String, key: String) -> Bool {
|
||||
return json_get_bool(body, key) || json_get_int(body, key) > 0
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// plain_chat_envelope — the JSON response contract for a non-agentic ("Tools: Off")
|
||||
// chat turn. Every /api/chat dispatch that calls layered_cycle goes through here, so
|
||||
// the three call sites cannot drift apart.
|
||||
//
|
||||
// WHY THE ENVELOPE IS BUILT HERE AND NOT INSIDE layered_cycle:
|
||||
// layered_cycle returns the user-facing text AFTER safety_validate has acted on it.
|
||||
// Keeping the JSON out of the cycle means the output gate always sees raw model text
|
||||
// and never an escaped blob — there is nothing to unwrap and re-wrap on the crisis
|
||||
// path, which is exactly the failure mode that made wiring handle_chat unsafe.
|
||||
// Escaping is the last thing that happens, strictly after the gate.
|
||||
//
|
||||
// FIELDS: `reply` and `response` carry the same validated text. Both are required by
|
||||
// live clients — the desktop app reads `reply` first (DaemonClient.parseChatResponse),
|
||||
// while the CLI tools and the Telegram gateway read `response` (the gateway reads only
|
||||
// `response`). Emitting one would break the other.
|
||||
//
|
||||
// EMPTY MEANS FAILURE, NOT AN EMPTY ANSWER: a hard bell returns the fixed crisis
|
||||
// message and a soft bell is padded to non-empty by safety_validate, so the only way
|
||||
// an empty string leaves the cycle is a failed model call. It is reported as an error
|
||||
// rather than dressed up as a successful blank reply.
|
||||
// ---------------------------------------------------------------------------
|
||||
fn plain_chat_envelope(validated: String, model: String) -> String {
|
||||
if str_eq(validated, "") {
|
||||
return "{\"error\":\"llm unavailable\",\"reply\":\"\",\"response\":\"\",\"agentic\":false,\"tools_used\":[]}"
|
||||
}
|
||||
let safe: String = json_safe(validated)
|
||||
return "{\"reply\":\"" + safe + "\""
|
||||
+ ",\"response\":\"" + safe + "\""
|
||||
+ ",\"model\":\"" + json_safe(model) + "\""
|
||||
+ ",\"agentic\":false"
|
||||
+ ",\"tools_used\":[]}"
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Rate limiting — simple in-memory per-IP sliding window counter.
|
||||
//
|
||||
@@ -243,8 +277,14 @@ fn handle_dharma_recv(body: String) -> String {
|
||||
} else if agentic_flag {
|
||||
handle_chat_agentic(chat_body)
|
||||
} else {
|
||||
let screened_reply: String = layered_cycle(raw_msg)
|
||||
screened_reply
|
||||
// Non-agentic ("Tools: Off"): the full L1→L2→L3→L1 cycle, which now generates
|
||||
// at L3 instead of echoing. Envelope built outside the cycle — see
|
||||
// plain_chat_envelope.
|
||||
// FIX B/E1 (2026-08-05): the cycle is told which conversation it is in, and
|
||||
// whether this generation is conversation at all. Same two arguments at all
|
||||
// three dispatch sites.
|
||||
let screened_reply: String = layered_cycle(raw_msg, json_get(chat_body, "session_id"), is_utility_request(chat_body, json_get(chat_body, "session_id")))
|
||||
plain_chat_envelope(screened_reply, chat_default_model())
|
||||
}
|
||||
auto_persist(chat_body, reply)
|
||||
return reply
|
||||
@@ -416,8 +456,11 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
} else if agentic_flag {
|
||||
handle_chat_agentic(body)
|
||||
} else {
|
||||
let screened_reply: String = layered_cycle(eff_msg)
|
||||
screened_reply
|
||||
// Non-agentic ("Tools: Off") — same cycle and same envelope as POST.
|
||||
// FIX B/E1: same threading. A GET probe usually carries no session_id, which
|
||||
// resolves to the anonymous window — the documented behaviour for this door.
|
||||
let screened_reply: String = layered_cycle(eff_msg, json_get(body, "session_id"), is_utility_request(body, json_get(body, "session_id")))
|
||||
plain_chat_envelope(screened_reply, chat_default_model())
|
||||
}
|
||||
auto_persist(body, reply)
|
||||
return reply
|
||||
@@ -580,8 +623,13 @@ fn handle_request(method: String, path: String, body: String) -> String {
|
||||
} else if agentic_flag {
|
||||
handle_chat_agentic(body)
|
||||
} else {
|
||||
let screened_reply: String = layered_cycle(raw_msg)
|
||||
screened_reply
|
||||
// Non-agentic ("Tools: Off") — the app's DEFAULT mode (AgentMode.NEVER).
|
||||
// Full L1→L2→L3→L1 cycle with real generation at L3; envelope built
|
||||
// outside the cycle so safety_validate always sees raw text.
|
||||
// FIX B/E1: same threading. This is the app's main plain-chat door, so this
|
||||
// is the site that ends the blank stare in practice.
|
||||
let screened_reply: String = layered_cycle(raw_msg, json_get(body, "session_id"), is_utility_request(body, json_get(body, "session_id")))
|
||||
plain_chat_envelope(screened_reply, chat_default_model())
|
||||
}
|
||||
auto_persist(body, reply)
|
||||
return reply
|
||||
|
||||
Executable
+108
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env bash
|
||||
# run-el-test.sh — compile and run one El test program from tests/.
|
||||
#
|
||||
# WHY THIS EXISTS (2026-08-07, issue #129):
|
||||
# tests/ has held 14 test programs for months with no way to run them. CI does
|
||||
# not run them. The convention printed in their own headers
|
||||
# (`elc soul.el && ./soul --test tests/x.el`) refers to a --test flag the El
|
||||
# runtime does not implement. So the tests were documentation, not gates —
|
||||
# which is how a P0 safety regression shipped with a test directory present.
|
||||
#
|
||||
# THE RECIPE, AND WHY IT IS THIS SHAPE:
|
||||
# Same discovery as gen-soul-amalgam.sh — `elc --target=c` emits only an extern
|
||||
# prototype for any module that has a .elh header next to it, and inlines the
|
||||
# module's bodies when it does not. A test that imports ../chat.el therefore
|
||||
# compiles to a 18 KB unit full of unresolved externs unless the headers are
|
||||
# out of the way. So: copy the sources into a scratch tree, delete every .elh
|
||||
# on the import chain, and compile the test there.
|
||||
#
|
||||
# Scratch copy on purpose: the worktree is shared with other terminals and
|
||||
# deleting headers in place would be a shared-tree mutation with no owner.
|
||||
#
|
||||
# EXIT STATUS IS THE GATE: non-zero if the binary fails to build, crashes, or if
|
||||
# its output contains a FAIL line or reports a non-zero failed count. Do not
|
||||
# "improve" this into something that only checks the exit code of the test
|
||||
# binary — these El tests print failures and still exit 0.
|
||||
#
|
||||
# usage: scripts/run-el-test.sh tests/test_history_amplification.el
|
||||
set -euo pipefail
|
||||
|
||||
TEST_REL="${1:?usage: run-el-test.sh tests/<test>.el}"
|
||||
SRC="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
TEST_NAME="$(basename "$TEST_REL" .el)"
|
||||
|
||||
ELC="${ELC:-$HOME/neuron-dev-stack/src/el/lang/dist/platform/elc}"
|
||||
[ -x "$ELC" ] || ELC="$HOME/el-sdk/elc"
|
||||
[ -x "$ELC" ] || { echo "[run-el-test] FAIL: no elc found (set ELC=)"; exit 1; }
|
||||
|
||||
RTC="${RTC:-$SRC/vendor/el-runtime/v1.0.0-20260501/el_runtime.c}"
|
||||
[ -f "$RTC" ] || RTC="$HOME/el-sdk/el_runtime.c"
|
||||
[ -f "$RTC" ] || { echo "[run-el-test] FAIL: no el_runtime.c found (set RTC=)"; exit 1; }
|
||||
RTDIR="$(dirname "$RTC")"
|
||||
|
||||
EL_REPO="${EL_REPO:-$HOME/Development/neuron-technologies/el}"
|
||||
SSL="${SSL_PREFIX:-/opt/homebrew/opt/openssl@3}"
|
||||
|
||||
GEN="$(mktemp -d "${TMPDIR:-/tmp}/el-test.XXXXXX")"
|
||||
trap 'rm -rf "$GEN"' EXIT
|
||||
|
||||
mkdir -p "$GEN/neuron/tests" "$GEN/foundation/el/elp/src"
|
||||
cp "$SRC"/*.el "$GEN/neuron/"
|
||||
cp "$SRC"/tests/*.el "$GEN/neuron/tests/" 2>/dev/null || true
|
||||
[ -d "$EL_REPO/elp/src" ] && cp "$EL_REPO"/elp/src/*.el "$GEN/foundation/el/elp/src/" 2>/dev/null || true
|
||||
# The whole recipe depends on there being no headers to short-circuit inlining.
|
||||
find "$GEN" -name '*.elh' -delete
|
||||
|
||||
echo "[run-el-test] compiling $TEST_REL"
|
||||
( cd "$GEN/neuron" && "$ELC" --target=c "tests/${TEST_NAME}.el" ) > "$GEN/${TEST_NAME}.c"
|
||||
|
||||
BODIES=$(grep -c '^el_val_t .*) {$' "$GEN/${TEST_NAME}.c" || true)
|
||||
echo "[run-el-test] $(wc -c < "$GEN/${TEST_NAME}.c" | tr -d ' ') bytes, ${BODIES} inlined function bodies"
|
||||
# A test that imports ../chat.el pulls in the bulk of the engine. A tiny body
|
||||
# count means an import was read from a header instead of inlined, and the test
|
||||
# would be exercising extern stubs rather than the real code.
|
||||
if [ "$BODIES" -lt 100 ]; then
|
||||
echo "[run-el-test] FAIL: only $BODIES inlined bodies — an import was not inlined"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
cc -O2 -DHAVE_CURL \
|
||||
-I"$RTDIR" -I"$SSL/include" -L"$SSL/lib" \
|
||||
"$GEN/${TEST_NAME}.c" "$RTC" \
|
||||
-lssl -lcrypto -lcurl -lpthread -lm \
|
||||
-o "$GEN/${TEST_NAME}" 2> "$GEN/cc.log" || {
|
||||
echo "[run-el-test] FAIL: compile error"; tail -30 "$GEN/cc.log"; exit 1; }
|
||||
|
||||
# arm64 pointer-truncation guard (cc-brain.sh's rule): an implicit declaration of
|
||||
# a runtime symbol truncates its returned pointer to 32 bits.
|
||||
if grep -E 'implicit.*(engram_|el_)' "$GEN/cc.log"; then
|
||||
echo "[run-el-test] FAIL: implicit declarations of runtime symbols"; exit 1; fi
|
||||
|
||||
# Throwaway HOME so a test can never read or write the live engram at ~/.neuron.
|
||||
TEST_HOME="$GEN/home"
|
||||
mkdir -p "$TEST_HOME"
|
||||
|
||||
echo "[run-el-test] running $TEST_NAME"
|
||||
set +e
|
||||
HOME="$TEST_HOME" NEURON_HOME="$TEST_HOME/.neuron" "$GEN/${TEST_NAME}" 2>&1 | tee "$GEN/out.txt"
|
||||
RC=${PIPESTATUS[0]}
|
||||
set -e
|
||||
|
||||
if [ "$RC" -ne 0 ]; then
|
||||
echo "[run-el-test] FAIL: $TEST_NAME exited $RC (crash or abort)"
|
||||
exit 1
|
||||
fi
|
||||
if grep -q " FAIL:" "$GEN/out.txt"; then
|
||||
echo "[run-el-test] FAIL: $TEST_NAME reported failing assertions"
|
||||
exit 1
|
||||
fi
|
||||
if grep -qE '[1-9][0-9]* failed' "$GEN/out.txt"; then
|
||||
echo "[run-el-test] FAIL: $TEST_NAME reported a non-zero failed count"
|
||||
exit 1
|
||||
fi
|
||||
if ! grep -q "PASS:" "$GEN/out.txt"; then
|
||||
echo "[run-el-test] FAIL: $TEST_NAME produced no assertions at all"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "[run-el-test] PASS: $TEST_NAME"
|
||||
Executable
+937
@@ -0,0 +1,937 @@
|
||||
#!/usr/bin/env python3
|
||||
"""state-key-audit.py — the analyzer behind scripts/verify-state-keys.sh.
|
||||
|
||||
Read that script's header for WHY this exists (issue #129). This file is the
|
||||
HOW: a small El reader that resolves the key expression at every state_get /
|
||||
state_set site, including keys that are computed.
|
||||
|
||||
WHAT IT PARSES
|
||||
El as this engine writes it: `fn f(a: T, b: T) -> T { ... }`, `let x: T = e`,
|
||||
`return e`, `if c { a } else { b }` as an expression, `+` concatenation,
|
||||
`"..."` with backslash escapes, `//` line comments. No block comments, no
|
||||
const/match/struct exist in this dialect (verified over the whole tree).
|
||||
|
||||
KEY PATTERNS — the only two things a key expression can resolve to
|
||||
EXACT "soul_model" the whole key is known
|
||||
PREFIX "session_hist_" a known head, then runtime text
|
||||
(plus UNRESOLVED, which is a report line and never a failure)
|
||||
|
||||
RESOLUTION — resolve_expr() returns a SET of patterns; unions are how branches,
|
||||
multiple returns, and multiple bindings of one name are represented.
|
||||
literal "k" -> {EXACT k}
|
||||
concat A + B -> fold left; all-static -> EXACT,
|
||||
static head + dynamic tail -> PREFIX
|
||||
if-expression if c {A} else {B} -> resolve(A) | resolve(B), except that
|
||||
str_eq(X,"") with X statically ""
|
||||
folds to the taken branch only
|
||||
call f(args) -> union over f's return expressions,
|
||||
with f's params bound to THIS call
|
||||
site's actual argument expressions
|
||||
local var let k = e; state_get(k)-> union over every `let k =` in the
|
||||
enclosing function
|
||||
parameter fn g(k) { state_get(k) }-> union over the argument at that
|
||||
position across every call site of g
|
||||
anything else json_get(...), env(...)-> UNRESOLVED
|
||||
Recursion is depth- and cycle-guarded; a guard trip yields UNRESOLVED, never a
|
||||
failure.
|
||||
|
||||
COVERAGE — a read is satisfied when some write can produce the same key:
|
||||
read EXACT k <- write EXACT k, or write PREFIX p where k starts with p
|
||||
read PREFIX p <- write EXACT k where k starts with p, or write PREFIX q
|
||||
where p and q are prefixes of each other
|
||||
Deliberately permissive at the boundaries: a gate that cries wolf gets deleted.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
MAX_DEPTH = 12
|
||||
|
||||
# ── patterns ────────────────────────────────────────────────────────────────
|
||||
EXACT = "exact"
|
||||
PREFIX = "prefix"
|
||||
|
||||
|
||||
def pat_exact(s):
|
||||
return (EXACT, s)
|
||||
|
||||
|
||||
def pat_prefix(s):
|
||||
# A prefix with no static text at all carries no information; that is the
|
||||
# UNRESOLVED case, not a pattern.
|
||||
return (PREFIX, s) if s else None
|
||||
|
||||
|
||||
def covers(write, read):
|
||||
"""Can a write of pattern `write` produce a key that `read` reads?
|
||||
|
||||
The prefix rule is DIRECTIONAL, and that direction is the whole point. A
|
||||
write namespace that is the same or BROADER than the read namespace covers
|
||||
it (write "rl:" covers read "rl:x"). A write namespace that is NARROWER does
|
||||
NOT (write "session_histv2_" does not cover read "session_hist_") — being
|
||||
permissive there re-opens the exact hole this gate exists to close: rename
|
||||
the producer, leave the readers, stay green. Verified with a control run
|
||||
that renames sessions.el's writer and leaves its four readers behind."""
|
||||
wk, wv = write
|
||||
rk, rv = read
|
||||
if rk == EXACT:
|
||||
return rv == wv if wk == EXACT else rv.startswith(wv)
|
||||
# read is a PREFIX: some key starting with rv is read
|
||||
if wk == EXACT:
|
||||
return wv.startswith(rv) # that one written key is in range
|
||||
return rv.startswith(wv) # write namespace same-or-broader
|
||||
|
||||
|
||||
# ── lexer ───────────────────────────────────────────────────────────────────
|
||||
TOK_STR, TOK_IDENT, TOK_PUNCT, TOK_NUM = "str", "ident", "punct", "num"
|
||||
IDENT_RE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
|
||||
NUM_RE = re.compile(r"[0-9]+(\.[0-9]+)?")
|
||||
|
||||
|
||||
class Tok:
|
||||
__slots__ = ("kind", "val", "line")
|
||||
|
||||
def __init__(self, kind, val, line):
|
||||
self.kind, self.val, self.line = kind, val, line
|
||||
|
||||
def __repr__(self):
|
||||
return "%s(%r)@%d" % (self.kind, self.val, self.line)
|
||||
|
||||
|
||||
def lex(src):
|
||||
toks, i, n, line = [], 0, len(src), 1
|
||||
while i < n:
|
||||
c = src[i]
|
||||
if c == "\n":
|
||||
line += 1
|
||||
i += 1
|
||||
continue
|
||||
if c in " \t\r":
|
||||
i += 1
|
||||
continue
|
||||
if c == "/" and i + 1 < n and src[i + 1] == "/":
|
||||
while i < n and src[i] != "\n":
|
||||
i += 1
|
||||
continue
|
||||
if c == '"':
|
||||
j, buf = i + 1, []
|
||||
while j < n:
|
||||
if src[j] == "\\" and j + 1 < n:
|
||||
esc = src[j + 1]
|
||||
buf.append({"n": "\n", "t": "\t", "r": "\r"}.get(esc, esc))
|
||||
j += 2
|
||||
continue
|
||||
if src[j] == '"':
|
||||
break
|
||||
if src[j] == "\n":
|
||||
line += 1
|
||||
buf.append(src[j])
|
||||
j += 1
|
||||
toks.append(Tok(TOK_STR, "".join(buf), line))
|
||||
i = j + 1
|
||||
continue
|
||||
m = IDENT_RE.match(src, i)
|
||||
if m:
|
||||
toks.append(Tok(TOK_IDENT, m.group(0), line))
|
||||
i = m.end()
|
||||
continue
|
||||
m = NUM_RE.match(src, i)
|
||||
if m:
|
||||
toks.append(Tok(TOK_NUM, m.group(0), line))
|
||||
i = m.end()
|
||||
continue
|
||||
toks.append(Tok(TOK_PUNCT, c, line))
|
||||
i += 1
|
||||
return toks
|
||||
|
||||
|
||||
def match_close(toks, i, open_ch, close_ch):
|
||||
"""toks[i] is open_ch; return index of its matching close_ch."""
|
||||
depth = 0
|
||||
while i < len(toks):
|
||||
if toks[i].kind == TOK_PUNCT:
|
||||
if toks[i].val == open_ch:
|
||||
depth += 1
|
||||
elif toks[i].val == close_ch:
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
return i
|
||||
i += 1
|
||||
return len(toks) - 1
|
||||
|
||||
|
||||
# ── program model ───────────────────────────────────────────────────────────
|
||||
class Func:
|
||||
def __init__(self, name, path, line, params, toks, start, end):
|
||||
self.name, self.path, self.line = name, path, line
|
||||
self.params = params # [param name]
|
||||
self.toks = toks # the whole file's token list
|
||||
self.start, self.end = start, end # body token range, exclusive of braces
|
||||
self.lets = None # name -> [expr token ranges], lazily built
|
||||
|
||||
|
||||
class Site:
|
||||
def __init__(self, kind, path, line, func, arg_range, text):
|
||||
self.kind = kind # "get" | "set"
|
||||
self.path, self.line = path, line
|
||||
self.func = func
|
||||
self.arg_range = arg_range
|
||||
self.text = text # source text of the key expression
|
||||
self.pats = set()
|
||||
self.unresolved = False
|
||||
self.literal = None # set when the key expression is a bare literal
|
||||
|
||||
|
||||
class Program:
|
||||
def __init__(self):
|
||||
self.files = {} # path -> toks
|
||||
self.funcs = {} # name -> [Func] (El allows no overloads, but be safe)
|
||||
self.toplevel = [] # [Func] one per file, params=[]
|
||||
self.sites = [] # [Site]
|
||||
self.calls = {} # callee name -> [(Func caller, [arg ranges])]
|
||||
|
||||
# -- loading ------------------------------------------------------------
|
||||
def load(self, path, rel):
|
||||
with open(path, "r", encoding="utf-8", errors="replace") as fh:
|
||||
src = fh.read()
|
||||
toks = lex(src)
|
||||
self.files[rel] = toks
|
||||
self._scan_funcs(rel, toks)
|
||||
|
||||
def _scan_funcs(self, rel, toks):
|
||||
covered = []
|
||||
i = 0
|
||||
while i < len(toks):
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and t.val == "fn" and i + 2 < len(toks) \
|
||||
and toks[i + 1].kind == TOK_IDENT and toks[i + 2].val == "(":
|
||||
name = toks[i + 1].val
|
||||
pclose = match_close(toks, i + 2, "(", ")")
|
||||
params = self._params(toks, i + 3, pclose)
|
||||
bopen = pclose + 1
|
||||
while bopen < len(toks) and toks[bopen].val != "{":
|
||||
bopen += 1
|
||||
bclose = match_close(toks, bopen, "{", "}")
|
||||
f = Func(name, rel, t.line, params, toks, bopen + 1, bclose)
|
||||
self.funcs.setdefault(name, []).append(f)
|
||||
covered.append((i, bclose))
|
||||
i = bclose + 1
|
||||
continue
|
||||
i += 1
|
||||
# everything outside a fn is the file's top-level "function"
|
||||
tl = Func("<toplevel:%s>" % rel, rel, 1, [], toks, 0, len(toks))
|
||||
tl.covered = covered
|
||||
self.toplevel.append(tl)
|
||||
|
||||
@staticmethod
|
||||
def _params(toks, i, end):
|
||||
"""`a: T, b: T` -> ['a','b'] (top-level commas only)."""
|
||||
names, depth, expect = [], 0, True
|
||||
while i < end:
|
||||
t = toks[i]
|
||||
if t.kind == TOK_PUNCT and t.val in "([{":
|
||||
depth += 1
|
||||
elif t.kind == TOK_PUNCT and t.val in ")]}":
|
||||
depth -= 1
|
||||
elif depth == 0 and t.kind == TOK_PUNCT and t.val == ",":
|
||||
expect = True
|
||||
elif depth == 0 and expect and t.kind == TOK_IDENT:
|
||||
names.append(t.val)
|
||||
expect = False
|
||||
i += 1
|
||||
return names
|
||||
|
||||
def func_at(self, rel, tok_index):
|
||||
for f in self.funcs_in(rel):
|
||||
if f.start <= tok_index < f.end:
|
||||
return f
|
||||
for f in self.toplevel:
|
||||
if f.path == rel:
|
||||
return f
|
||||
return None
|
||||
|
||||
def funcs_in(self, rel):
|
||||
for fl in self.funcs.values():
|
||||
for f in fl:
|
||||
if f.path == rel:
|
||||
yield f
|
||||
|
||||
# -- indexing -----------------------------------------------------------
|
||||
def index(self):
|
||||
for rel, toks in self.files.items():
|
||||
i = 0
|
||||
while i < len(toks):
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and i + 1 < len(toks) and toks[i + 1].val == "(" \
|
||||
and t.val not in KEYWORDS \
|
||||
and not (i > 0 and toks[i - 1].kind == TOK_IDENT
|
||||
and toks[i - 1].val == "fn"):
|
||||
# ^ the `fn f(a: T)` declaration is not a call site; counting
|
||||
# it as one makes every parameter resolve to its own name
|
||||
# and reports the whole function UNRESOLVED.
|
||||
close = match_close(toks, i + 1, "(", ")")
|
||||
args = split_args(toks, i + 2, close)
|
||||
self.calls.setdefault(t.val, []).append(
|
||||
(self.func_at(rel, i), args, rel, t.line))
|
||||
if t.val in ("state_get", "state_set") and args:
|
||||
self.sites.append(Site(
|
||||
"get" if t.val == "state_get" else "set",
|
||||
rel, t.line, self.func_at(rel, i), args[0],
|
||||
render(toks, *args[0])))
|
||||
i += 1
|
||||
|
||||
# -- resolution ---------------------------------------------------------
|
||||
def lets_of(self, f):
|
||||
if f.lets is not None:
|
||||
return f.lets
|
||||
f.lets = {}
|
||||
toks = f.toks
|
||||
skip = getattr(f, "covered", [])
|
||||
i = f.start
|
||||
while i < f.end:
|
||||
if any(a <= i <= b for a, b in skip):
|
||||
i = max(b for a, b in skip if a <= i <= b) + 1
|
||||
continue
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and t.val == "let" and i + 1 < f.end \
|
||||
and toks[i + 1].kind == TOK_IDENT:
|
||||
name = toks[i + 1].val
|
||||
j = i + 2
|
||||
if j < f.end and toks[j].val == ":": # skip the type
|
||||
while j < f.end and toks[j].val != "=":
|
||||
j += 1
|
||||
if j < f.end and toks[j].val == "=":
|
||||
s = j + 1
|
||||
e = stmt_end(toks, s, f.end)
|
||||
f.lets.setdefault(name, []).append((s, e))
|
||||
i = e
|
||||
continue
|
||||
i += 1
|
||||
return f.lets
|
||||
|
||||
def returns_of(self, ctx, depth=0, seen=None):
|
||||
"""The value expressions of a function, in the context it was CALLED in.
|
||||
|
||||
Context-sensitive on purpose. `conv_hist_key` is written as a guard:
|
||||
|
||||
if str_eq(session_id, "") { return "conv_history" }
|
||||
return "session_hist_" + session_id
|
||||
|
||||
Collecting both returns flat would make state_set(conv_hist_key("")) — the
|
||||
dead handle_chat() write — claim to produce the session_hist_ namespace
|
||||
too. That is a producer this engine does not actually have, and claiming
|
||||
it would let the gate stay green if sessions.el's real writer vanished:
|
||||
a masking hole in the exact namespace #129 lives in. So a guard whose
|
||||
condition folds is honoured, and the branch not taken is dropped."""
|
||||
out = []
|
||||
self._values(ctx.toks, ctx.start, ctx.end, ctx, depth,
|
||||
seen if seen is not None else set(), out)
|
||||
return out
|
||||
|
||||
def _values(self, toks, s, e, ctx, depth, seen, out):
|
||||
"""Append the value expressions of a statement sequence.
|
||||
Returns True when the sequence definitely returns (rest unreachable)."""
|
||||
if depth > MAX_DEPTH:
|
||||
return False
|
||||
i = s
|
||||
while i < e:
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and t.val == "return":
|
||||
j = stmt_end(toks, i + 1, e)
|
||||
if j > i + 1:
|
||||
out.append((i + 1, j))
|
||||
return True
|
||||
if t.kind == TOK_IDENT and t.val == "let":
|
||||
i = stmt_end(toks, i + 2, e)
|
||||
continue
|
||||
if t.kind == TOK_IDENT and t.val == "if":
|
||||
i = self._if_stmt(toks, i, e, ctx, depth, seen, out)
|
||||
if i is True:
|
||||
return True
|
||||
continue
|
||||
if t.kind == TOK_PUNCT and t.val in "([{":
|
||||
i = match_close(toks, i, t.val,
|
||||
{"(": ")", "[": "]", "{": "}"}[t.val]) + 1
|
||||
continue
|
||||
en = stmt_end(toks, i, e)
|
||||
if en <= i:
|
||||
i += 1
|
||||
continue
|
||||
if en >= e: # trailing expression = the value
|
||||
out.append((i, en))
|
||||
i = en
|
||||
return False
|
||||
|
||||
def _if_stmt(self, toks, i, e, ctx, depth, seen, out):
|
||||
"""Walk one if / else-if / else chain. Returns the next index, or True
|
||||
if the chain definitely returns on every reachable branch."""
|
||||
bopen = i + 1
|
||||
while bopen < e and toks[bopen].val != "{":
|
||||
bopen += 1
|
||||
if bopen >= e:
|
||||
return e
|
||||
bclose = match_close(toks, bopen, "{", "}")
|
||||
fold = self._fold_cond(toks, i + 1, bopen, ctx, depth, seen)
|
||||
|
||||
j = bclose + 1
|
||||
else_s = else_e = None
|
||||
if j < e and toks[j].kind == TOK_IDENT and toks[j].val == "else":
|
||||
if j + 1 < e and toks[j + 1].val == "{":
|
||||
ec = match_close(toks, j + 1, "{", "}")
|
||||
else_s, else_e = j + 2, ec
|
||||
j = ec + 1
|
||||
else: # `else if ...` — the rest of the chain
|
||||
else_s = j + 1
|
||||
else_e = stmt_end(toks, j + 1, e)
|
||||
j = else_e
|
||||
|
||||
then_ret = else_ret = False
|
||||
if fold is not False:
|
||||
then_ret = self._values(toks, bopen + 1, bclose, ctx, depth + 1, seen, out)
|
||||
if fold is not True and else_s is not None:
|
||||
else_ret = self._values(toks, else_s, else_e, ctx, depth + 1, seen, out)
|
||||
|
||||
if fold is True and then_ret:
|
||||
return True
|
||||
if fold is False and else_s is not None and else_ret:
|
||||
return True
|
||||
if fold is None and else_s is not None and then_ret and else_ret:
|
||||
return True
|
||||
return j
|
||||
|
||||
def resolve(self, rng, func, depth=0, seen=None):
|
||||
"""-> (set of patterns, unresolved_flag)"""
|
||||
if seen is None:
|
||||
seen = set()
|
||||
if depth > MAX_DEPTH:
|
||||
return set(), True
|
||||
return self._expr(func.toks, rng[0], rng[1], func, depth, seen)
|
||||
|
||||
# -- expression walker --------------------------------------------------
|
||||
def _expr(self, toks, s, e, func, depth, seen):
|
||||
parts, cur, d = [], s, 0
|
||||
i = s
|
||||
while i < e: # split on top-level '+'
|
||||
v = toks[i].val
|
||||
if toks[i].kind == TOK_PUNCT and v in "([{":
|
||||
d += 1
|
||||
elif toks[i].kind == TOK_PUNCT and v in ")]}":
|
||||
d -= 1
|
||||
elif d == 0 and toks[i].kind == TOK_PUNCT and v == "+" and i > s:
|
||||
parts.append((cur, i))
|
||||
cur = i + 1
|
||||
i += 1
|
||||
parts.append((cur, e))
|
||||
if len(parts) == 1:
|
||||
return self._primary(toks, s, e, func, depth, seen)
|
||||
|
||||
# concatenation: keep folding while every operand so far is EXACT
|
||||
head, unres = "", False
|
||||
static = True
|
||||
for (ps, pe) in parts:
|
||||
pats, u = self._primary(toks, ps, pe, func, depth, seen)
|
||||
exacts = {p[1] for p in pats if p[0] == EXACT}
|
||||
if static and len(exacts) == 1 and not u and len(pats) == 1:
|
||||
head += exacts.pop()
|
||||
continue
|
||||
if static and pats and all(p[0] == EXACT for p in pats) and len(pats) > 1:
|
||||
# a branchy static operand: keep the shared head only
|
||||
static = False
|
||||
head += os.path.commonprefix(sorted({p[1] for p in pats}))
|
||||
break
|
||||
static = False
|
||||
# first non-static operand: everything after it is runtime text
|
||||
if (ps, pe) == parts[0]:
|
||||
for p in pats:
|
||||
if p[0] == PREFIX:
|
||||
head = p[1]
|
||||
break
|
||||
if not head:
|
||||
unres = True
|
||||
break
|
||||
if static:
|
||||
return {pat_exact(head)}, False
|
||||
p = pat_prefix(head)
|
||||
return ({p} if p else set()), (unres or not p)
|
||||
|
||||
def _primary(self, toks, s, e, func, depth, seen):
|
||||
while s < e and toks[s].kind == TOK_PUNCT and toks[s].val == "(" \
|
||||
and match_close(toks, s, "(", ")") == e - 1:
|
||||
s, e = s + 1, e - 1
|
||||
if s >= e:
|
||||
return set(), True
|
||||
t = toks[s]
|
||||
|
||||
if t.kind == TOK_STR and e == s + 1:
|
||||
return {pat_exact(t.val)}, False
|
||||
|
||||
if t.kind == TOK_IDENT and t.val == "if":
|
||||
return self._if_expr(toks, s, e, func, depth, seen)
|
||||
|
||||
if t.kind == TOK_IDENT and s + 1 < e and toks[s + 1].val == "(":
|
||||
close = match_close(toks, s + 1, "(", ")")
|
||||
if close == e - 1:
|
||||
return self._call(toks, t.val, split_args(toks, s + 2, close),
|
||||
func, depth, seen)
|
||||
|
||||
if t.kind == TOK_IDENT and e == s + 1:
|
||||
return self._var(t.val, func, depth, seen)
|
||||
|
||||
return set(), True
|
||||
|
||||
def _if_expr(self, toks, s, e, func, depth, seen):
|
||||
bopen = s + 1
|
||||
while bopen < e and toks[bopen].val != "{":
|
||||
bopen += 1
|
||||
cond = (s + 1, bopen)
|
||||
bclose = match_close(toks, bopen, "{", "}")
|
||||
then_rng = block_tail(toks, bopen + 1, bclose) or (bopen + 1, bclose)
|
||||
|
||||
else_rng = None
|
||||
j = bclose + 1
|
||||
if j < e and toks[j].kind == TOK_IDENT and toks[j].val == "else":
|
||||
if j + 1 < e and toks[j + 1].val == "{":
|
||||
ec = match_close(toks, j + 1, "{", "}")
|
||||
else_rng = block_tail(toks, j + 2, ec) or (j + 2, ec)
|
||||
else:
|
||||
else_rng = (j + 1, e) # `else if ...`
|
||||
|
||||
taken = self._fold_cond(toks, cond[0], cond[1], func, depth, seen)
|
||||
rngs = []
|
||||
if taken is not False:
|
||||
rngs.append(then_rng)
|
||||
if taken is not True and else_rng:
|
||||
rngs.append(else_rng)
|
||||
|
||||
pats, unres = set(), False
|
||||
for r in rngs:
|
||||
p, u = self._expr(toks, r[0], r[1], func, depth + 1, seen)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
def _fold_cond(self, toks, s, e, func, depth, seen):
|
||||
"""Constant-fold `str_eq(X, "")` / `!str_eq(X, "")` so a helper called with
|
||||
a literal (conv_hist_key("")) yields only the branch it really takes.
|
||||
Returns True / False / None(unknown)."""
|
||||
neg = False
|
||||
if s < e and toks[s].kind == TOK_PUNCT and toks[s].val == "!":
|
||||
neg, s = True, s + 1
|
||||
if not (s < e and toks[s].kind == TOK_IDENT and toks[s].val == "str_eq"
|
||||
and s + 1 < e and toks[s + 1].val == "("):
|
||||
return None
|
||||
close = match_close(toks, s + 1, "(", ")")
|
||||
if close != e - 1:
|
||||
return None
|
||||
args = split_args(toks, s + 2, close)
|
||||
if len(args) != 2:
|
||||
return None
|
||||
va, ua = self._expr(toks, args[0][0], args[0][1], func, depth + 1, seen)
|
||||
vb, ub = self._expr(toks, args[1][0], args[1][1], func, depth + 1, seen)
|
||||
if ua or ub or len(va) != 1 or len(vb) != 1:
|
||||
return None
|
||||
(ka, sa), (kb, sb) = va.pop(), vb.pop()
|
||||
if ka != EXACT or kb != EXACT:
|
||||
return None
|
||||
r = (sa == sb)
|
||||
return (not r) if neg else r
|
||||
|
||||
def _call(self, toks, name, args, func, depth, seen):
|
||||
cands = self.funcs.get(name)
|
||||
if not cands:
|
||||
return set(), True # builtin: json_get, env, ...
|
||||
pats, unres = set(), False
|
||||
for callee in cands:
|
||||
key = ("fn", callee.path, callee.name, tuple(args))
|
||||
if key in seen:
|
||||
unres = True
|
||||
continue
|
||||
seen = seen | {key}
|
||||
# bind the callee's params to THIS call site's argument expressions
|
||||
binding = {}
|
||||
for idx, pname in enumerate(callee.params):
|
||||
if idx < len(args):
|
||||
binding[pname] = (args[idx], func)
|
||||
callee_ctx = _Bound(callee, binding)
|
||||
for r in self.returns_of(callee_ctx, depth + 1, seen):
|
||||
p, u = self._expr(callee.toks, r[0], r[1], callee_ctx,
|
||||
depth + 1, seen)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
def _var(self, name, func, depth, seen):
|
||||
real = func.func if isinstance(func, _Bound) else func
|
||||
|
||||
# 1. a parameter bound by the call site we came through
|
||||
if isinstance(func, _Bound) and name in func.binding:
|
||||
rng, caller_ctx = func.binding[name]
|
||||
return self._expr(caller_ctx.toks, rng[0], rng[1], caller_ctx,
|
||||
depth + 1, seen)
|
||||
|
||||
# 2. a local `let` in the enclosing function
|
||||
lets = self.lets_of(real)
|
||||
if name in lets:
|
||||
key = ("let", real.path, real.name, name)
|
||||
if key in seen:
|
||||
return set(), True
|
||||
seen = seen | {key}
|
||||
pats, unres = set(), False
|
||||
for rng in lets[name]:
|
||||
p, u = self._expr(real.toks, rng[0], rng[1], real, depth + 1, seen)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
# 3. an unbound parameter -> look at every call site of the enclosing fn
|
||||
if name in real.params:
|
||||
key = ("param", real.path, real.name, name)
|
||||
if key in seen:
|
||||
return set(), True
|
||||
seen = seen | {key}
|
||||
idx = real.params.index(name)
|
||||
pats, unres = set(), False
|
||||
sites = self.calls.get(real.name, [])
|
||||
if not sites:
|
||||
return set(), True
|
||||
for caller, args, _rel, _line in sites:
|
||||
if caller is None or idx >= len(args):
|
||||
unres = True
|
||||
continue
|
||||
p, u = self._expr(caller.toks, args[idx][0], args[idx][1],
|
||||
caller, depth + 1, seen)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
# 4. a file-level / cross-file top-level `let`
|
||||
for tl in self.toplevel:
|
||||
lets = self.lets_of(tl)
|
||||
if name in lets:
|
||||
key = ("let", tl.path, tl.name, name)
|
||||
if key in seen:
|
||||
return set(), True
|
||||
seen2 = seen | {key}
|
||||
pats, unres = set(), False
|
||||
for rng in lets[name]:
|
||||
p, u = self._expr(tl.toks, rng[0], rng[1], tl, depth + 1, seen2)
|
||||
pats |= p
|
||||
unres = unres or u
|
||||
return pats, unres
|
||||
|
||||
return set(), True
|
||||
|
||||
|
||||
class _Bound:
|
||||
"""A callee view that also knows what its params were called with."""
|
||||
|
||||
def __init__(self, func, binding):
|
||||
self.func, self.binding = func, binding
|
||||
self.toks, self.start, self.end = func.toks, func.start, func.end
|
||||
self.params, self.path, self.name = func.params, func.path, func.name
|
||||
|
||||
def __getattr__(self, k):
|
||||
return getattr(self.func, k)
|
||||
|
||||
|
||||
# ── token helpers ───────────────────────────────────────────────────────────
|
||||
def split_args(toks, s, e):
|
||||
out, cur, d = [], s, 0
|
||||
i = s
|
||||
while i < e:
|
||||
v = toks[i].val
|
||||
if toks[i].kind == TOK_PUNCT and v in "([{":
|
||||
d += 1
|
||||
elif toks[i].kind == TOK_PUNCT and v in ")]}":
|
||||
d -= 1
|
||||
elif d == 0 and toks[i].kind == TOK_PUNCT and v == ",":
|
||||
out.append((cur, i))
|
||||
cur = i + 1
|
||||
i += 1
|
||||
if cur < e:
|
||||
out.append((cur, e))
|
||||
return out
|
||||
|
||||
|
||||
STMT_START = {"let", "return", "if", "while", "for"}
|
||||
KEYWORDS = {"if", "while", "for", "return", "fn", "let", "else", "match"}
|
||||
|
||||
|
||||
def stmt_end(toks, s, limit):
|
||||
"""End of the expression starting at s: the next top-level statement
|
||||
boundary. El has no semicolons, so a newline that starts a new statement
|
||||
ends this one."""
|
||||
d, i = 0, s
|
||||
while i < limit:
|
||||
t = toks[i]
|
||||
if t.kind == TOK_PUNCT and t.val in "([":
|
||||
d += 1
|
||||
elif t.kind == TOK_PUNCT and t.val in ")]":
|
||||
d -= 1
|
||||
if d < 0:
|
||||
return i
|
||||
elif t.kind == TOK_PUNCT and t.val == "{":
|
||||
# a brace at depth 0 belongs to this expression only when it is an
|
||||
# if/else block that is part of it
|
||||
d += 1
|
||||
elif t.kind == TOK_PUNCT and t.val == "}":
|
||||
d -= 1
|
||||
if d < 0:
|
||||
return i
|
||||
elif d == 0 and t.kind == TOK_PUNCT and t.val == ",":
|
||||
return i
|
||||
elif d == 0 and i > s and t.kind == TOK_IDENT and t.val in STMT_START:
|
||||
if t.val == "if" and toks[i - 1].kind == TOK_IDENT and toks[i - 1].val == "else":
|
||||
i += 1
|
||||
continue
|
||||
return i
|
||||
elif d == 0 and i > s and t.kind == TOK_IDENT and t.val == "fn":
|
||||
return i
|
||||
i += 1
|
||||
return limit
|
||||
|
||||
|
||||
def block_tail(toks, s, e):
|
||||
"""The trailing expression of a block, if the block ends in one."""
|
||||
i, last = s, None
|
||||
while i < e:
|
||||
t = toks[i]
|
||||
if t.kind == TOK_IDENT and t.val in ("let", "return"):
|
||||
i = stmt_end(toks, i + 1, e)
|
||||
last = None
|
||||
continue
|
||||
if t.kind == TOK_PUNCT and t.val in "([{":
|
||||
i = match_close(toks, i, t.val, {"(": ")", "[": "]", "{": "}"}[t.val]) + 1
|
||||
continue
|
||||
st = i
|
||||
en = stmt_end(toks, i, e)
|
||||
if en <= st:
|
||||
i = st + 1
|
||||
continue
|
||||
last = (st, en)
|
||||
i = en
|
||||
return last
|
||||
|
||||
|
||||
def render(toks, s, e):
|
||||
out = []
|
||||
for t in toks[s:e]:
|
||||
out.append('"%s"' % t.val if t.kind == TOK_STR else t.val)
|
||||
return " ".join(out)
|
||||
|
||||
|
||||
# ── the gate ────────────────────────────────────────────────────────────────
|
||||
def collect(root, include_tests):
|
||||
files = []
|
||||
for dirpath, dirnames, filenames in os.walk(root):
|
||||
dirnames[:] = [d for d in dirnames
|
||||
if d not in ("dist", "vendor", ".git", "node_modules")]
|
||||
rel_dir = os.path.relpath(dirpath, root)
|
||||
if not include_tests and rel_dir.split(os.sep)[0] == "tests":
|
||||
continue
|
||||
for fn in sorted(filenames):
|
||||
if fn.endswith(".el"):
|
||||
rel = os.path.normpath(os.path.join(rel_dir, fn))
|
||||
files.append((os.path.join(dirpath, fn), rel))
|
||||
return sorted(files, key=lambda x: x[1])
|
||||
|
||||
|
||||
def is_bare_literal(prog, site):
|
||||
toks = prog.files[site.path]
|
||||
s, e = site.arg_range
|
||||
return e == s + 1 and toks[s].kind == TOK_STR
|
||||
|
||||
|
||||
def read_decl(path):
|
||||
"""A declaration file: one entry per line, `# ...` comments stripped."""
|
||||
out = []
|
||||
if not path or not os.path.exists(path):
|
||||
return out
|
||||
with open(path) as fh:
|
||||
for ln in fh:
|
||||
ln = ln.split("#", 1)[0].strip()
|
||||
if ln:
|
||||
out.append(ln)
|
||||
return out
|
||||
|
||||
|
||||
def opt(argv, name, default=None):
|
||||
for i, a in enumerate(argv):
|
||||
if a == name and i + 1 < len(argv):
|
||||
return argv[i + 1]
|
||||
return default
|
||||
|
||||
|
||||
def main(argv):
|
||||
root = os.path.abspath(argv[1]) if len(argv) > 1 and not argv[1].startswith("-") else "."
|
||||
include_tests = "--include-tests" in argv
|
||||
verbose = "--verbose" in argv
|
||||
baseline_path = opt(argv, "--baseline")
|
||||
external_path = opt(argv, "--external")
|
||||
|
||||
prog = Program()
|
||||
for path, rel in collect(root, include_tests):
|
||||
prog.load(path, rel)
|
||||
prog.index()
|
||||
for site in prog.sites:
|
||||
pats, unres = prog.resolve(site.arg_range, site.func)
|
||||
site.pats, site.unresolved = {p for p in pats if p}, unres
|
||||
if is_bare_literal(prog, site):
|
||||
site.literal = prog.files[site.path][site.arg_range[0]].val
|
||||
|
||||
writes = [s for s in prog.sites if s.kind == "set"]
|
||||
reads = [s for s in prog.sites if s.kind == "get"]
|
||||
write_pats = set()
|
||||
for w in writes:
|
||||
write_pats |= w.pats
|
||||
|
||||
# Declared host-set keys: written by something outside the El tree (an
|
||||
# operator, the installer, a host process). Each entry must carry a reason.
|
||||
external = []
|
||||
for ln in read_decl(external_path):
|
||||
parts = ln.split(None, 1)
|
||||
if len(parts) != 2 or parts[0] not in (EXACT, PREFIX):
|
||||
print("bad --external line (want `exact|prefix <key>`): %r" % ln,
|
||||
file=sys.stderr)
|
||||
return 2
|
||||
external.append((parts[0], parts[1]))
|
||||
write_pats |= set(external)
|
||||
|
||||
# F1 — a read of a key no write in the tree produces.
|
||||
f1 = []
|
||||
for r in reads:
|
||||
for p in sorted(r.pats):
|
||||
if not any(covers(w, p) for w in write_pats):
|
||||
f1.append((r, p))
|
||||
|
||||
# F2 — a key namespace owned by a helper, accessed by a hand-rolled literal.
|
||||
# This is the #129 shape: the producer moved behind conv_hist_key() and
|
||||
# one consumer kept spelling the old key out by hand.
|
||||
owners = {} # helper fn name -> its value set
|
||||
for s in prog.sites:
|
||||
toks = prog.files[s.path]
|
||||
a, b = s.arg_range
|
||||
if toks[a].kind == TOK_IDENT and a + 1 < b and toks[a + 1].val == "(" \
|
||||
and match_close(toks, a + 1, "(", ")") == b - 1 \
|
||||
and toks[a].val in prog.funcs:
|
||||
name = toks[a].val
|
||||
if name not in owners:
|
||||
vals = set()
|
||||
for callee in prog.funcs[name]:
|
||||
# No call context here on purpose: the OWNED namespace is
|
||||
# every key the helper can ever produce, over all call sites.
|
||||
for rng in prog.returns_of(callee):
|
||||
p, _ = prog._expr(callee.toks, rng[0], rng[1], callee, 0, set())
|
||||
vals |= {x for x in p if x}
|
||||
owners[name] = vals
|
||||
f2 = []
|
||||
for s in prog.sites:
|
||||
if s.literal is None:
|
||||
continue
|
||||
for owner, vals in sorted(owners.items()):
|
||||
for v in sorted(vals):
|
||||
if covers(v, pat_exact(s.literal)):
|
||||
f2.append((s, owner, v))
|
||||
break
|
||||
else:
|
||||
continue
|
||||
break
|
||||
|
||||
unresolved = [s for s in prog.sites if s.unresolved or not s.pats]
|
||||
|
||||
# Baseline signatures carry NO line number on purpose: an unrelated edit that
|
||||
# shifts a line must not un-mute an accepted finding (that is crying wolf),
|
||||
# but a GROWTH in count must not hide either. So a baseline entry is
|
||||
# `<file> <CODE> <detail> [xN]` and only the first N matches are muted.
|
||||
baseline, bad_baseline = {}, []
|
||||
for ln in read_decl(baseline_path):
|
||||
n, key = 1, ln
|
||||
parts = ln.rsplit(" x", 1)
|
||||
if len(parts) == 2 and parts[1].isdigit():
|
||||
key, n = parts[0].strip(), int(parts[1])
|
||||
baseline[key] = n
|
||||
|
||||
def sig(path, code, detail):
|
||||
return "%s %s %s" % (path, code, detail)
|
||||
|
||||
findings = []
|
||||
for r, p in f1:
|
||||
findings.append((sig(r.path, "DEAD-READ", "%s:%s" % p), r.line,
|
||||
" %s:%d state_get(%s)\n resolves to %s %r — no state_set in the tree produces it"
|
||||
% (r.path, r.line, r.text, p[0].upper(), p[1])))
|
||||
for s, owner, v in f2:
|
||||
findings.append((sig(s.path, "HAND-ROLLED", "%s<-%s()" % (s.literal, owner)), s.line,
|
||||
" %s:%d state_%s(\"%s\")\n %s() owns this key namespace (%s %r) — go through the helper, "
|
||||
"or a rename orphans this site silently" % (s.path, s.line, s.kind, s.literal, owner, v[0].upper(), v[1])))
|
||||
findings.sort(key=lambda f: (f[0], f[1]))
|
||||
|
||||
live, muted, budget = [], [], dict(baseline)
|
||||
for f in findings:
|
||||
if budget.get(f[0], 0) > 0:
|
||||
budget[f[0]] -= 1
|
||||
muted.append(f)
|
||||
else:
|
||||
live.append(f)
|
||||
stale = sorted(k for k, v in budget.items() if v > 0)
|
||||
|
||||
print("── state-key audit ─────────────────────────────────────────────")
|
||||
print("scanned %d .el files%s" % (len(prog.files),
|
||||
"" if include_tests else " (tests/ excluded)"))
|
||||
print("sites %d state_set, %d state_get" % (len(writes), len(reads)))
|
||||
print("keys %d distinct write patterns" % len(write_pats))
|
||||
print("")
|
||||
|
||||
if verbose:
|
||||
print("WRITE PATTERNS")
|
||||
for k, v in sorted(write_pats):
|
||||
print(" %-6s %s" % (k, v))
|
||||
print("")
|
||||
|
||||
if external:
|
||||
print("DECLARED HOST-SET (%d) — %s" % (len(external), external_path))
|
||||
for k, v in sorted(external):
|
||||
print(" %-6s %s" % (k, v))
|
||||
print("")
|
||||
|
||||
print("UNRESOLVED (%d) — reported, never fails the build" % len(unresolved))
|
||||
if not unresolved:
|
||||
print(" (none)")
|
||||
for s in sorted(unresolved, key=lambda x: (x.path, x.line)):
|
||||
print(" %s:%d state_%s(%s)%s"
|
||||
% (s.path, s.line, s.kind, s.text,
|
||||
" [partial: %s]" % ", ".join("%s %r" % p for p in sorted(s.pats))
|
||||
if s.pats else ""))
|
||||
print("")
|
||||
|
||||
if muted:
|
||||
print("BASELINED (%d) — pre-existing debt accepted in %s. NOT clean; fix these."
|
||||
% (len(muted), baseline_path))
|
||||
for sg, line, _ in muted:
|
||||
print(" %s (line %d)" % (sg, line))
|
||||
print("")
|
||||
if stale:
|
||||
print("STALE BASELINE (%d) — entries that no longer match anything; delete them:"
|
||||
% len(stale))
|
||||
for sg in stale:
|
||||
print(" %s" % sg)
|
||||
print("")
|
||||
|
||||
print("FINDINGS (%d)" % len(live))
|
||||
if not live:
|
||||
print(" (none)")
|
||||
for _, _, body in live:
|
||||
print(body)
|
||||
print("")
|
||||
|
||||
if live:
|
||||
print("FAIL: %d state-key finding(s). See scripts/verify-state-keys.sh "
|
||||
"for why this gate exists (issue #129)." % len(live))
|
||||
return 1
|
||||
print("PASS: every resolvable state_get key has a producer, and no key "
|
||||
"namespace is spelled two ways.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
@@ -0,0 +1,28 @@
|
||||
# state-key-baseline.txt — findings that already existed when this gate landed
|
||||
# (2026-08-07). Each one is a REAL defect of the #129 class, not a false
|
||||
# positive. They are muted only so the gate can be turned on today instead of
|
||||
# being deferred until the debt is paid; every run still prints them under
|
||||
# BASELINED with the word "debt".
|
||||
#
|
||||
# THIS FILE SHOULD ONLY EVER SHRINK. Adding a line means you are shipping a
|
||||
# known silent-"" read. If you must, date it and say why in the comment.
|
||||
#
|
||||
# format: <file> <CODE> <detail> [xN] # N = how many sites are accepted
|
||||
# No line numbers on purpose: an unrelated edit must not un-mute an accepted
|
||||
# finding, but a GROWTH in count is NOT muted — the extra site fails the build.
|
||||
#
|
||||
chat.el DEAD-READ exact:soul_identity x5
|
||||
# ^ soul.el used to run `state_set("soul_identity", soul_identity)`. It was
|
||||
# deleted on 2026-05-13 in b163fa6 ("feat(awareness): route ISE writes to HTTP
|
||||
# Engram ..."), a commit about something else entirely, and the five readers in
|
||||
# chat.el were left behind. Since that date build_system_prompt (737), the
|
||||
# vision handler (1745), the agentic system prompt (2620), the council
|
||||
# transcript handler (3425) and 3480 have all been prefixing "" — exactly the
|
||||
# #129 shape, found by this gate on its first run. Sites: 737, 1745, 2620,
|
||||
# 3425, 3480. Fix = restore the boot-time write or delete the reads; not done
|
||||
# here because this branch must not change engine behaviour.
|
||||
|
||||
studio.el DEAD-READ exact:soul_principal x1
|
||||
# ^ studio.el:57 dharma_registry() emits "principal":"" on every call — no
|
||||
# producer has ever existed in the tree's history (git log -S finds none).
|
||||
# Never-wired rather than orphaned, same silent-"" result.
|
||||
@@ -0,0 +1,16 @@
|
||||
# state-key-external.txt — state keys the engine READS but deliberately never
|
||||
# WRITES, because a host outside the El tree sets them (an operator, the
|
||||
# installer, a deployment env). Read scripts/verify-state-keys.sh for why this
|
||||
# list has to exist and why it has to stay short.
|
||||
#
|
||||
# THE RULE FOR ADDING A LINE: the read site must already treat "" as a defined
|
||||
# default (`if str_eq(x, "") { <default> }`) AND the source must say so in a
|
||||
# comment. "I could not find the writer" is NOT a reason — that is the #129
|
||||
# defect, and it belongs in state-key-baseline.txt with a date, not here.
|
||||
#
|
||||
# format: exact|prefix <key> # why, and where the source says so
|
||||
#
|
||||
exact soul_rate_limit # routes.el:59-61 — "configurable via soul state key ... Falls back to 60 req/min if not set."
|
||||
exact web_search_tool_version # chat.el:1884-1910 — version lives in state "so a future bump is a config write, not a recompile"; defaults to web_search_20250305
|
||||
exact platform_auth # stewardship.el:92 — host-set capability flag; fail-CLOSED (anything but "true" denies the platform tool)
|
||||
exact security_research_authorized # awareness.el:991-996 — state override for env SECURITY_RESEARCH_TOKEN; fail-closed, defaults false
|
||||
Executable
+118
@@ -0,0 +1,118 @@
|
||||
#!/usr/bin/env bash
|
||||
# verify-state-keys.sh — the state-key gate. Retires a defect class at build time.
|
||||
#
|
||||
# ── WHY THIS EXISTS. DO NOT DELETE IT AS NOISE. ──────────────────────────────
|
||||
#
|
||||
# The engine keeps runtime values in a key-value store: state_set("k", v) writes,
|
||||
# state_get("k") reads. A read of a key that NOTHING writes returns an empty
|
||||
# string. Silently. No error, no warning, no log line. The El compiler cannot see
|
||||
# it, no test sees it, and the product keeps running — just with a hole in it.
|
||||
#
|
||||
# That is how issue #129 happened. ff421d3 (2026-08-05) correctly moved
|
||||
# conversation history to a per-session key behind conv_hist_key(session_id). One
|
||||
# consumer did not move with it: the agentic path's L1 safety screen kept reading
|
||||
# the old anonymous "conv_history" bucket. The desktop app always mints a session
|
||||
# id, so history was always written under session_hist_<id> and that read always
|
||||
# returned "". The half of the crisis score that receives history is the
|
||||
# ESCALATION half — the one that exists for distress building across several
|
||||
# turns, where no single message trips the bell on its own. It scored 0 on every
|
||||
# real conversation for two days, and nothing failed.
|
||||
#
|
||||
# The line that broke carried a comment describing this exact bug being fixed
|
||||
# once already, under issue #9. A comment is not a gate. This is the gate.
|
||||
#
|
||||
# ── WHAT IT CHECKS ──────────────────────────────────────────────────────────
|
||||
#
|
||||
# DEAD-READ a state_get whose key resolves to something no state_set in the
|
||||
# tree produces. The direct form of the class.
|
||||
#
|
||||
# HAND-ROLLED a state_get/state_set that spells out a literal belonging to a
|
||||
# key namespace a helper function owns (e.g. "conv_history", owned
|
||||
# by conv_hist_key()). This is #129's actual shape: the producer
|
||||
# moved behind the helper and one consumer kept the old spelling
|
||||
# by hand. DEAD-READ alone does NOT catch #129, because the dead
|
||||
# handle_chat() still writes that key through the helper — so this
|
||||
# second check is the one that earns the gate its keep.
|
||||
#
|
||||
# ── WHY IT DOES NOT CRY WOLF ────────────────────────────────────────────────
|
||||
#
|
||||
# Keys are usually COMPUTED, not literal, so a naive grep would flood and get
|
||||
# switched off within a day. scripts/state-key-audit.py resolves computed keys:
|
||||
# string concatenation (matched on the static prefix), helper functions (resolved
|
||||
# to their possible return values), keys built into a local variable, and keys
|
||||
# arriving as a function parameter (resolved through the call sites). Where a key
|
||||
# genuinely cannot be resolved it is printed under UNRESOLVED and does NOT fail
|
||||
# the build — visible, never silently ignored. Keep that list short.
|
||||
#
|
||||
# On this tree it resolves 278 of 278 sites: UNRESOLVED is 0 and FINDINGS is 0.
|
||||
#
|
||||
# Two declaration files, both of which should only ever shrink:
|
||||
# scripts/state-key-external.txt keys a host outside the El tree writes
|
||||
# scripts/state-key-baseline.txt findings that predate the gate (real debt)
|
||||
#
|
||||
# ── PROVEN TO DISCRIMINATE (2026-08-07) ─────────────────────────────────────
|
||||
#
|
||||
# 1. Synthetic: a scratch copy of this tree with agentic_safety_screen reverted
|
||||
# to the pre-fix state_get("conv_history") — ONE line, nothing else — FAILS
|
||||
# with `chat.el:2536 ... conv_hist_key() owns this key namespace`. The tree
|
||||
# as shipped PASSES. One variable, opposite verdicts.
|
||||
# 2. Independent: run read-only against origin/feat/soul-openai-tools-v2, which
|
||||
# carries the same defect on its own, the gate reported chat.el:2937 — the
|
||||
# exact line 43d0449's commit message had named by hand. Against that
|
||||
# branch's fix (origin/fix/129-on-openai-tools) it passes.
|
||||
# 3. Producer-moved controls: renaming the sole writer of an EXACT key
|
||||
# (soul_model) orphans 3 readers across 3 files; renaming the sole writer of
|
||||
# a PREFIX namespace (agent_workspace_root_*) orphans 3 readers — including
|
||||
# when the producer moves to a NARROWER namespace, which an earlier,
|
||||
# sloppier prefix rule let through.
|
||||
#
|
||||
# It also found, on its first run, a defect nobody was looking for: soul.el's
|
||||
# `state_set("soul_identity", ...)` was deleted on 2026-05-13 in b163fa6 (a
|
||||
# commit about awareness/ISE writes) and five readers in chat.el were left
|
||||
# behind — the system prompt, the vision handler, the agentic prompt and the
|
||||
# council handler have been prefixing "" ever since. See state-key-baseline.txt.
|
||||
#
|
||||
# ── SAFETY ──────────────────────────────────────────────────────────────────
|
||||
# Pure static read of .el sources. Starts nothing, opens no port, touches no
|
||||
# daemon, and never reads or writes ~/.neuron.
|
||||
#
|
||||
# ── USAGE ───────────────────────────────────────────────────────────────────
|
||||
# scripts/verify-state-keys.sh gate the repo (honours baseline)
|
||||
# scripts/verify-state-keys.sh --strict ignore the baseline: show the debt
|
||||
# scripts/verify-state-keys.sh --verbose also dump every write pattern
|
||||
# scripts/verify-state-keys.sh --root DIR audit a different tree
|
||||
# exit 0 = clean; 1 = finding(s); 2 = the gate itself could not run.
|
||||
set -uo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
||||
STRICT=0
|
||||
PASS_THROUGH=()
|
||||
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--strict) STRICT=1; shift ;;
|
||||
--root) ROOT="${2:?--root needs a directory}"; shift 2 ;;
|
||||
-h|--help) awk 'NR>1 && /^#/ {print; next} NR>1 {exit}' "${BASH_SOURCE[0]}"; exit 0 ;;
|
||||
*) PASS_THROUGH+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
|
||||
command -v python3 >/dev/null 2>&1 || {
|
||||
echo "[state-keys] CANNOT RUN: python3 not found" >&2; exit 2; }
|
||||
[ -d "$ROOT" ] || { echo "[state-keys] CANNOT RUN: no such tree: $ROOT" >&2; exit 2; }
|
||||
|
||||
AUDIT="$SCRIPT_DIR/state-key-audit.py"
|
||||
[ -f "$AUDIT" ] || { echo "[state-keys] CANNOT RUN: missing $AUDIT" >&2; exit 2; }
|
||||
|
||||
ARGS=("$ROOT" "--external" "$SCRIPT_DIR/state-key-external.txt")
|
||||
[ "$STRICT" -eq 0 ] && ARGS+=("--baseline" "$SCRIPT_DIR/state-key-baseline.txt")
|
||||
[ ${#PASS_THROUGH[@]} -gt 0 ] && ARGS+=("${PASS_THROUGH[@]}")
|
||||
|
||||
python3 "$AUDIT" "${ARGS[@]}"
|
||||
RC=$?
|
||||
if [ "$RC" -gt 1 ]; then
|
||||
echo "[state-keys] CANNOT RUN: the audit itself failed (exit $RC)" >&2
|
||||
exit 2
|
||||
fi
|
||||
exit "$RC"
|
||||
@@ -379,9 +379,23 @@ fn emit_session_start_event() -> Void {
|
||||
// layered_cycle — routes user-facing requests through the 4-layer consciousness stack.
|
||||
// L0 (core) → L1 (safety screen) → L2a (continuity + behavioral profiling) → L2b (mission alignment) → L3 (imprint) → L1 (safety validate)
|
||||
// Internal cognition (heartbeat, proactive, memory ops) bypasses layers — use one_cycle directly.
|
||||
fn layered_cycle(raw_input: String) -> String {
|
||||
let history: String = state_get("conv_history")
|
||||
let session_id: String = state_get("current_session_id")
|
||||
//
|
||||
// FIX B (2026-08-05) — the cycle now knows which conversation it is in.
|
||||
//
|
||||
// session_id: the caller's session, threaded from the route. Was previously read from the
|
||||
// state key "current_session_id", which is read HERE and written NOWHERE in the entire
|
||||
// source — verified across every .el file. So this value was unconditionally "", and every
|
||||
// downstream consumer of it silently fell back to a process-global bucket: conversation
|
||||
// history, and the steward's continuity tracking (TODO reliability #4, below, describes the
|
||||
// cross-session bleed this caused; threading the real id closes it). The plain path's blank
|
||||
// stare and the agentic path's scoped history were the same defect seen from two sides.
|
||||
//
|
||||
// utility: true when the generation is not part of the user's conversation — the app's
|
||||
// title and insight passes. Answered normally, never recorded. See is_utility_request.
|
||||
fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String {
|
||||
// Safety-screen history amplification now reads the SAME window the turn will be
|
||||
// recorded into, so a session's own escalation pattern is what gets scored.
|
||||
let history: String = state_get(conv_hist_key(session_id))
|
||||
|
||||
// L1 in: safety screen
|
||||
let screen_result: String = safety_screen(raw_input, history)
|
||||
@@ -423,8 +437,10 @@ fn layered_cycle(raw_input: String) -> String {
|
||||
let cont_action: String = json_get(continuity, "action")
|
||||
|
||||
// Store continuity status so imprint can adjust its response register.
|
||||
// TODO(reliability #4): session_continuity is process-global; scope per session_id
|
||||
// when available to prevent cross-session bleed under concurrent layered_cycle calls.
|
||||
// TODO(reliability #4) CLOSED 2026-08-05: this line was already written to scope per
|
||||
// session — it just never received a session id, because the only source was a state key
|
||||
// nothing wrote. It is now threaded from the route, so named sessions genuinely get their
|
||||
// own continuity state and only anonymous callers share the global one.
|
||||
let cont_key: String = if str_eq(session_id, "") { "session_continuity" } else { "session_continuity:" + session_id }
|
||||
state_set(cont_key, cont_status)
|
||||
|
||||
@@ -453,40 +469,24 @@ fn layered_cycle(raw_input: String) -> String {
|
||||
let lc_aff_cutoff: Int = time_now() - 259200
|
||||
let lc_bell_nodes: String = engram_search_json("bell:soft bell:hard BellEvent affective", 2)
|
||||
let lc_has_bell: Bool = !str_eq(lc_bell_nodes, "") && !str_eq(lc_bell_nodes, "[]")
|
||||
// CRASH FIX 2026-08-05 (BUG-PLAINCHAT-1): the " | ts:" parser used to be inline here.
|
||||
// Inside this block-expression initializer elc compiled `lbmp + str_len(lbm)` to
|
||||
// el_str_concat() on two integers, which segfaulted the whole daemon the moment a
|
||||
// distress turn followed an earlier affective turn — i.e. exactly on the crisis path.
|
||||
// Verified against the unmodified baseline binary AND present in the committed
|
||||
// dist/soul.c. affective_node_ts() is a top-level function, where the same expression
|
||||
// compiles to integer addition. Do not inline it back.
|
||||
let lc_bell_note: String = if lc_has_bell {
|
||||
let lb0: String = json_array_get(lc_bell_nodes, 0)
|
||||
let lb_c: String = json_get(lb0, "content")
|
||||
let lbm: String = " | ts:"
|
||||
let lbmp: Int = str_index_of(lb_c, lbm)
|
||||
let lb_ts_raw: String = if lbmp >= 0 {
|
||||
let lbs: Int = lbmp + str_len(lbm)
|
||||
let lbr: String = str_slice(lb_c, lbs, str_len(lb_c))
|
||||
let lbn: Int = str_index_of(lbr, " | ")
|
||||
if lbn < 0 { lbr } else { str_slice(lbr, 0, lbn) }
|
||||
} else {
|
||||
let lbca: String = json_get(lb0, "created_at")
|
||||
if str_eq(lbca, "") { json_get(lb0, "updated_at") } else { lbca }
|
||||
}
|
||||
let lb_ts: Int = if str_eq(lb_ts_raw, "") { 0 } else { str_to_int(lb_ts_raw) }
|
||||
let lb_ts: Int = affective_node_ts(lb0)
|
||||
if lb_ts > lc_aff_cutoff { "[AFFECTIVE NOTE: User was in distress in a recent session.]" } else { "" }
|
||||
} else { "" }
|
||||
let lc_pos_nodes: String = engram_search_json("PositiveEvent joy:high joy:low affective", 2)
|
||||
let lc_has_pos: Bool = !str_eq(lc_pos_nodes, "") && !str_eq(lc_pos_nodes, "[]")
|
||||
// Same crash fix as the bell note above (BUG-PLAINCHAT-1).
|
||||
let lc_pos_note: String = if lc_has_pos && str_eq(lc_bell_note, "") {
|
||||
let lp0: String = json_array_get(lc_pos_nodes, 0)
|
||||
let lp_c: String = json_get(lp0, "content")
|
||||
let lpm: String = " | ts:"
|
||||
let lpmp: Int = str_index_of(lp_c, lpm)
|
||||
let lp_ts_raw: String = if lpmp >= 0 {
|
||||
let lps: Int = lpmp + str_len(lpm)
|
||||
let lpr: String = str_slice(lp_c, lps, str_len(lp_c))
|
||||
let lpn: Int = str_index_of(lpr, " | ")
|
||||
if lpn < 0 { lpr } else { str_slice(lpr, 0, lpn) }
|
||||
} else {
|
||||
let lpca: String = json_get(lp0, "created_at")
|
||||
if str_eq(lpca, "") { json_get(lp0, "updated_at") } else { lpca }
|
||||
}
|
||||
let lp_ts: Int = if str_eq(lp_ts_raw, "") { 0 } else { str_to_int(lp_ts_raw) }
|
||||
let lp_ts: Int = affective_node_ts(lp0)
|
||||
if lp_ts > lc_aff_cutoff { "[AFFECTIVE NOTE: User shared positive news in a recent session.]" } else { "" }
|
||||
} else { "" }
|
||||
let lc_affective_note: String = if !str_eq(lc_bell_note, "") { lc_bell_note } else { lc_pos_note }
|
||||
@@ -498,11 +498,47 @@ fn layered_cycle(raw_input: String) -> String {
|
||||
}
|
||||
state_set("layered_cycle_safety_system_addendum", augmented_addendum)
|
||||
|
||||
// L3: imprint responds
|
||||
let output: String = imprint_respond(aligned, imprint_id)
|
||||
// L3: imprint responds — applies the active imprint's voice/domain annotation to the
|
||||
// steward-aligned input. This produces the PROMPT, not the answer.
|
||||
let prompt: String = imprint_respond(aligned, imprint_id)
|
||||
|
||||
// L1 out: validate output before delivery
|
||||
return safety_validate(output, screen_action)
|
||||
// L3b: the imprint SPEAKS (added 2026-08-05).
|
||||
//
|
||||
// Until now the cycle stopped at the annotation above, so /api/chat with agentic:false
|
||||
// handed the user's own screened text back as the "reply" — every gate ran, but nothing
|
||||
// ever generated. The generation is placed HERE, inside the cycle, rather than by
|
||||
// pointing the route at handle_chat(): handle_chat has no enforcing input gate and no
|
||||
// enforcing output gate, so calling it instead of this cycle would have traded the whole
|
||||
// safety pipeline for a working reply. Composing keeps both.
|
||||
//
|
||||
// Order is deliberate and must not be rearranged: this call sits strictly AFTER the L1
|
||||
// screen, the safe-mode guard, the hard-bell short-circuit and the L2 stewardship layers,
|
||||
// and strictly BEFORE the L1 output gate. A hard bell never reaches a model — the branch
|
||||
// above returns first. Tools are not offered on this turn; see layered_generate.
|
||||
let output: String = layered_generate(prompt, imprint_id, session_id)
|
||||
|
||||
// L1 out: validate output before delivery. Still the terminal gate — nothing below this
|
||||
// line can change the string this function returns.
|
||||
let validated: String = safety_validate(output, screen_action)
|
||||
|
||||
// Turn bookkeeping. Records the VALIDATED text, never the raw model output, and is only
|
||||
// reachable on the non-bell path: both bell branches above return before this point, so
|
||||
// bell turns still never enter conversation history. Pure state side effect — it cannot
|
||||
// alter what is returned.
|
||||
//
|
||||
// FIX A: the receipt is unconditional and always negative on this path, because on this
|
||||
// path it is structurally true — layered_generate offers no tools at all (build_system_prompt
|
||||
// chat mode + a request body with no "tools" key). Recording "no tools ran" is not padding:
|
||||
// it is the only thing that distinguishes "nothing ran" from "we forgot to write down what
|
||||
// ran", and that ambiguity is what made the model confess to a search it had performed.
|
||||
//
|
||||
// FIX E1: a utility generation is answered but not recorded. Guarded here rather than at
|
||||
// the route so every /api/chat dispatch site inherits it from one place.
|
||||
let receipt: String = tool_receipt("", "")
|
||||
if !utility {
|
||||
conv_history_record(session_id, raw_input, validated, receipt)
|
||||
}
|
||||
return validated
|
||||
}
|
||||
|
||||
let soul_cgi_id_raw: String = env("SOUL_CGI_ID")
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
// auto-generated by elc --emit-header - do not edit
|
||||
extern fn init_soul_edges() -> Void
|
||||
extern fn ensure_self_canonical_bridge() -> Void
|
||||
extern fn aff_try_slot(slot_json: String, aff_7d_ts: Int, acc_key: String) -> Void
|
||||
extern fn load_identity_context() -> Void
|
||||
extern fn seed_persona_from_env() -> Void
|
||||
extern fn emit_session_start_event() -> Void
|
||||
extern fn layered_cycle(raw_input: String) -> String
|
||||
extern fn layered_cycle(raw_input: String, session_id: String, utility: Bool) -> String
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
// ── test_history_amplification.el ─────────────────────────────────────────────
|
||||
//
|
||||
// REGRESSION TEST FOR ISSUE #129 (P0, SAFETY).
|
||||
//
|
||||
// What this guards: on the agentic path, the crisis score has two halves — the
|
||||
// message you just sent, and the distress that has accumulated across the
|
||||
// conversation. The second half is the whole reason the escalation logic exists:
|
||||
// someone whose distress builds over several turns never sends one message that
|
||||
// trips the bell on its own.
|
||||
//
|
||||
// The defect this test was written against (ff421d3, 2026-08-05 → fixed
|
||||
// 2026-08-07): conversation history moved to a per-session key via
|
||||
// conv_hist_key(session_id), but the agentic path's safety screen was left
|
||||
// reading the old anonymous "conv_history" bucket. The desktop app always sends
|
||||
// a session_id, so the screen received "" on every real conversation and the
|
||||
// escalation half always scored 0. Nothing failed. Nothing logged. The comment
|
||||
// above the defective line documented this same bug being fixed once before.
|
||||
//
|
||||
// THE INVARIANT UNDER TEST, stated so it survives future renames:
|
||||
// the window the safety screen READS must be the window conv_history_record
|
||||
// WRITES. Not "must be called conv_history" — must AGREE.
|
||||
//
|
||||
// This test is deliberately written to fail loudly on the pre-fix source. If it
|
||||
// ever passes on code where the screen reads a key nothing writes, it is broken.
|
||||
//
|
||||
// To run (macOS, from the worktree root):
|
||||
// scripts/run-el-test.sh tests/test_history_amplification.el
|
||||
// ──────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
import "../chat.el"
|
||||
import "../safety.el"
|
||||
import "../sessions.el"
|
||||
|
||||
// Program class. Without this an El program compiles as a 'utility', and a
|
||||
// utility may not call the self-formation primitives (llm_call_system,
|
||||
// llm_vision) that chat.el's agentic loop references — the unit fails to
|
||||
// compile with a capability violation even though the test never calls them.
|
||||
// Declaring 'cgi' matches how soul.el declares itself.
|
||||
//
|
||||
// The endpoints below are deliberately DEAD: this test must never reach a live
|
||||
// engram, and nothing it asserts depends on one. Port 9 is discard.
|
||||
cgi "neuron-test-history-amplification" {
|
||||
dharma_id: "ntn-test@http://127.0.0.1:9",
|
||||
principal: "test-harness",
|
||||
network: "dharma-testnet",
|
||||
engram: "http://127.0.0.1:9"
|
||||
}
|
||||
|
||||
// ── Counters ──────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// NOTE for anyone copying this harness: the idiom used by the older tests in
|
||||
// this directory — `let pass_count = pass_count + 1` inside an assert function —
|
||||
// does NOT mutate the module-level binding. It declares a new local that dies
|
||||
// with the call, so those suites all print "0 passed, 0 failed" no matter what
|
||||
// happened. Counters go through the state store here so the summary is real.
|
||||
|
||||
fn bump(counter: String) -> Void {
|
||||
let cur: String = state_get(counter)
|
||||
let n: Int = if str_eq(cur, "") { 0 } else { str_to_int(cur) }
|
||||
state_set(counter, int_to_str(n + 1))
|
||||
}
|
||||
|
||||
fn counter_of(counter: String) -> Int {
|
||||
let cur: String = state_get(counter)
|
||||
if str_eq(cur, "") { return 0 }
|
||||
return str_to_int(cur)
|
||||
}
|
||||
|
||||
fn assert_eq(label: String, got: String, expected: String) -> Void {
|
||||
if str_eq(got, expected) {
|
||||
bump("t_pass")
|
||||
println(" PASS: " + label)
|
||||
} else {
|
||||
bump("t_fail")
|
||||
println(" FAIL: " + label)
|
||||
println(" got: " + got)
|
||||
println(" expected: " + expected)
|
||||
}
|
||||
}
|
||||
|
||||
fn assert_true(label: String, cond: Bool) -> Void {
|
||||
if cond {
|
||||
bump("t_pass")
|
||||
println(" PASS: " + label)
|
||||
} else {
|
||||
bump("t_fail")
|
||||
println(" FAIL: " + label)
|
||||
}
|
||||
}
|
||||
|
||||
// ── Fixtures ──────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// ESCALATION_MSG scores 50 on its own ("can't go on", safety_score_crisis s9).
|
||||
// 50 is >= soft (35) and < hard (70): a soft bell by itself, which is the point.
|
||||
// It is the message that is only a crisis IN CONTEXT.
|
||||
//
|
||||
// DISTRESS_TURN carries five history phrases:
|
||||
// hopeless 15 + worthless 15 + nobody cares 15 + can't take it anymore 20
|
||||
// + want to disappear 20 = 85; contributed at 1/3 weight = 28.
|
||||
// 50 + 28 = 78 >= 70 → hard_bell. That 28-point swing IS the escalation logic,
|
||||
// and it is exactly what the defect suppressed.
|
||||
|
||||
let TEST_SESSION: String = "sess-test-129-history-amplification"
|
||||
let ESCALATION_MSG: String = "honestly i can't go on like this"
|
||||
let DISTRESS_TURN: String = "i feel hopeless and worthless, nobody cares, i can't take it anymore, i want to disappear"
|
||||
|
||||
println("")
|
||||
println("test_history_amplification.el — issue #129 (agentic multi-turn crisis escalation)")
|
||||
|
||||
// ── 1. Baseline: the message alone is a SOFT bell, not a hard one ─────────────
|
||||
//
|
||||
// If this ever returns hard_bell, the test below proves nothing — the message
|
||||
// would trip the bell without any history and the amplification would be
|
||||
// invisible. This assertion is what keeps the real test honest.
|
||||
|
||||
println("")
|
||||
println("1. baseline — escalation message with NO history is a soft bell")
|
||||
|
||||
let baseline: String = safety_screen(ESCALATION_MSG, "")
|
||||
assert_eq("no history -> soft_bell (not hard)", json_get(baseline, "action"), "soft_bell")
|
||||
|
||||
// ── 2. Producer sanity: history lands in the session's own window ─────────────
|
||||
|
||||
println("")
|
||||
println("2. producer — conv_history_record writes the session's window")
|
||||
|
||||
conv_history_record(TEST_SESSION, DISTRESS_TURN, "i hear you, that sounds heavy", "")
|
||||
|
||||
let written: String = state_get(conv_hist_key(TEST_SESSION))
|
||||
assert_true("session window is non-empty after record", !str_eq(written, ""))
|
||||
assert_true("session window contains the distress turn", str_contains(written, "hopeless"))
|
||||
|
||||
// ── 3. THE REGRESSION: the agentic screen must SEE that window ────────────────
|
||||
//
|
||||
// Pre-fix this returns soft_bell, because agentic_safety_screen read the
|
||||
// anonymous bucket and got "". Post-fix it returns hard_bell.
|
||||
|
||||
println("")
|
||||
println("3. REGRESSION #129 — agentic screen reads the session's own window")
|
||||
|
||||
let screened: String = agentic_safety_screen(TEST_SESSION, ESCALATION_MSG)
|
||||
assert_eq(
|
||||
"distress history escalates the agentic screen to hard_bell",
|
||||
json_get(screened, "action"),
|
||||
"hard_bell"
|
||||
)
|
||||
|
||||
// ── 4. The invariant, stated directly ─────────────────────────────────────────
|
||||
//
|
||||
// Independent of thresholds and phrase lists: whatever the screen reads for a
|
||||
// session must equal what the recorder wrote for that session. This is the
|
||||
// assertion that survives a future rename of either side.
|
||||
|
||||
println("")
|
||||
println("4. invariant — read window == written window")
|
||||
|
||||
let read_back: String = state_get(conv_hist_key(TEST_SESSION))
|
||||
assert_true("screen input is the recorded window, not empty", !str_eq(read_back, ""))
|
||||
assert_eq("read window is byte-identical to written window", read_back, written)
|
||||
|
||||
// ── 5. No false positive: a calm session does not escalate ────────────────────
|
||||
//
|
||||
// A test that only ever asserts "hard_bell" would pass on code that hard-bells
|
||||
// every message. This is the other leg, and it runs BEFORE the anonymous case
|
||||
// below on purpose: that case writes the shared bucket, and under the defect a
|
||||
// calm session would then inherit it.
|
||||
|
||||
println("")
|
||||
println("5. specificity — a calm history does NOT escalate")
|
||||
|
||||
let CALM_SESSION: String = "sess-test-129-calm"
|
||||
state_set("conv_history", "")
|
||||
conv_history_record(CALM_SESSION, "what is the weather like today", "clear and mild", "")
|
||||
let calm: String = agentic_safety_screen(CALM_SESSION, ESCALATION_MSG)
|
||||
assert_eq("calm history stays at soft_bell", json_get(calm, "action"), "soft_bell")
|
||||
|
||||
// ── 6. Cross-session leakage ──────────────────────────────────────────────────
|
||||
//
|
||||
// The same defect had a second face: because the screen read one shared bucket,
|
||||
// a calm session could be scored against a DIFFERENT session's distress. That is
|
||||
// wrong in both directions — it fabricates a crisis for the calm user and it
|
||||
// leaks the distressed user's content into another session's scoring.
|
||||
|
||||
println("")
|
||||
println("6. isolation — one session's distress must not score another session")
|
||||
|
||||
state_set("conv_history", "")
|
||||
let OTHER_SESSION: String = "sess-test-129-other"
|
||||
conv_history_record(OTHER_SESSION, DISTRESS_TURN, "i hear you", "")
|
||||
let isolated: String = agentic_safety_screen(CALM_SESSION, ESCALATION_MSG)
|
||||
assert_eq(
|
||||
"a distressed OTHER session does not escalate the calm session",
|
||||
json_get(isolated, "action"),
|
||||
"soft_bell"
|
||||
)
|
||||
|
||||
// ── 7. Anonymous sessions still work ──────────────────────────────────────────
|
||||
//
|
||||
// conv_hist_key("") deliberately falls back to the shared "conv_history" bucket.
|
||||
// The fix must not break the no-session_id path older callers rely on. Runs last
|
||||
// because it writes that shared bucket.
|
||||
|
||||
println("")
|
||||
println("7. anonymous path — empty session_id still screens against the shared window")
|
||||
|
||||
state_set("conv_history", "[{\"role\":\"user\",\"content\":\"" + DISTRESS_TURN + "\"}]")
|
||||
let anon: String = agentic_safety_screen("", ESCALATION_MSG)
|
||||
assert_eq("anonymous session escalates too", json_get(anon, "action"), "hard_bell")
|
||||
|
||||
// ── Summary ───────────────────────────────────────────────────────────────────
|
||||
|
||||
println("")
|
||||
println("history amplification tests: " + int_to_str(counter_of("t_pass")) + " passed, " + int_to_str(counter_of("t_fail")) + " failed")
|
||||
+97
-11
@@ -41,6 +41,7 @@
|
||||
#include <fcntl.h>
|
||||
#include <dirent.h>
|
||||
#include <errno.h>
|
||||
#include <signal.h> /* SIGPIPE disposition — see el_runtime_ignore_sigpipe */
|
||||
#include <pthread.h>
|
||||
#include <curl/curl.h>
|
||||
|
||||
@@ -1238,16 +1239,77 @@ static const char* http_reason_phrase(int status) {
|
||||
}
|
||||
}
|
||||
|
||||
/* Best-effort send with retry on partial writes. */
|
||||
/* ── A departing client MUST NOT be able to kill the daemon ──────────────────
|
||||
* (2026-08-06, round 9.1 / ADR 0006 item 4.)
|
||||
*
|
||||
* Measured field failure: a client cancelled its request at 25 s; the handler
|
||||
* finished its work at 116.9 s and wrote the reply into the departed client's
|
||||
* socket. The second send() on a reset connection raised SIGPIPE, whose DEFAULT
|
||||
* disposition terminates the process — `exited due to SIGPIPE ... ran for
|
||||
* 361177ms`. launchd respawned 4 ms later, so EVERY other in-flight request on
|
||||
* that daemon lost its work, silently.
|
||||
*
|
||||
* Two independent guards, because one of them can be undone from outside this
|
||||
* file (an embedder may reset signal dispositions) and the other cannot:
|
||||
* 1. process-wide SIGPIPE -> SIG_IGN, installed at runtime init;
|
||||
* 2. per-send suppression at the syscall (MSG_NOSIGNAL where the platform has
|
||||
* it, SO_NOSIGPIPE on the accepted socket on macOS/BSD).
|
||||
* With either in force, send() reports the peer's departure as EPIPE and the
|
||||
* caller decides — which is the point: this is an ordinary I/O outcome, not a
|
||||
* fatal condition.
|
||||
*
|
||||
* It deliberately does NOT swallow the error. http_send_response() below
|
||||
* classifies the errno and logs: "client left" for a departure, and a real
|
||||
* "send failed: <strerror>" for anything else, so a genuine write fault is
|
||||
* still visible in the log (spec round-9.1 §5.3). */
|
||||
|
||||
#ifndef MSG_NOSIGNAL
|
||||
#define MSG_NOSIGNAL 0
|
||||
#endif
|
||||
|
||||
void el_runtime_ignore_sigpipe(void) {
|
||||
static int done = 0;
|
||||
if (done) return;
|
||||
done = 1;
|
||||
struct sigaction sa;
|
||||
memset(&sa, 0, sizeof(sa));
|
||||
sa.sa_handler = SIG_IGN;
|
||||
sigemptyset(&sa.sa_mask);
|
||||
sigaction(SIGPIPE, &sa, NULL);
|
||||
}
|
||||
|
||||
/* Suppress SIGPIPE for one accepted connection (macOS/BSD have no
|
||||
* MSG_NOSIGNAL; they have the socket option instead). Best effort. */
|
||||
static void http_socket_nosigpipe(int fd) {
|
||||
#ifdef SO_NOSIGPIPE
|
||||
int on = 1;
|
||||
setsockopt(fd, SOL_SOCKET, SO_NOSIGPIPE, &on, sizeof(on));
|
||||
#else
|
||||
(void)fd;
|
||||
#endif
|
||||
}
|
||||
|
||||
/* Best-effort send with retry on partial writes.
|
||||
* Returns 0 on success, -1 on failure with errno preserved for the caller. */
|
||||
static int http_send_all(int fd, const char* p, size_t left) {
|
||||
while (left > 0) {
|
||||
ssize_t w = send(fd, p, left, 0);
|
||||
if (w <= 0) return -1;
|
||||
ssize_t w = send(fd, p, left, MSG_NOSIGNAL);
|
||||
if (w < 0) {
|
||||
if (errno == EINTR) continue; /* not an error — retry */
|
||||
return -1; /* errno stays set for caller */
|
||||
}
|
||||
if (w == 0) { errno = EPIPE; return -1; }
|
||||
p += w; left -= (size_t)w;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Did this write fail because the client is gone, or because something is
|
||||
* actually wrong with the socket? Only the first is routine. */
|
||||
static int http_write_err_is_client_gone(int e) {
|
||||
return e == EPIPE || e == ECONNRESET || e == ENOTCONN || e == ESHUTDOWN;
|
||||
}
|
||||
|
||||
/* Discriminator that http_response() embeds at the start of its envelope.
|
||||
* A handler returning a string starting with this exact prefix is treated
|
||||
* as a structured response; anything else is treated as a raw body. */
|
||||
@@ -1468,14 +1530,30 @@ static void http_send_response(int fd, const char* body) {
|
||||
free(env_body); free(hdrs.buf); return;
|
||||
}
|
||||
|
||||
if (http_send_all(fd, status_line, (size_t)sl) == 0
|
||||
&& http_send_all(fd, hdrs.buf, hdrs.len) == 0
|
||||
&& http_send_all(fd, tail, (size_t)tl) == 0
|
||||
&& (head_only
|
||||
/* HEAD requests echo headers + Content-Length but no body. */
|
||||
? 1
|
||||
: http_send_all(fd, eff_body, blen) == 0)) {
|
||||
/* sent successfully */
|
||||
/* The reply is written in four pieces; any of them can find the client
|
||||
* already gone. errno is captured at the first failure, before any later
|
||||
* library call can clobber it, and classified once below. */
|
||||
errno = 0;
|
||||
int send_err = 0;
|
||||
if (http_send_all(fd, status_line, (size_t)sl) != 0) send_err = errno;
|
||||
else if (http_send_all(fd, hdrs.buf, hdrs.len) != 0) send_err = errno;
|
||||
else if (http_send_all(fd, tail, (size_t)tl) != 0) send_err = errno;
|
||||
else if (!head_only /* HEAD echoes headers + Content-Length, no body. */
|
||||
&& http_send_all(fd, eff_body, blen) != 0) send_err = errno;
|
||||
|
||||
if (send_err) {
|
||||
if (http_write_err_is_client_gone(send_err)) {
|
||||
/* ROUTINE. The user closed the window, quit the app, or cancelled.
|
||||
* The work is done and the daemon keeps serving everyone else. */
|
||||
fprintf(stderr, "[http] client left before the reply was written "
|
||||
"(%zu-byte body, %s) - request completed, reply discarded\n",
|
||||
blen, strerror(send_err));
|
||||
} else {
|
||||
/* NOT routine — a real write fault. Never let the client-gone case
|
||||
* above hide this one. */
|
||||
fprintf(stderr, "[http] send failed: %s (%zu-byte body)\n",
|
||||
strerror(send_err), blen);
|
||||
}
|
||||
}
|
||||
|
||||
if (env_parsed_root) el_release(env_parsed_root);
|
||||
@@ -1491,6 +1569,7 @@ static void* http_worker(void* arg) {
|
||||
HttpWorkerArg* a = (HttpWorkerArg*)arg;
|
||||
int fd = a->fd;
|
||||
free(a);
|
||||
http_socket_nosigpipe(fd);
|
||||
char *method = NULL, *path = NULL, *body = NULL;
|
||||
if (http_read_request(fd, &method, &path, &body, NULL) == 0) {
|
||||
http_handler_fn h = http_lookup_active();
|
||||
@@ -1531,6 +1610,7 @@ static void* http_worker(void* arg) {
|
||||
}
|
||||
|
||||
void http_serve(el_val_t port, el_val_t handler) {
|
||||
el_runtime_ignore_sigpipe(); /* serving implies clients that leave */
|
||||
/* If `handler` looks like a string name, register it as the active handler. */
|
||||
const char* hname = EL_CSTR(handler);
|
||||
if (hname && looks_like_string(handler)) {
|
||||
@@ -1634,6 +1714,7 @@ static void* _http_serve_async_loop(void* raw) {
|
||||
}
|
||||
|
||||
void http_serve_async(el_val_t port, el_val_t handler) {
|
||||
el_runtime_ignore_sigpipe(); /* serving implies clients that leave */
|
||||
const char* hname = EL_CSTR(handler);
|
||||
if (hname && looks_like_string(handler)) {
|
||||
http_set_handler(handler);
|
||||
@@ -1821,6 +1902,7 @@ static void* http_worker_v2(void* arg) {
|
||||
HttpWorkerArg* a = (HttpWorkerArg*)arg;
|
||||
int fd = a->fd;
|
||||
free(a);
|
||||
http_socket_nosigpipe(fd);
|
||||
char *method = NULL, *path = NULL, *body = NULL, *hdr_block = NULL;
|
||||
if (http_read_request(fd, &method, &path, &body, &hdr_block) == 0) {
|
||||
http_handler4_fn h = http_lookup_active_v2();
|
||||
@@ -1858,6 +1940,7 @@ static void* http_worker_v2(void* arg) {
|
||||
}
|
||||
|
||||
void http_serve_v2(el_val_t port, el_val_t handler) {
|
||||
el_runtime_ignore_sigpipe(); /* serving implies clients that leave */
|
||||
const char* hname = EL_CSTR(handler);
|
||||
if (hname && looks_like_string(handler)) {
|
||||
http_set_handler_v2(handler);
|
||||
@@ -5511,6 +5594,9 @@ el_val_t getpid_now(void) {
|
||||
static el_val_t _el_args_list = 0;
|
||||
|
||||
void el_runtime_init_args(int argc, char** argv) {
|
||||
/* First line of every generated main(): a client that leaves must never be
|
||||
* able to signal this process to death. See el_runtime_ignore_sigpipe. */
|
||||
el_runtime_ignore_sigpipe();
|
||||
_el_args_list = el_list_empty();
|
||||
for (int i = 1; i < argc; i++) {
|
||||
_el_args_list = el_list_append(_el_args_list, EL_STR(argv[i]));
|
||||
|
||||
Reference in New Issue
Block a user