Merge pull request 'nsbx: one-command dev onboarding (branch + worktree + isolated engram)' (#99) from feat/nsbx-dev-env into dev
El SDK CI - dev / build-and-test (push) Failing after 13m56s

This commit was merged in pull request #99.
This commit is contained in:
2026-08-15 19:52:35 +00:00
7 changed files with 2295 additions and 8 deletions
+577 -8
View File
@@ -3240,6 +3240,37 @@ static void jb_init(JsonBuf* b) {
b->buf[0] = '\0';
}
/* jb_init_cap — jb_init with a caller-supplied starting capacity.
*
* WHY THIS EXISTS (2026-08-11 self-review). jb_init starts at 64 BYTES and
* jb_reserve grows by doubling. That is right for the hundreds of small JSON
* responses this runtime builds per minute and catastrophic for the one that
* is 64 MEGABYTES: serializing the canonical snapshot walked the buffer
* 64B 128B ... 128MB, about twenty reallocs, each copying everything
* written so far. Roughly 128MB of memcpy per save, and the part that
* actually hurt a fresh large span from the allocator every time.
*
* MEASURED (13,129 nodes / 43,400 edges, macOS arm64): RSS climbed +63MB per
* snapshot write, linearly, 14 for 14 writes, no plateau 204MB to 1,028MB.
* `leaks` reported only 15KB genuinely unreachable, which is what makes this
* subtle: nothing is leaked in the reachable/unreachable sense. engram_save
* frees b.buf correctly on every path. The growth is the allocator declining
* to return large freed spans to the OS, and the doubling walk guaranteeing
* that each save asks for a differently-sized region than the last free made
* available. Every durable write path calls this node create, edge create,
* the Hebbian batch write-back so on the live daemon it grows without bound
* until the process dies.
*
* The fix is to ask for the right size once. With a stable capacity the
* allocator hands back the same span on every save and RSS flattens. */
static void jb_init_cap(JsonBuf* b, size_t cap) {
if (cap < 64) cap = 64;
b->cap = cap; b->len = 0;
b->buf = malloc(b->cap);
if (!b->buf) { fputs("el_runtime: out of memory\n", stderr); exit(1); }
b->buf[0] = '\0';
}
static void jb_reserve(JsonBuf* b, size_t add) {
if (b->len + add + 1 > b->cap) {
while (b->len + add + 1 > b->cap) b->cap *= 2;
@@ -6516,6 +6547,26 @@ static float* _eg_ctx_c = NULL;
static int32_t _eg_ctx_dim = 0;
static double _eg_act_ctx_cos = -2.0;
/* Fan-effect gauges (2026-08-11 self-review). Per-call, like ctx_cos: they
* describe THIS activation, not process history. Without these the degree
* normalization is an unobservable change to the most important scoring path
* in the runtime, and "did it do anything" would be unanswerable which is
* exactly the failure the Hebbian learning rate had before it was measured.
* fan_mean mean applied factor over every propagation step. 1.0 means the
* correction never bound (graph is flat, or d_ref is above every
* pair's geometric mean degree). Falling toward FAN_MIN means
* traversal is running through hubs.
* fan_min_seen / fan_hits the worst single penalty and how many steps were
* penalized at all, so a low mean caused by one pathological hub
* is distinguishable from broad hub saturation.
* fan_dref the live mean degree the correction is calibrated against;
* publishing it makes densification visible over time. */
static double _eg_act_fan_sum = 0.0;
static double _eg_act_fan_min = 1.0;
static int64_t _eg_act_fan_n = 0;
static int64_t _eg_act_fan_hits = 0;
static double _eg_act_fan_dref = 0.0;
static int _eg_embed_consec_fail = 0;
static int64_t _eg_embed_breaker_until = 0;
@@ -6534,6 +6585,27 @@ static int64_t _eg_embed_breaker_until = 0;
* rates keep the previous reading and diff. Restart legitimately resets to 0. */
static int64_t _eg_act_breakthroughs = 0; /* forced promotions at the floor, cumulative */
static int64_t _eg_act_wm_evicted = 0; /* ALL WM evictions, cumulative (see below) */
/* ── Eviction CAUSE decomposition (2026-08-14 self-review) ──────────────────
* _eg_act_wm_evicted is incremented from six sites with four distinct causes,
* and every one of them collapsed into that single integer. Today's review
* measured 175,547 evictions over 13.5h (~216/min against 24 slots) and could
* not tell healthy rotation from cap thrashing from duplicate churn, because
* the only available number counts all three the same way.
*
* That is this file's most-repeated defect. The 08-02 and 08-06 reviews were
* each diagnosable only because someone first added a NEW gauge; dup_wm and
* dup_wm_global exist precisely because the aggregate could not answer "why".
* These three finish the decomposition, so that
* evicted == floor + cap + bll + dup_wm + dup_wm_global
* holds as an identity and each term names a different corrective action:
* floor - candidates below the absolute admission bar. High = weak retrieval.
* cap - lost the rank contest for 24 slots. High = genuine contention.
* bll - carried-over residents that decayed under the ACT-R tau. High =
* healthy forgetting, NOT pressure.
* Confusing the third with the second is what makes WM churn unreadable. */
static int64_t _eg_act_evict_floor = 0; /* below ENGRAM_WM_FLOOR (both passes) */
static int64_t _eg_act_evict_cap = 0; /* over ENGRAM_WM_CAP (both passes) */
static int64_t _eg_act_evict_bll = 0; /* carry-over decayed under BLL tau */
/* Redundancy suppression counters (2026-08-05 self-review) — see
* ENGRAM_DEDUP_COS. dup_seeds = semantic seed slots reclaimed from redundant
* copies; dup_wm = WM candidates dropped for duplicating a higher-ranked
@@ -6874,6 +6946,10 @@ typedef struct EngramStore {
int* adj_to_len;
int adj_dirty; /* 1 = rebuild needed before next BFS */
int64_t adj_node_count; /* node_count at time of last adj_rebuild */
/* Nodes with degree >= 1 at last adj_rebuild. The denominator for the
* fan-effect reference degree see eg_fan_factor for why isolated nodes
* must not be counted. (2026-08-11 self-review) */
int64_t adj_connected;
} EngramStore;
static EngramStore* engram_global = NULL;
@@ -7237,11 +7313,16 @@ static void engram_adj_rebuild(EngramStore* g) {
if (ti >= 0 && g->adj_to[ti])
g->adj_to[ti][to_pos[ti]++] = (int)ei;
}
/* Copy counts */
/* Copy counts. Also tally how many nodes have any edge at all — the
* fan-effect denominator. Free here, in the O(V) pass that already exists,
* rather than as a separate scan. (2026-08-11 self-review) */
int64_t connected = 0;
for (int64_t i = 0; i < g->node_count; i++) {
g->adj_from_len[i] = from_cnt[i];
g->adj_to_len[i] = to_cnt[i];
if (from_cnt[i] + to_cnt[i] > 0) connected++;
}
g->adj_connected = connected;
free(from_cnt); free(to_cnt); free(from_pos); free(to_pos);
g->adj_node_count = g->node_count;
g->adj_dirty = 0;
@@ -8496,6 +8577,7 @@ static void eg_wm_carry_over(EngramNode* cn, int64_t now_ms, int64_t* evict_ctr)
cn->working_memory_weight = 0.0;
cn->wm_anchor = 0.0;
if (evict_ctr) (*evict_ctr)++;
_eg_act_evict_bll++;
} else {
cn->working_memory_weight = w;
}
@@ -8662,6 +8744,108 @@ static double engram_activation_dampen(const EngramNode* n) {
return 1.0 / (1.0 + log(1.0 + (double)n->activation_count));
}
/* ── ACT-R fan effect: degree normalization for spreading activation ─────────
* (2026-08-11 self-review. Closes the other half of a mechanism that has been
* half-implemented since the BLL work.)
*
* THE GAP. This runtime implements ACT-R's base-level learning term
* B_i = ln(Σ t_k^-d) (engram_bll_base_level) but never implemented the
* ASSOCIATIVE term that goes with it:
*
* A_i = B_i + Σ_j W_j · S_ji where S_ji = S ln(fan_j)
*
* fan_j is the number of things j is associated with. The whole point of the
* fan effect (Anderson 1974; Anderson & Reder 1999) is that a source spreads a
* FIXED budget of activation across its associations so being connected to
* many things makes each individual connection weaker. Without it, degree is
* pure advantage: a node wins retrieval by being popular rather than by being
* relevant. That is backwards, and it is what this graph has been doing.
*
* MEASURED ON THE LIVE STORE (13,129 nodes / 43,400 edges, 2026-08-11):
* degree p50=14 p90=34 p95=82 p99=275 max=357 mean=23.3
* the top 1% of nodes by degree touch 21.2% of all edges
* So the most-connected node had a 25x propagation advantage over the median
* node for no reason other than accumulated connections. The top hubs are not
* even semantically central several are duplicate pairs of the same document
* left over from the redundancy census of the 2026-08-05 review.
*
* The hub problem was already recognized twice and patched narrowly both
* times: InternalStateEvent nodes were cut out of propagation entirely (see
* the frontier loop) and eg_hebb_node_budget caps per-node Hebbian mass. Both
* are special cases of this general law. This is the general fix.
*
* FORM. Symmetric normalization, w / (deg(u)^β · deg(v)^β) with β = 0.5 the
* normalized-Laplacian / GCN form, which penalizes a hub both for sending and
* for receiving. Both failure modes are live here: a hub source floods its
* neighborhood, and a hub target gets reached by everything regardless of
* relevance. Written relative to the graph's own mean degree:
*
* fan(u,v) = clamp( d_ref / sqrt(deg(u) · deg(v)), FAN_MIN, 1.0 )
* d_ref = 2·|E| / |V| (mean degree, O(1), live)
*
* WHY IT IS CLAMPED AT 1.0 ON TOP this is the load-bearing safety property,
* not a detail. The factor can only ever REDUCE propagation, never amplify it.
* Every constant downstream of this multiply is calibrated against today's
* activation magnitudes: the 0.02 firing threshold, SPREAD_DECAY = 0.7, the
* 0.15 WM promotion threshold, the 24-slot WM cap. A normalization that
* boosted low-degree nodes would inflate the frontier, change how many nodes
* clear 0.02, and silently recalibrate working memory as a side effect of a
* change that was supposed to be about hubs. Capping at 1.0 means every pair
* at or below mean degree the common case propagates EXACTLY as it does
* today, and the only behavior that changes is that above-mean hubs stop
* winning on degree alone. Strictly monotone, strictly conservative, and the
* blast radius is confined to the nodes the change is aimed at.
*
* Self-calibrating: d_ref is recomputed from the live graph, so the correction
* tracks densification instead of drifting against a constant that was right
* in August 2026 and wrong a year later. Change is the signal.
*
* FAN_MIN = 0.30 bottoms the penalty at ~3.3x rather than the ~15x that raw
* 1/deg would give at max degree. Same reasoning as ENGRAM_QGATE_FLOOR: damp
* the uninformative path, never sever it. A hub is usually a hub for a reason;
* it just should not also get a free win.
*
* Sources: Anderson & Reder 1999 (fan effect, S=1.6-2.0, d=0.5) ·
* arXiv:2405.14831 HippoRAG (node specificity) · Systems 9(2):22
* (normalized-Laplacian spreading activation) · arXiv:2606.30133 (β is a
* low-sensitivity knob; gating and fan normalization carry the effect). */
/* FAN_MIN 0.50, not the 0.30 this shipped as on the first build. Measured on
* the live graph, β=0.5 with a 0.30 floor damped 96% of propagation steps to a
* mean factor of 0.34 and that number is not a bug in the correction, it is
* an honest measurement of how hub-dominated traversal here actually is. But a
* ~3x near-uniform damp is a bigger global change than one A/B run justifies,
* and it cost a working-memory promotion (5 4) on the one query measured
* cleanly. A 0.50 floor keeps the full mechanism and the whole [0.5, 1.0]
* dynamic range for separating hubs from non-hubs, at half the blast radius.
* The fan_mean / fan_hits gauges make the next review's tuning evidence-based
* rather than another guess: loosen it when the data says WM can afford it. */
#define ENGRAM_FAN_MIN 0.50
/* eg_node_degree — total (in + out) degree from the adjacency index. The index
* is rebuilt at the top of engram_activate whenever topology changed, so this
* is current. adj_node_count is the count at BUILD time and can lag
* node_count; out-of-range indices report 0 and are treated as unpenalized. */
static int eg_node_degree(const EngramStore* g, int64_t idx) {
if (idx < 0 || idx >= g->adj_node_count) return 0;
if (!g->adj_from_len || !g->adj_to_len) return 0;
return g->adj_from_len[idx] + g->adj_to_len[idx];
}
static double eg_fan_factor(const EngramStore* g, double d_ref,
int64_t u_idx, int64_t v_idx) {
if (d_ref <= 0.0) return 1.0;
int du = eg_node_degree(g, u_idx);
int dv = eg_node_degree(g, v_idx);
/* Degree 0 is only reachable when the adjacency index is stale or absent;
* an actually-isolated node is never on the frontier. Do not penalize what
* we cannot measure. */
if (du <= 0 || dv <= 0) return 1.0;
double f = d_ref / sqrt((double)du * (double)dv);
if (f > 1.0) return 1.0; /* never amplify — see above */
if (f < ENGRAM_FAN_MIN) return ENGRAM_FAN_MIN;
return f;
}
/* Temporal proximity bonus: boost propagation along edges connecting
* co-temporal nodes. Returns a multiplier bonus in [0, 0.2]. */
static double engram_temporal_proximity_bonus(int64_t node_created,
@@ -8829,6 +9013,8 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
* miss nearly all events between beats; see the definition site).
* ctx_cos stays per-call: it is a gauge of THIS query vs the centroid. */
_eg_act_ctx_cos = -2.0;
_eg_act_fan_sum = 0.0; _eg_act_fan_min = 1.0;
_eg_act_fan_n = 0; _eg_act_fan_hits = 0;
/* ── Embedding backfill + query embedding (2026-07-24, bl-b2d1c944) ──
* Backfill: embed up to N un-embedded eligible nodes per call, newest
@@ -9061,6 +9247,29 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
ftail++;
}
const double SPREAD_DECAY = 0.7;
/* Reference degree for the fan-effect correction: mean degree over
* CONNECTED nodes, 2|E| / |{v : deg(v) > 0}|. O(1) adj_connected is
* tallied during adjacency rebuild.
*
* NOT 2|E|/|V|. That was the first cut and instrumentation caught it
* immediately: on the live graph it gives d_ref = 6.61, while the median
* degree of a node that actually has edges is 14. Isolated nodes cannot
* be on the frontier spreading activation only ever traverses connected
* ones so including them in the denominator deflates the reference below
* anything traversal will ever see, and the correction pins to
* ENGRAM_FAN_MIN on every step. Measured on the first build:
* fan_mean 0.3026 with fan_hits 579/579 a uniform 0.30 multiplier, which
* is not a fan effect at all. It is just a weaker SPREAD_DECAY, and it
* would have quietly recalibrated the 0.02 firing threshold and WM
* competition while appearing to be a targeted change.
*
* Over connected nodes the reference is ~23, above the median, so typical
* traversal rides the 1.0 cap unchanged and only genuine hubs are damped
* which is the whole intent. The gauge that caught this is the reason it
* was worth adding the gauge. */
const double FAN_DREF = (g->adj_connected > 0)
? (2.0 * (double)g->edge_count / (double)g->adj_connected) : 0.0;
_eg_act_fan_dref = FAN_DREF;
while (fhead < ftail) {
Frontier f = fr[fhead++];
if (f.hops >= max_depth) continue;
@@ -9127,16 +9336,51 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
* ~4x, never killed); unembedded targets pass ungated (no
* information, no penalty); cosq == NULL (embedder down) means
* no gating at all same graceful degradation as seeding. */
/* Rescale before gating (2026-08-14 self-review). Raw cosine from
* nomic-embed is compressed into a narrow high band, so feeding it
* to the gate directly makes the gate nearly a constant. Measured
* on this store: 400 random UNRELATED node pairs gave median 0.562,
* central 98% span [0.381, 0.743]. The raw gate therefore passed a
* typical unrelated node at 0.25 + 0.75*0.562 = 0.67 two thirds
* strength for a node with no semantic relation to the query. That
* is not a gate, it is a small tax.
*
* Shift-and-floor about ENGRAM_EMBED_S0, exactly as the Pass-2 WM
* term at ENGRAM_EMBED_WM_WEIGHT already does. The constant was in
* this file for this reason; the propagation gate simply never used
* it. Same store, same 400 pairs, after the rescale: the median
* unrelated pair drops to 0.40 while the top of the range is
* preserved (0.85 vs 0.92), and gate spread widens 0.42 -> 0.60.
* Only 8.5% of pairs fall to the floor, so lexical/structural
* pathways through dissimilar nodes are damped, never severed.
* Cf. arXiv:2512.15922, which rescales w' = (w-c)/(1-c) about
* c = 0.4 for precisely this reason ("prevent overactivation and
* context explosion"). */
double qgate = 1.0;
if (cosq && cosq[oi] > -1.5) {
double c = cosq[oi] > 0.0 ? cosq[oi] : 0.0;
double c = (cosq[oi] - ENGRAM_EMBED_S0) / (1.0 - ENGRAM_EMBED_S0);
if (c < 0.0) c = 0.0;
if (c > 1.0) c = 1.0;
qgate = ENGRAM_QGATE_FLOOR + (1.0 - ENGRAM_QGATE_FLOOR) * c;
}
/* ── ACT-R fan effect (2026-08-11 self-review) ──
* Symmetric degree normalization over the (source, target) pair.
* The query gate above prunes branches that are semantically
* irrelevant; this prunes branches that are merely POPULAR. They
* are different failure modes a duplicate document with 357
* edges can be highly cosine-similar to the query and still be
* the wrong thing to spread through. Only ever <= 1.0, so it
* cannot inflate the frontier. See eg_fan_factor. */
double fan = eg_fan_factor(g, FAN_DREF, cur, oi);
_eg_act_fan_sum += fan;
_eg_act_fan_n++;
if (fan < 1.0) _eg_act_fan_hits++;
if (fan < _eg_act_fan_min) _eg_act_fan_min = fan;
/* eg_edge_eff_weight, not e->weight: edges that have repeatedly
* carried co-activated pairs propagate more strongly. Identity on
* an unlearned edge. (2026-08-04 self-review.) */
double new_act = f.act * eg_edge_eff_weight(e) * SPREAD_DECAY
* (1.0 + tbonus) * tdecay * dampen * qgate;
* (1.0 + tbonus) * tdecay * dampen * qgate * fan;
/* Firing threshold per classic spreading-activation: sub-threshold
* activation neither updates the target nor enqueues it, so weak
* signals die out instead of flooding the whole graph with tiny
@@ -9426,6 +9670,7 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
if (wm_weights[i] > 0.0 && wm_weights[i] < ENGRAM_WM_FLOOR) {
wm_weights[i] = 0.0;
_eg_act_wm_evicted++;
_eg_act_evict_floor++;
}
}
int64_t cap_count = 0;
@@ -9461,6 +9706,7 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
}
wm_weights[i] = 0.0; /* over cap: evict */
_eg_act_wm_evicted++;
_eg_act_evict_cap++;
}
}
/* If malloc failed, skip cap — WM unbounded this call, no corruption. */
@@ -9572,6 +9818,7 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
fn->working_memory_weight = 0.0;
fn->wm_anchor = 0.0;
_eg_act_wm_evicted++;
_eg_act_evict_floor++;
}
}
/* ── Global redundancy suppression (2026-08-06 self-review) ──────────
@@ -9680,6 +9927,7 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
n->working_memory_weight = 0.0; /* evict: over global cap */
n->wm_anchor = 0.0; /* keep anchor coherent */
_eg_act_wm_evicted++; /* was uncounted before 2026-08-02 */
_eg_act_evict_cap++;
}
}
/* If malloc failed, skip — WM over cap this call, no data corruption. */
@@ -10109,11 +10357,24 @@ static void engram_emit_edge_json(JsonBuf* b, const EngramEdge* e) {
jb_putc(b, '}');
}
/* Size of the last snapshot this process serialized. Seeds the next save's
* buffer so the doubling walk never runs on the big document. See jb_init_cap
* for the measurement that motivated it. (2026-08-11 self-review) */
static size_t _eg_save_cap_hint = 0;
el_val_t engram_save(el_val_t path) {
const char* p = EL_CSTR(path);
if (!p || !*p) return 0;
EngramStore* g = engram_get();
JsonBuf b; jb_init(&b);
/* Pre-size from the previous save plus 12.5% headroom, so ordinary growth
* between snapshots does not trigger a realloc and the request size stays
* stable enough for the allocator to reuse the same span. First save of
* the process has no hint and starts at 1MB still 14 doublings better
* than 64 bytes. */
JsonBuf b;
jb_init_cap(&b, _eg_save_cap_hint
? _eg_save_cap_hint + (_eg_save_cap_hint >> 3) + 1024
: (size_t)1 << 20);
jb_puts(&b, "{\"nodes\":[");
for (int64_t i = 0; i < g->node_count; i++) {
if (i > 0) jb_putc(&b, ',');
@@ -10153,6 +10414,10 @@ el_val_t engram_save(el_val_t path) {
jb_putc(&b, '}');
}
jb_puts(&b, "]}");
/* Remember the size BEFORE the write: the hint is about how much buffer
* the next serialization needs, which is a property of the graph, not of
* whether this particular fopen succeeded. */
_eg_save_cap_hint = b.len;
FILE* f = fopen(p, "wb");
if (!f) { free(b.buf); return 0; }
size_t w = fwrite(b.buf, 1, b.len, f);
@@ -11727,8 +11992,11 @@ el_val_t engram_act_stats_json(void) {
}
/* 768, not 512: the write-back gauges added 2026-08-07 push the worst-case
* rendering past the old bound, and snprintf would truncate the JSON into
* an unparseable tail rather than fail loudly. */
char buf[896];
* an unparseable tail rather than fail loudly.
* 1152, not 896: the five fan-effect gauges added 2026-08-11 add ~90 bytes
* worst-case. Same reasoning headroom is cheaper than a truncated tail
* that every downstream JSON parser rejects as a whole. */
char buf[1152];
/* ctx_cos (2026-07-29): cos(query, context centroid) at the LAST
* activate call, measured before the query was folded in. ~1.0 =
* context aligned with current query; low = divergence (expected at
@@ -11736,6 +12004,15 @@ el_val_t engram_act_stats_json(void) {
* embedder down. The drift gauge for the context-centroid mechanism. */
snprintf(buf, sizeof(buf),
"{\"wm_evicted\":%lld,\"breakthroughs\":%lld,"
/* Eviction cause decomposition (2026-08-14 self-review):
* wm_evicted == evict_floor + evict_cap + evict_bll
* + dup_wm + dup_wm_global.
* Read them as a ratio, not a level. cap-dominant = real
* contention for the 24 slots; bll-dominant = healthy decay of
* carried-over residents; floor-dominant = retrieval is returning
* weak candidates. The aggregate alone cannot distinguish these
* and every prior WM incident needed a new gauge to diagnose. */
"\"evict_floor\":%lld,\"evict_cap\":%lld,\"evict_bll\":%lld,"
"\"embed_breaker_open\":%d,\"embed_consec_fail\":%d,"
"\"ctx_cos\":%.3f,"
"\"hebb_edges\":%lld,\"hebb_max\":%.4f,\"hebb_mass\":%.3f,"
@@ -11749,9 +12026,19 @@ el_val_t engram_act_stats_json(void) {
* any climb means a write path is mangling text again. Cheap
* (counted at creation) the full census lives in
* engram_text_health_json. (2026-08-08 self-review) */
"\"txt_damaged\":%lld}",
"\"txt_damaged\":%lld,"
/* Fan-effect gauges (2026-08-11 self-review) — see the
* _eg_act_fan_* definitions. fan_mean == 1.0 with fan_hits == 0
* means the degree correction never bound on the last activation;
* a mean drifting toward ENGRAM_FAN_MIN means traversal is
* running through hubs and the correction is doing work. */
"\"fan_mean\":%.4f,\"fan_min\":%.4f,\"fan_hits\":%lld,"
"\"fan_steps\":%lld,\"fan_dref\":%.2f}",
(long long)_eg_act_wm_evicted,
(long long)_eg_act_breakthroughs,
(long long)_eg_act_evict_floor,
(long long)_eg_act_evict_cap,
(long long)_eg_act_evict_bll,
breaker_open, _eg_embed_consec_fail,
_eg_act_ctx_cos,
(long long)hebb_edges, hebb_max, hebb_mass,
@@ -11761,7 +12048,10 @@ el_val_t engram_act_stats_json(void) {
(long long)_eg_hebb_wb_dropped,
(long long)_eg_act_dup_seeds, (long long)_eg_act_dup_wm,
(long long)_eg_act_dup_wm_global,
(long long)_eg_txt_write_damaged);
(long long)_eg_txt_write_damaged,
(_eg_act_fan_n > 0 ? _eg_act_fan_sum / (double)_eg_act_fan_n : 1.0),
_eg_act_fan_min, (long long)_eg_act_fan_hits,
(long long)_eg_act_fan_n, _eg_act_fan_dref);
return el_wrap_str(el_strdup(buf));
}
@@ -11894,6 +12184,285 @@ el_val_t engram_label_df(el_val_t term) {
return (el_val_t)df;
}
/* ── Salient-term extraction (2026-08-13 self-review) ────────────────────────
* THE MEASUREMENT. auto_term_empty_streak, the counter added by the 2026-08-06
* review precisely to catch this class of silent death, read 50 and climbing.
* Fifty consecutive curiosity scans in which the soul's dynamic seeding path
* produced NOTHING and the loop fell back to its four hardcoded rotating
* phrases. Dumping the live WM top says why in one look:
*
* Memory 0.390 memory:remembered
* Memory 0.378 memory:remembered
* Memory 0.377 memory:remembered
* Memory 0.373 memory:remembered
* Memory 0.370 memory:remembered
*
* Every slot at the top of working memory is a Memory node, and every Memory
* node written by remember() carries the sentinel label "memory:remembered".
* auto_term_try_slot reads the LABEL and only the label; the colon-no-space
* guard (correctly) rejects sentinels as carrying no seed signal; so the
* extractor had nothing to work with and returned empty, forever.
*
* THE ACTUAL DEFECT is not the sentinel guard that guard is right. It is
* that the extractor was built against Knowledge nodes, which have real
* titles, and is structurally blind to the node type that in fact dominates
* working memory. The label is not the content. A Memory node's topic is in
* its text; the runtime just never looked there.
*
* WHY NOT ANOTHER GUARD. The extractor's whole history is guards: genre words
* (07-23), quoted titles (07-25), English stopwords (07-30), label-df
* (08-03). Four reviews, four blocklists, each written after watching a flood
* happen. That is a losing shape, and 08-03 said so explicitly before adding
* the fifth. The reason it keeps recurring is the algorithm underneath:
* TAKE THE FIRST WORD, THEN CHECK WHETHER IT IS ACCEPTABLE. A first-word
* extractor has no notion of term quality, so quality has to be bolted on as
* rejection, and rejection can only encode the past.
*
* THE FIX is to invert it: score EVERY candidate token in the text and take
* the argmax. Then term quality is the selection criterion rather than a
* veto, and a bad token does not need to be on a list to lose it only needs
* a better token in the same text, which is the common case.
*
* SCORING (YAKE, Campos et al., Information Sciences 509:257-289, 2020
* lightweight unsupervised single-document keyword extraction). YAKE scores
* candidates on casing, position, frequency, context relatedness and sentence
* dispersion, and beats RAKE/TextRank/SingleRank across twenty datasets. Two
* of its five features port directly and cheaply; the other three are
* within-document proxies for a corpus YAKE deliberately does not have. This
* system DOES have the corpus 12.7k labelled nodes so real IDF is
* substituted where YAKE has to approximate:
*
* score(t) = idf(t) · position(t) · casing(t)
*
* idf = ln((N+1)/(df+1)) real corpus specificity (Spärck
* Jones 1972), strictly better than
* YAKE's TF-based stand-in
* position = 1/ln(e + i) YAKE T_Position: earlier tokens are
* more topical. Keeps the old
* first-word bias as a SOFT preference
* instead of an absolute rule
* casing = 1.30 acronym / 1.15 capitalised / 1.00 otherwise
* YAKE T_Case
*
* THE min_df GATE. The df ceiling (08-03) rejects corpus-frequent markup and
* sentinels. A floor was added alongside it for an independent reason: a term
* appearing in ZERO labels cannot lexically reach anything, so it is a bad
* seed however specific it looks.
*
* An earlier draft of this comment claimed the floor also subsumes the 73
* hand-listed stopwords that 08-03 measured label-df as missing (Whose:0,
* Would:0, Could:0). MEASURED, AND THAT CLAIM IS FALSE. Under word-boundary
* df on the live store, function words are rare in labels but not absent:
* about:2, whole:1, them:2, head:2. They clear a floor of 1. What actually
* keeps them from winning is the argmax itself they carry no position
* advantage and lose to a topical term in the same text on every node
* measured. The stopword list therefore STAYS as a real defense for the
* Title-case cases, not as vestigial belt-and-braces. Recording the
* correction rather than the tidier story: the floor buys lexical
* reachability, the argmax buys quality, and the list still earns its keep.
*
* TABU IS APPLIED DURING THE ARGMAX, not after it. The old code picked a term
* and then discarded it if it was tabu, which turned inhibition-of-return
* into another source of empty scans. Excluding tabu terms from the candidate
* set instead yields the best NON-TABU term, so rotation costs quality rather
* than costing the whole scan.
*
* COST. One pass over g->nodes scoring all candidates at once (12.7k labels ×
* <=32 candidates, short strings, good locality), twice per 30 s scan.
*
* POLICY LIVES IN THE SOUL. Thresholds arrive as arguments; the runtime
* measures and ranks, awareness.el decides. Same split as engram_label_df.
*
* Returns the winning token, or "" when the node is missing, has no usable
* text, or every candidate is gated out "" remains the honest signal that
* this slot yielded no seed, and auto_term_empty_streak still counts it. */
#define ENGRAM_ST_MAXCAND 32
#define ENGRAM_ST_TOKLEN 64
#define ENGRAM_ST_SCANCHARS 400
/* Trim leading/trailing non-alphanumerics, then accept only tokens whose core
* is alphanumeric plus '-' and '_' with at least 3 letters. This subsumes the
* quoted-title guard (2026-07-25) and the "<!--" flood (2026-08-03)
* structurally: markup and punctuation-bearing tokens never become
* candidates, rather than being blocklisted after the fact. */
static int eg_st_clean_token(const char* raw, size_t rawlen,
char* out, size_t outcap) {
size_t s = 0, e = rawlen;
while (s < e && !isalnum((unsigned char)raw[s])) s++;
while (e > s && !isalnum((unsigned char)raw[e - 1])) e--;
size_t len = e - s;
if (len < 4 || len >= outcap) return 0;
int alpha = 0;
for (size_t i = 0; i < len; i++) {
unsigned char c = (unsigned char)raw[s + i];
if (isalpha(c)) alpha++;
else if (!isdigit(c) && c != '-' && c != '_') return 0;
}
if (alpha < 3) return 0;
memcpy(out, raw + s, len);
out[len] = '\0';
return 1;
}
/* ENGRAM_ST_DEBUG=1 dumps the full scored candidate set to stderr. One
* cached branch in production. This exists because the first live run of this
* function returned five ALL-CAPS terms in a row and there was no way to see
* whether that was the corpus or the casing weight without guessing the
* lesson this system keeps relearning. */
static int _eg_st_debug(void) {
static int v = -1;
if (v < 0) { const char* e = getenv("ENGRAM_ST_DEBUG"); v = (e && *e == '1'); }
return v;
}
/* Word-boundary document frequency. engram_label_df uses istr_contains, i.e.
* SUBSTRING matching, and that is the wrong estimator for term specificity on
* short tokens: "them" hits inside "theme" and "anthem", "about" and "whole"
* come back with df 2 and 1 rather than 0. That matters here specifically
* because the min_df floor is what rejects English function words, and it can
* only do that job if their df is honestly zero. Substring df quietly handed
* them a survival ticket. Measured on the live store before this fix, "whole"
* (df=1, idf=8.76) and "about" (df=2, idf=8.36) were outscoring real topical
* terms and losing only on position one node whose text happened to open
* with a function word would have seeded on it.
*
* engram_label_df keeps substring semantics: it is a separate published
* measure with existing callers, and changing it underneath them is not this
* change's business. */
static int eg_st_label_has_word(const char* hay, const char* word) {
size_t wl = strlen(word);
for (const char* p = hay; *p; p++) {
if (strncasecmp(p, word, wl) != 0) continue;
char before = (p == hay) ? '\0' : p[-1];
char after = p[wl];
if (before && (isalnum((unsigned char)before) || before == '_')) continue;
if (after && (isalnum((unsigned char)after) || after == '_')) continue;
return 1;
}
return 0;
}
/* YAKE T_Case, adapted to this corpus. YAKE up-weights all-caps tokens
* because in ordinary prose an acronym is rare and carries topic. That
* assumption does not hold here: memory content written by remember()
* conventionally OPENS WITH AN ALL-CAPS HEADER ("FRAME-ROUTER UPGRADE —
* RESULTS", "THE GAP", "CENSUS"), so a flat acronym bonus systematically
* hands the seed to whatever word the header happens to start with and lets
* casing override the specificity signal it is supposed to only nudge.
* Measured on the live store: the first five WM nodes returned PRIMING,
* CONVERSATION, OCCUPATION, RELATIONAL, SELF-OCCUPATION every one an
* all-caps header word, none chosen on its merits.
*
* Genuine acronyms are SHORT (VBD, CCR, MCP, HTTP); shouty headers are long
* words that happen to be capitalised. So the acronym bonus is restricted to
* tokens of <= 5 characters, where all-caps is actually evidence of an
* acronym rather than evidence of a heading. Longer all-caps tokens fall
* through to the ordinary Title-case nudge they still compete, they just
* compete on specificity instead of on volume. */
static double eg_st_casing(const char* t) {
int upper = 0, lower = 0;
size_t len = 0;
for (const char* q = t; *q; q++, len++) {
if (isupper((unsigned char)*q)) upper++;
else if (islower((unsigned char)*q)) lower++;
}
if (lower == 0 && upper >= 2 && len <= 5) return 1.30; /* acronym */
if (isupper((unsigned char)t[0])) return 1.15; /* Title/hdr */
return 1.0;
}
el_val_t engram_salient_term(el_val_t node_id, el_val_t max_df_v,
el_val_t min_df_v, el_val_t tabu_v) {
EngramStore* g = engram_get();
int64_t ix = engram_find_node_index(EL_CSTR(node_id));
if (ix < 0) return el_wrap_str(el_strdup(""));
EngramNode* n = &g->nodes[ix];
int64_t max_df = (int64_t)max_df_v;
int64_t min_df = (int64_t)min_df_v;
if (max_df <= 0) max_df = g->node_count;
if (min_df < 0) min_df = 0;
const char* tabu = EL_CSTR(tabu_v);
/* Source selection. Prefer the label — it is a curated title when it is
* one. Fall back to content when the label is absent or a sentinel
* ("memory:remembered": a colon and no space). This single line is what
* makes Memory nodes visible to the extractor at all. */
const char* src = n->label;
if (!src || !*src) {
src = n->content;
} else if (strchr(src, ':') != NULL && strchr(src, ' ') == NULL) {
src = n->content;
}
if (!src || !*src) return el_wrap_str(el_strdup(""));
/* Collect distinct candidates from the head of the text. */
char cand[ENGRAM_ST_MAXCAND][ENGRAM_ST_TOKLEN];
int pos[ENGRAM_ST_MAXCAND];
int64_t df[ENGRAM_ST_MAXCAND];
int ncand = 0, tokidx = 0;
const char* p = src;
const char* lim = src + strnlen(src, ENGRAM_ST_SCANCHARS);
while (p < lim && ncand < ENGRAM_ST_MAXCAND) {
while (p < lim && isspace((unsigned char)*p)) p++;
if (p >= lim) break;
const char* tk = p;
while (p < lim && !isspace((unsigned char)*p)) p++;
char buf[ENGRAM_ST_TOKLEN];
int slot = tokidx++;
if (!eg_st_clean_token(tk, (size_t)(p - tk), buf, sizeof(buf))) continue;
/* Tabu exclusion, applied here so the argmax runs over eligible
* terms only. tabu arrives pipe-delimited: "|t0|t1|t2|t3|". */
if (tabu && *tabu) {
char pat[ENGRAM_ST_TOKLEN + 2];
snprintf(pat, sizeof(pat), "|%s|", buf);
if (istr_contains(tabu, pat)) continue;
}
int dup = 0;
for (int i = 0; i < ncand; i++)
if (strcasecmp(cand[i], buf) == 0) { dup = 1; break; }
if (dup) continue;
memcpy(cand[ncand], buf, strlen(buf) + 1);
pos[ncand] = slot;
df[ncand] = 0;
ncand++;
}
if (ncand == 0) return el_wrap_str(el_strdup(""));
/* One pass over the store, all candidates at once. */
for (int64_t i = 0; i < g->node_count; i++) {
const char* lbl = g->nodes[i].label;
if (!lbl || !*lbl) continue;
for (int c = 0; c < ncand; c++)
if (eg_st_label_has_word(lbl, cand[c])) df[c]++;
}
/* Argmax over idf · position · casing, subject to the df band. */
int best = -1;
double best_score = 0.0;
for (int c = 0; c < ncand; c++) {
if (df[c] > max_df) continue;
if (df[c] < min_df) continue;
double idf = log(((double)g->node_count + 1.0) / ((double)df[c] + 1.0));
if (idf <= 0.0) continue;
double position = 1.0 / log(2.718281828459045 + (double)pos[c]);
double casing = eg_st_casing(cand[c]);
double score = idf * position * casing;
if (_eg_st_debug()) {
fprintf(stderr, " cand %-24s df=%-5lld idf=%.2f pos=%d p=%.2f "
"case=%.2f score=%.3f\n",
cand[c], (long long)df[c], idf, pos[c], position,
casing, score);
}
if (score > best_score) { best_score = score; best = c; }
}
if (best < 0) return el_wrap_str(el_strdup(""));
return el_wrap_str(el_strdup(cand[best]));
}
/* engram_embed_backfill — explicitly drive the lazy embedding backfill.
* (2026-07-25 self-review.) The per-activate backfill (8 nodes/call) only
* runs inside engram_activate, and on the authoritative HTTP store nothing