Merge pull request 'engram: reconcile #105's embed-cache/beam-BFS latency fixes onto current dev' (#115) from fix/engram-search-latency-reconciled into dev
El SDK CI - dev / build-and-test (push) Failing after 4m6s
El SDK CI - dev / build-and-test (push) Failing after 4m6s
This commit was merged in pull request #115.
This commit is contained in:
+114
-20
@@ -6847,10 +6847,16 @@ static float* eg_embed_fetch(const char* text, int32_t* out_dim) {
|
|||||||
else esc[w++] = (char)c;
|
else esc[w++] = (char)c;
|
||||||
}
|
}
|
||||||
esc[w] = '\0';
|
esc[w] = '\0';
|
||||||
size_t blen = w + strlen(eg_embed_model()) + 64;
|
size_t blen = w + strlen(eg_embed_model()) + 96;
|
||||||
char* body = malloc(blen);
|
char* body = malloc(blen);
|
||||||
if (!body) { free(esc); return NULL; }
|
if (!body) { free(esc); return NULL; }
|
||||||
snprintf(body, blen, "{\"model\":\"%s\",\"prompt\":\"%s\"}",
|
/* keep_alive:-1 pins the embed model resident in Ollama indefinitely
|
||||||
|
* (2026-08-15, PR #105 port). Without it the tiny embed model is evicted
|
||||||
|
* whenever a larger generation model loads (unified-memory pressure), so
|
||||||
|
* the NEXT search pays a cold model reload — measured cold reload up to
|
||||||
|
* ~2.2s vs ~0.02-0.05s warm, well inside ENGRAM_EMBED_TIMEOUT_MS but a
|
||||||
|
* real tax on every activate() call that lands cold. Pinning removes it. */
|
||||||
|
snprintf(body, blen, "{\"model\":\"%s\",\"keep_alive\":-1,\"prompt\":\"%s\"}",
|
||||||
eg_embed_model(), esc);
|
eg_embed_model(), esc);
|
||||||
free(esc);
|
free(esc);
|
||||||
struct curl_slist* h = curl_slist_append(NULL, "Content-Type: application/json");
|
struct curl_slist* h = curl_slist_append(NULL, "Content-Type: application/json");
|
||||||
@@ -9508,6 +9514,36 @@ static inline double eg_cosq_at(EngramStore* g, double* cosq, unsigned char* cos
|
|||||||
return cosq[i];
|
return cosq[i];
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* ── Beam cap for engram_activate spreading activation (2026-08-15, PR #105
|
||||||
|
* port) ──────────────────────────────────────────────────────────────────
|
||||||
|
* #105 measured the OLD (pre-adjacency-index, pre-qgate, pre-fan-effect)
|
||||||
|
* BFS reaching multi-second/crash territory at depth 2-3 from unbounded
|
||||||
|
* hub-node fan-out. That specific failure mode is already substantially
|
||||||
|
* mitigated here by mechanisms #105's branch predates: the adjacency index
|
||||||
|
* (O(degree) not O(E) per hop), the query-aware qgate (prunes semantically
|
||||||
|
* irrelevant branches), the ACT-R fan-effect correction (dampens popular-
|
||||||
|
* hub over-connectivity), and the 0.02 firing threshold. A beam cap is still
|
||||||
|
* a genuine additional, orthogonal bound: it caps WORST-CASE per-hop
|
||||||
|
* expansion width regardless of how many targets happen to pass the soft
|
||||||
|
* gates above, so it is kept as defense in depth rather than dropped as
|
||||||
|
* redundant.
|
||||||
|
*
|
||||||
|
* Bounds the number of frontier nodes EXPANDED per hop-level (see the
|
||||||
|
* level-batching in the BFS below). Every reached node still gets its
|
||||||
|
* best_bg[]/reached[] recorded and appears in the returned/promoted set —
|
||||||
|
* the cap bounds only how far ASSOCIATIVE SPREAD continues past a level,
|
||||||
|
* never the direct seed matches or the reported result set. Tunable via
|
||||||
|
* ENGRAM_ACTIVATE_BEAM (default 128, matching #105); set very high (e.g.
|
||||||
|
* the node count) to recover the pre-cap unbounded-per-level behaviour. */
|
||||||
|
static int64_t engram_activate_beam(void) {
|
||||||
|
static int64_t v = -1;
|
||||||
|
if (v >= 0) return v;
|
||||||
|
const char* s = getenv("ENGRAM_ACTIVATE_BEAM");
|
||||||
|
int64_t d = 128;
|
||||||
|
if (s && *s) { char* e = NULL; long t = strtol(s, &e, 10); if (e != s && t > 0) d = (int64_t)t; }
|
||||||
|
v = d; return v;
|
||||||
|
}
|
||||||
|
|
||||||
el_val_t engram_activate(el_val_t query, el_val_t depth) {
|
el_val_t engram_activate(el_val_t query, el_val_t depth) {
|
||||||
EngramStore* g = engram_get();
|
EngramStore* g = engram_get();
|
||||||
const char* q = EL_CSTR(query);
|
const char* q = EL_CSTR(query);
|
||||||
@@ -9552,25 +9588,43 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
|
|||||||
backfilled++;
|
backfilled++;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
/* Query embedding, cached single-slot: the curiosity loop re-issues the
|
/* Query embedding cache (2026-08-15, PR #105 port: single-slot -> direct-
|
||||||
* same 4 rotating phrases, so consecutive identical queries skip the
|
* mapped multi-slot). The single-slot cache below this comment's history
|
||||||
* HTTP round-trip entirely. */
|
* only remembered the LAST query, so "the curiosity loop re-issues the
|
||||||
static char* _eg_qcache_text = NULL;
|
* same 4 rotating phrases" only hit when two CONSECUTIVE calls used the
|
||||||
static float* _eg_qcache_emb = NULL;
|
* SAME phrase — any rotation among >1 phrase evicted the slot before it
|
||||||
static int32_t _eg_qcache_dim = 0;
|
* could be reused. #105 measured this as a real cost (a repeated query
|
||||||
|
* costing a full Ollama round-trip whenever a different phrase intervened)
|
||||||
|
* and fixed it with a direct-mapped, FNV-1a-keyed cache sized for the
|
||||||
|
* rotation. Ported here on TOP of the existing cosq/e_eff semantic layer
|
||||||
|
* (this cache only ever supplies q_emb/q_dim into that unchanged
|
||||||
|
* pipeline) rather than replacing it — see the M8/#105 reconciliation
|
||||||
|
* note above eg_cosq_at. ENGRAM_QCACHE_SIZE must be a power of two (mask
|
||||||
|
* indexing below). Full strcmp on lookup rejects hash collisions; each
|
||||||
|
* slot owns its `text`/`vec` and is freed on eviction, matching the old
|
||||||
|
* single-slot free/replace contract — q_emb below still points at cache-
|
||||||
|
* owned memory the caller must NOT free, just as before. */
|
||||||
|
#define ENGRAM_QCACHE_SIZE 1024
|
||||||
|
typedef struct { char* text; uint64_t hash; float* vec; int32_t dim; } EgQCacheEntry;
|
||||||
|
static EgQCacheEntry _eg_qcache[ENGRAM_QCACHE_SIZE];
|
||||||
float* q_emb = NULL;
|
float* q_emb = NULL;
|
||||||
int32_t q_dim = 0;
|
int32_t q_dim = 0;
|
||||||
if (_eg_qcache_text && strcmp(_eg_qcache_text, q) == 0) {
|
{
|
||||||
q_emb = _eg_qcache_emb; q_dim = _eg_qcache_dim;
|
uint64_t qh = engram_id_hash(q);
|
||||||
} else {
|
EgQCacheEntry* slot = &_eg_qcache[qh & (ENGRAM_QCACHE_SIZE - 1)];
|
||||||
int32_t d = 0;
|
if (slot->vec && slot->hash == qh && slot->text && strcmp(slot->text, q) == 0) {
|
||||||
float* v = eg_embed_fetch(q, &d);
|
q_emb = slot->vec; q_dim = slot->dim;
|
||||||
if (v) {
|
} else {
|
||||||
free(_eg_qcache_text); free(_eg_qcache_emb);
|
int32_t d = 0;
|
||||||
_eg_qcache_text = strdup(q);
|
float* v = eg_embed_fetch(q, &d);
|
||||||
_eg_qcache_emb = v;
|
if (v) {
|
||||||
_eg_qcache_dim = d;
|
free(slot->text); free(slot->vec); /* evict prior occupant */
|
||||||
q_emb = v; q_dim = d;
|
slot->text = strdup(q);
|
||||||
|
slot->hash = qh;
|
||||||
|
slot->vec = v;
|
||||||
|
slot->dim = d;
|
||||||
|
q_emb = v; q_dim = d;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
/* ── Context centroid fold-in (2026-07-29) ──────────────────────────
|
/* ── Context centroid fold-in (2026-07-29) ──────────────────────────
|
||||||
@@ -9997,8 +10051,45 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
|
|||||||
const double FAN_DREF = (g->adj_connected > 0)
|
const double FAN_DREF = (g->adj_connected > 0)
|
||||||
? (2.0 * (double)g->edge_count / (double)g->adj_connected) : 0.0;
|
? (2.0 * (double)g->edge_count / (double)g->adj_connected) : 0.0;
|
||||||
_eg_act_fan_dref = FAN_DREF;
|
_eg_act_fan_dref = FAN_DREF;
|
||||||
|
const int64_t activate_beam = engram_activate_beam();
|
||||||
while (fhead < ftail) {
|
while (fhead < ftail) {
|
||||||
Frontier f = fr[fhead++];
|
/* Level-batch (2026-08-15, PR #105 port): entries sharing .hops are
|
||||||
|
* always contiguous — hop k+1 entries are appended only while
|
||||||
|
* processing hop k, strictly after the current ftail, so they form one
|
||||||
|
* block right after hop k's block (see engram_activate_beam's comment
|
||||||
|
* for why this holds even with the improve-and-re-enqueue behavior
|
||||||
|
* below). Find this level's extent, then beam-select which of it
|
||||||
|
* EXPANDS; every entry in the level still gets recorded via
|
||||||
|
* reached[]/best_bg[] regardless (that happened when it was enqueued,
|
||||||
|
* one level up) — the cap bounds propagation width only. */
|
||||||
|
int64_t level_hops = fr[fhead].hops;
|
||||||
|
int64_t level_start = fhead;
|
||||||
|
int64_t level_end = fhead;
|
||||||
|
while (level_end < ftail && fr[level_end].hops == level_hops) level_end++;
|
||||||
|
int64_t level_n = level_end - level_start;
|
||||||
|
unsigned char* expand = NULL;
|
||||||
|
if (level_n > activate_beam) {
|
||||||
|
expand = calloc((size_t)level_n, 1);
|
||||||
|
if (expand) {
|
||||||
|
/* Partial selection: mark the top-`activate_beam` entries by
|
||||||
|
* .act. O(beam*level_n) — beam is the small tunable. */
|
||||||
|
for (int64_t bsel = 0; bsel < activate_beam; bsel++) {
|
||||||
|
int64_t best = -1;
|
||||||
|
for (int64_t k = 0; k < level_n; k++) {
|
||||||
|
if (expand[k]) continue;
|
||||||
|
if (best < 0 || fr[level_start+k].act > fr[level_start+best].act)
|
||||||
|
best = k;
|
||||||
|
}
|
||||||
|
if (best < 0) break;
|
||||||
|
expand[best] = 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/* OOM on the selection map: expand stays NULL -> this level runs
|
||||||
|
* unbounded, same as if beam were disabled. Never silently wrong. */
|
||||||
|
}
|
||||||
|
for (int64_t lk = level_start; lk < level_end; lk++) {
|
||||||
|
if (expand && !expand[lk - level_start]) continue;
|
||||||
|
Frontier f = fr[lk];
|
||||||
if (f.hops >= max_depth) continue;
|
if (f.hops >= max_depth) continue;
|
||||||
int64_t cur = f.idx;
|
int64_t cur = f.idx;
|
||||||
int64_t new_hops = f.hops + 1;
|
int64_t new_hops = f.hops + 1;
|
||||||
@@ -10130,6 +10221,9 @@ el_val_t engram_activate(el_val_t query, el_val_t depth) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
free(expand);
|
||||||
|
fhead = level_end;
|
||||||
}
|
}
|
||||||
/* Persist layer-1 background_activation to node store. */
|
/* Persist layer-1 background_activation to node store. */
|
||||||
for (int64_t i = 0; i < g->node_count; i++) {
|
for (int64_t i = 0; i < g->node_count; i++) {
|
||||||
|
|||||||
Reference in New Issue
Block a user