Files
el/lang/runtime/vindex_bench.c
T
bigmerge 728207aabf engram: real Metal batch-cosine kernel, wired into vindex_bench's brute-force oracle
The original brief targeted engram_activate's O(N*D) cosq prescan, but #109
(this branch) already retires that loop algorithmically (HNSW seed selection
+ lazy memoized cosine) — GPU-accelerating a loop being deleted isn't real
work, so that target was dropped rather than forced.

Re-investigated for a genuine remaining GPU-shaped call site (not a
manufactured one): HNSW insert's candidate-list distance work is bounded-
degree (M=24-48) and sequential/adaptive — too fine-grained for a GPU
dispatch to pay off. No O(N^2) pairwise cosine pass exists (dedup only
checks the K=8 already-selected seed slots). No concurrent multi-query
traffic exists (server.el: the soul's curiosity loop is a single in-process
caller). vindex_bench.c's brute_topk — the correctness oracle this same PR
adds to validate HNSW recall — is the one real, unforced fit: genuine
1-query-vs-N-vectors, embarrassingly parallel, no adaptivity.

Adds:
  - eg_cosine_batch.metal: batched cosine kernel, single- and multi-query
    variants, same -2.0 dim-mismatch/zero-norm sentinel as eg_cosine().
  - eg_metal_cosine.h/.m: C-callable Objective-C bridge. Lazy one-time
    device/pipeline init, MTLResourceStorageModeShared buffers, returns
    false on ANY failure so callers fall back to the scalar CPU loop
    unconditionally — never partial, never throws.
  - eg_metal_cosine_stub.c: zero-dependency CPU-only implementation for
    non-Darwin builds (Linux CI) — same symbols, always returns false, no
    #ifdef needed at any call site.
  - build_vindex_bench.sh: one-command build, real bridge + Metal frameworks
    on Darwin, stub everywhere else.

vindex_bench.c: brute_topk_metal / brute_topk_metal_batch call the bridge,
falling back to the existing CPU brute_topk on any failure or
EL_METAL_COSINE=0. The multi-query batched path exists because the first
version (one GPU call per query) measured SLOWER than CPU at N~13.7k — it
re-uploaded the full N*D matrix every query. Fixed by uploading the matrix
once per query batch.

Measured against a real nsbx-sandboxed clone of the live store (never
:8742/:7770), 13,671 real embedded nodes, dim=768, 300 real queries:
  BRUTE-FORCE (CPU):    2.013 ms/query
  BRUTE-METAL (GPU):    0.117 ms/query   (17.2x)
  id-recall vs CPU oracle: 0.9990 over 300 queries
  same-rank |Δdist|: max 2.98e-07, mean 7.53e-08 (float32 rounding, not a bug)

Synthetic scaling sweep (13k -> 50k nodes, same dim/queries) shows the GPU
speedup holding (~11x) as N grows toward the mathematical-foundations doc's
1.3M-node target, with CPU brute-force cost growing linearly as expected.

Not wired into engram_activate or the daemon build (nsbx's _build_binary) —
vindex_bench is a standalone offline tool, not part of the request-serving
binary, so no engram_activate/server-latency claim is made here. The bridge
is a reusable primitive (single eg_cosine_batch_metal + batched
eg_cosine_batch_metal_multi) other call sites can adopt later without
re-deriving any of this.

Based on feat/reframe-region-setop (PR #109), not dev directly: the only
genuine batch-cosine call site (vindex_bench.c) exists solely on this
branch. Flagged explicitly in the PR description as a deliberate deviation
from the original "base off dev" instruction.
2026-08-15 16:48:01 -05:00

400 lines
19 KiB
C
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/* vindex_bench.c — standalone proof harness for the engram HNSW ANN index.
*
* Measures brute-force cosine top-k (the correctness ORACLE) vs vindex_search
* (HNSW) on: (a) the REAL paged store harvested read-only, and (b) synthetic
* clustered data at several sizes to trace the scaling curve. Reports build time,
* per-query latency (brute vs HNSW), and recall@k (HNSW top-k vs brute top-k).
*
* Read-only: never opens a socket, never writes the store. Safe on an nsbx clone.
*
* Also runs the brute-force oracle a second way, through
* eg_cosine_batch_metal() (Apple/Metal only — see eg_metal_cosine.h), and
* reports its latency + a correctness check against the CPU oracle
* side-by-side with the existing CPU-vs-HNSW numbers. EL_METAL_COSINE=0
* forces CPU-only.
*
* Build (macOS):
* cc -O2 -std=c11 -x objective-c -c eg_metal_cosine.m -o eg_metal_cosine.o \
* -framework Metal -framework Foundation
* cc -O2 -std=c11 vindex_bench.c engram_vindex.c eg_metal_cosine.o -lm \
* -framework Metal -framework Foundation -o vindex_bench
* Build (Linux / no Metal): omit eg_metal_cosine.o entirely and instead link
* a CPU-only stub translation unit that defines eg_cosine_batch_metal() /
* eg_cosine_batch_metal_available() returning false — this file never
* references Metal directly, only the plain-C header.
* Usage: vindex_bench store <neuron.egm> <dim> [nqueries] [k] [ef_csv]
* vindex_bench synth <N> [dim] [clusters] [nqueries] [k] [ef_csv]
*/
#include "engram_vindex.h"
#include "eg_metal_cosine.h"
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <math.h>
#include <stdint.h>
#include <time.h>
/* ── deterministic PRNG (splitmix64) so runs are reproducible ─────────────── */
static uint64_t g_seed = 0xD1B54A32D192ED03ULL;
static uint64_t sm(void){
uint64_t z = (g_seed += 0x9E3779B97F4A7C15ULL);
z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9ULL;
z = (z ^ (z >> 27)) * 0x94D049BB133111EBULL;
return z ^ (z >> 31);
}
static double urand(void){ return (double)((sm() >> 11) + 1) * (1.0/9007199254740993.0); }
static double grand(void){ /* Box-Muller */
double u1 = urand(), u2 = urand();
return sqrt(-2.0*log(u1)) * cos(2.0*M_PI*u2);
}
static double now_s(void){
struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts);
return (double)ts.tv_sec + (double)ts.tv_nsec*1e-9;
}
/* L2-normalise a row in place. */
static void l2norm(float* v, int dim){
double ss = 0; for (int i=0;i<dim;i++) ss += (double)v[i]*v[i];
if (ss > 0){ float inv = (float)(1.0/sqrt(ss)); for (int i=0;i<dim;i++) v[i]*=inv; }
}
/* Brute-force top-k by cosine distance (1 - dot on normalised vecs).
* data is n*dim, already L2-normalised. Writes k node ids (row indices) into
* out_ids ascending by distance. Returns nothing; assumes k<=n. */
static void brute_topk(const float* data, int n, int dim, const float* q,
int k, int* out_ids, float* out_d){
/* maintain a small sorted array of the k best (ascending distance). */
for (int i=0;i<k;i++){ out_ids[i]=-1; out_d[i]=2.0f+1.0f; }
for (int i=0;i<n;i++){
const float* r = data + (size_t)i*dim;
float s0=0,s1=0,s2=0,s3=0; int j=0;
for (; j+4<=dim; j+=4){ s0+=q[j]*r[j]; s1+=q[j+1]*r[j+1]; s2+=q[j+2]*r[j+2]; s3+=q[j+3]*r[j+3]; }
float dot=(s0+s1)+(s2+s3); for (; j<dim; j++) dot+=q[j]*r[j];
float d = 1.0f - dot;
if (d >= out_d[k-1]) continue;
int p = k-1;
while (p>0 && out_d[p-1] > d){ out_d[p]=out_d[p-1]; out_ids[p]=out_ids[p-1]; p--; }
out_d[p]=d; out_ids[p]=i;
}
}
/* GPU-accelerated variant of brute_topk: same oracle, same contract, same
* output — computes all n distances via eg_cosine_batch_metal() instead of
* one C loop, then does the identical top-k selection over the result.
*
* data is already L2-normalised (vindex_bench's convention throughout), so
* eg_cosine()'s general unnormalised cosine and this file's "distance =
* 1 - dot" both reduce to the same number here (a unit vector's norm is 1,
* so cosine == dot). Passing pre-normalised rows through the general-purpose
* batch kernel is deliberate: it proves the SAME primitive that would serve
* el_runtime.c's raw/unnormalised embeddings also serves this oracle without
* a second code path.
*
* Returns false (out_ids/out_d untouched) if the GPU path is unavailable or
* fails for any reason — caller must fall back to brute_topk(). Never
* partial: either the full top-k was computed on GPU, or nothing was. */
/* EL_METAL_COSINE: 0/off/false disables the GPU path outright (falls back to
* brute_topk() every time), matching el_runtime.c's own gate for the same
* env var. Unset or any other value = auto (try Metal, fall back on failure). */
static bool g_metal_env_checked = false;
static bool g_metal_disabled_by_env = false;
static void eg_metal_check_env_once(void){
if (g_metal_env_checked) return;
g_metal_env_checked = true;
const char* v = getenv("EL_METAL_COSINE");
if (v && (v[0]=='0' || v[0]=='n' || v[0]=='N' || v[0]=='f' || v[0]=='F'))
g_metal_disabled_by_env = true;
}
static bool brute_topk_metal(const float* data, int n, int dim, const float* q,
int k, int* out_ids, float* out_d){
eg_metal_check_env_once();
if (g_metal_disabled_by_env) return false;
const float** row_ptrs = malloc((size_t)n * sizeof(float*));
int32_t* dims = malloc((size_t)n * sizeof(int32_t));
double* scores = malloc((size_t)n * sizeof(double));
if (!row_ptrs || !dims || !scores) { free(row_ptrs); free(dims); free(scores); return false; }
for (int i = 0; i < n; i++) {
row_ptrs[i] = data + (size_t)i * dim;
dims[i] = dim;
}
bool ok = eg_cosine_batch_metal(q, dim, row_ptrs, dims, n, scores);
free(row_ptrs); free(dims);
if (!ok) { free(scores); return false; }
for (int i = 0; i < k; i++) { out_ids[i] = -1; out_d[i] = 3.0f; }
for (int i = 0; i < n; i++) {
float d = 1.0f - (float)scores[i]; /* same distance convention as brute_topk */
if (d >= out_d[k-1]) continue;
int p = k - 1;
while (p > 0 && out_d[p-1] > d) { out_d[p] = out_d[p-1]; out_ids[p] = out_ids[p-1]; p--; }
out_d[p] = d; out_ids[p] = i;
}
free(scores);
return true;
}
/* Batched sibling of brute_topk_metal: computes top-k for ALL nq queries in
* ONE eg_cosine_batch_metal_multi() call, uploading node_matrix exactly
* once instead of once per query. out_ids/out_d are nq*k, row-major
* (query i's results at out_ids+i*k / out_d+i*k) — same layout run_bench
* already uses for `gt`/per-query scratch. Returns false (nothing written)
* on any failure; caller falls back to the per-query CPU brute_topk loop. */
static bool brute_topk_metal_batch(const float* data, int n, int dim,
const float* queries, int nq,
int k, int* out_ids, float* out_d){
eg_metal_check_env_once();
if (g_metal_disabled_by_env) return false;
const float** row_ptrs = malloc((size_t)n * sizeof(float*));
int32_t* dims = malloc((size_t)n * sizeof(int32_t));
double* scores = malloc((size_t)nq * (size_t)n * sizeof(double));
if (!row_ptrs || !dims || !scores) { free(row_ptrs); free(dims); free(scores); return false; }
for (int i = 0; i < n; i++) { row_ptrs[i] = data + (size_t)i * dim; dims[i] = dim; }
bool ok = eg_cosine_batch_metal_multi(queries, dim, nq, row_ptrs, dims, n, scores);
free(row_ptrs); free(dims);
if (!ok) { free(scores); return false; }
for (int qi = 0; qi < nq; qi++) {
int* ids = out_ids + (size_t)qi * k;
float* ds = out_d + (size_t)qi * k;
const double* srow = scores + (size_t)qi * n;
for (int i = 0; i < k; i++) { ids[i] = -1; ds[i] = 3.0f; }
for (int i = 0; i < n; i++) {
float d = 1.0f - (float)srow[i];
if (d >= ds[k-1]) continue;
int p = k - 1;
while (p > 0 && ds[p-1] > d) { ds[p] = ds[p-1]; ids[p] = ids[p-1]; p--; }
ds[p] = d; ids[p] = i;
}
}
free(scores);
return true;
}
/* recall@k: |brute_topk ∩ hnsw_topk| / k. Both are id arrays of length k. */
static double recall_at_k(const int* gt, const uint64_t* ann, int nann, int k){
int hit = 0;
for (int i=0;i<k;i++){
if (gt[i] < 0) continue;
for (int j=0;j<nann;j++){ if ((int)ann[j] == gt[i]){ hit++; break; } }
}
return (double)hit / (double)k;
}
/* Parse "64,128,256" into an int array; returns count. */
static int parse_csv(const char* s, int* out, int maxo){
int n=0; if(!s||!*s) return 0;
const char* p=s;
while(*p && n<maxo){ out[n++]=atoi(p); while(*p && *p!=',') p++; if(*p==',') p++; }
return n;
}
/* Generate n unit vectors on a LOW-DIMENSIONAL MANIFOLD, the property that makes
* real text embeddings tractable for ANN: each vector is a fixed random linear map
* A (dim × LATENT) applied to a latent gaussian z ∈ R^LATENT, plus small ambient
* noise, then L2-normalised. Points therefore lie near a `latent`-dim subspace, so
* every point has a well-defined tight neighbourhood (high recall) and the HNSW
* graph is cheap to build — unlike near-isotropic 768-d gaussians, where the curse
* of dimensionality makes all points near-equidistant (no structure → slow build,
* low recall) and unlike tight clusters (near-duplicates → artificial top-k ties).
* `sigma` is the ambient-noise scale. This reproduces the intrinsic-dimensionality
* regime of nomic embeddings, so the scaling curve reflects real-corpus behaviour. */
#define SYNTH_LATENT 48
static void gen_synth(float* data, int n, int dim, int clusters, double sigma){
(void)clusters;
float* A = malloc((size_t)dim*SYNTH_LATENT*sizeof(float)); /* fixed random basis */
for (size_t i=0;i<(size_t)dim*SYNTH_LATENT;i++) A[i]=(float)grand();
float z[SYNTH_LATENT];
for (int i=0;i<n;i++){
for (int l=0;l<SYNTH_LATENT;l++) z[l]=(float)grand();
float* v = data+(size_t)i*dim;
for (int j=0;j<dim;j++){
float acc = (float)(sigma*grand());
const float* row = A + (size_t)j*SYNTH_LATENT;
for (int l=0;l<SYNTH_LATENT;l++) acc += row[l]*z[l];
v[j]=acc;
}
l2norm(v, dim);
}
free(A);
}
/* Build M / ef_construction come from env (VIDX_M / VIDX_EFC) so the scaling
* sweep can trade build cost against graph quality without a recompile. 0 = default. */
static int env_int(const char* k, int dflt){ const char* s=getenv(k); return (s&&*s)?atoi(s):dflt; }
/* Run the full brute-vs-HNSW comparison over an already-normalised dataset. */
static void run_bench(const char* label, float* data, int n, int dim,
int nq, int k, int* efs, int nef, double build_s){
(void)build_s;
int bM = env_int("VIDX_M", 0), bEFC = env_int("VIDX_EFC", 0);
printf("\n=== %s : N=%d dim=%d k=%d queries=%d ===\n", label, n, dim, k, nq);
/* build the index once (shared across ef settings). */
double t0 = now_s();
VIndex* ix = vindex_create(dim, bM, bEFC);
for (int i=0;i<n;i++) vindex_insert(ix, (uint64_t)i, data + (size_t)i*dim);
double bt = now_s()-t0;
printf("HNSW build: M=%d ef_construction=%d -> %.3f s (%.1f k nodes/s)\n",
bM?bM:VINDEX_DEFAULT_M, bEFC?bEFC:VINDEX_DEFAULT_EF_CONSTRUCTION, bt, n/1000.0/bt);
/* choose query vectors: perturb random dataset rows (near-but-not-identical). */
int* qidx = malloc((size_t)nq*sizeof(int));
float* qv = malloc((size_t)nq*dim*sizeof(float));
for (int i=0;i<nq;i++){
int r = (int)(sm() % (uint64_t)n);
qidx[i]=r;
float* dst = qv+(size_t)i*dim; const float* src = data+(size_t)r*dim;
for (int j=0;j<dim;j++) dst[j] = src[j] + (float)(0.01*grand());
l2norm(dst, dim);
}
/* ground truth: brute-force top-k for every query (also the oracle latency).
* gd is nq*k (one real slot per query, not a shared scratch buffer) so the
* GPU comparison below can diff against every query's actual distances,
* not just whichever query happened to run last. */
int* gt = malloc((size_t)nq*k*sizeof(int));
float* gd = malloc((size_t)nq*k*sizeof(float));
double tb0 = now_s();
for (int i=0;i<nq;i++) brute_topk(data, n, dim, qv+(size_t)i*dim, k, gt+(size_t)i*k, gd+(size_t)i*k);
double brute_ms = (now_s()-tb0)*1000.0/nq;
printf("BRUTE-FORCE : %8.3f ms/query (oracle; O(N*D), CPU)\n", brute_ms);
/* GPU-accelerated oracle: SAME nq queries, SAME top-k contract, via ONE
* eg_cosine_batch_metal_multi() call (uploads node_matrix once, not once
* per query — see brute_topk_metal_batch). Run only if the GPU path is
* actually available (checked once) — never fabricated, never assumed.
* Verified against the CPU ground truth computed above: id-recall across
* ALL nq queries, plus the actual max distance delta across every
* (query, rank) pair that was compared — not a single spot check. */
eg_metal_check_env_once();
if (!g_metal_disabled_by_env && eg_cosine_batch_metal_available()) {
int* gtm = malloc((size_t)nq*k*sizeof(int));
float* gdm = malloc((size_t)nq*k*sizeof(float));
double tm0 = now_s();
bool ok = brute_topk_metal_batch(data, n, dim, qv, nq, k, gtm, gdm);
double metal_ms = (now_s()-tm0)*1000.0/nq;
if (ok) {
double rec_sum = 0; double max_ddiff = 0; double sum_ddiff = 0; int compared = 0;
for (int i=0;i<nq;i++) {
const int* ids_gt = gt+(size_t)i*k;
const float* d_gt = gd+(size_t)i*k;
const int* ids_m = gtm+(size_t)i*k;
const float* d_m = gdm+(size_t)i*k;
uint64_t idset[512]; int m = (k<512)?k:512;
for (int j=0;j<m;j++) idset[j] = (uint64_t)ids_m[j];
rec_sum += recall_at_k(ids_gt, idset, m, k);
/* same-rank distance delta — valid whenever both sides agree on
* the id at that rank (true almost always, given ~100% recall;
* a rank where they disagree isn't a meaningful delta to diff). */
for (int j=0;j<k;j++) {
if (ids_gt[j] == ids_m[j]) {
double diff = fabs((double)d_gt[j]-(double)d_m[j]);
if (diff>max_ddiff) max_ddiff=diff;
sum_ddiff += diff; compared++;
}
}
}
printf("BRUTE-METAL : %8.3f ms/query (%.1fx vs CPU brute; id-recall %.4f vs CPU oracle over %d queries; same-rank |Δdist|: max %.2e, mean %.2e over %d compared)\n",
metal_ms, brute_ms/metal_ms, rec_sum/nq, nq, max_ddiff, compared?sum_ddiff/compared:0.0, compared);
} else {
printf("BRUTE-METAL : GPU batch call failed/unavailable mid-run — skipped\n");
}
free(gtm); free(gdm);
} else {
printf("BRUTE-METAL : no Metal device/pipeline available — CPU-only\n");
}
/* HNSW at each ef. */
uint64_t* aid = malloc((size_t)k*sizeof(uint64_t));
float* ad = malloc((size_t)k*sizeof(float));
printf("%-6s %14s %12s %10s\n", "ef", "HNSW ms/query", "speedup", "recall@k");
for (int e=0;e<nef;e++){
int ef = efs[e];
double th0 = now_s();
double rec_sum = 0;
for (int i=0;i<nq;i++){
int m = vindex_search(ix, qv+(size_t)i*dim, k, ef, aid, ad);
rec_sum += recall_at_k(gt+(size_t)i*k, aid, m, k);
}
double hnsw_ms = (now_s()-th0)*1000.0/nq;
printf("%-6d %14.4f %11.1fx %10.4f\n", ef, hnsw_ms, brute_ms/hnsw_ms, rec_sum/nq);
}
free(qidx); free(qv); free(gt); free(gd); free(aid); free(ad);
vindex_free(ix);
}
int main(int argc, char** argv){
setvbuf(stdout, NULL, _IOLBF, 0); /* line-buffered so progress streams to a log */
if (argc < 2){ fprintf(stderr,"usage: %s store <path> <dim> [nq] [k] [ef_csv] | synth <N> [dim] [clusters] [nq] [k] [ef_csv] | sweep <dim> <N_csv> [nq] [k] [ef_csv]\n", argv[0]); return 2; }
int defef[8]; int ndef;
if (strcmp(argv[1],"sweep")==0){
if (argc < 4){ fprintf(stderr,"sweep needs <dim> <N_csv>\n"); return 2; }
int dim = atoi(argv[2]);
int Ns[16]; int nN = parse_csv(argv[3], Ns, 16);
int nq = (argc>4)?atoi(argv[4]):200;
int k = (argc>5)?atoi(argv[5]):10;
ndef = (argc>6)?parse_csv(argv[6],defef,8):parse_csv("64,128,200",defef,8);
for (int s=0;s<nN;s++){
int N = Ns[s];
float* data = malloc((size_t)N*dim*sizeof(float));
if (!data){ fprintf(stderr,"OOM at N=%d\n",N); continue; }
int clusters = N/100; if (clusters < 64) clusters = 64;
gen_synth(data, N, dim, clusters, 1.0);
char lbl[64]; snprintf(lbl,sizeof lbl,"SYNTH N=%d", N);
run_bench(lbl, data, N, dim, nq, k, defef, ndef, 0.0);
free(data);
}
return 0;
}
if (strcmp(argv[1],"store")==0){
if (argc < 4){ fprintf(stderr,"store needs <path> <dim>\n"); return 2; }
const char* path = argv[2]; int dim = atoi(argv[3]);
int nq = (argc>4)?atoi(argv[4]):500;
int k = (argc>5)?atoi(argv[5]):10;
ndef = (argc>6)?parse_csv(argv[6],defef,8):parse_csv("32,64,128,200,400",defef,8);
printf("Harvesting emb vectors from %s (dim=%d) ...\n", path, dim);
float* data=NULL; int n=0;
double t0=now_s();
int h = vindex_harvest_from_store(path, dim, &data, NULL, &n);
double harvest_s = now_s()-t0;
if (h < 0 || n == 0){ fprintf(stderr,"harvest failed (h=%d n=%d) — wrong dim or path?\n", h, n); return 1; }
printf("Harvested %d live embedded nodes in %.2f s\n", n, harvest_s);
for (int i=0;i<n;i++) l2norm(data+(size_t)i*dim, dim); /* oracle needs normalised */
if (nq > n) nq = n;
run_bench("REAL STORE", data, n, dim, nq, k, defef, ndef, 0.0);
free(data);
return 0;
}
if (strcmp(argv[1],"synth")==0){
if (argc < 3){ fprintf(stderr,"synth needs <N>\n"); return 2; }
int N = atoi(argv[2]);
int dim = (argc>3)?atoi(argv[3]):768;
int clusters = (argc>4)?atoi(argv[4]):200;
int nq = (argc>5)?atoi(argv[5]):500;
int k = (argc>6)?atoi(argv[6]):10;
ndef = (argc>7)?parse_csv(argv[7],defef,8):parse_csv("64,128,200",defef,8);
printf("Generating %d synthetic clustered vectors (dim=%d clusters=%d) ...\n", N, dim, clusters);
float* data = malloc((size_t)N*dim*sizeof(float));
if (!data){ fprintf(stderr,"OOM allocating %zu bytes\n", (size_t)N*dim*sizeof(float)); return 1; }
gen_synth(data, N, dim, clusters, 0.35);
char lbl[64]; snprintf(lbl,sizeof lbl,"SYNTH");
run_bench(lbl, data, N, dim, nq, k, defef, ndef, 0.0);
free(data);
return 0;
}
fprintf(stderr,"unknown mode '%s'\n", argv[1]);
return 2;
}