/* vindex_bench.c — standalone proof harness for the engram HNSW ANN index. * * Measures brute-force cosine top-k (the correctness ORACLE) vs vindex_search * (HNSW) on: (a) the REAL paged store harvested read-only, and (b) synthetic * clustered data at several sizes to trace the scaling curve. Reports build time, * per-query latency (brute vs HNSW), and recall@k (HNSW top-k vs brute top-k). * * Read-only: never opens a socket, never writes the store. Safe on an nsbx clone. * * Also runs the brute-force oracle a second (and third) way, through the * batch-cosine Strategies behind eg_cosine_batch_strategy.h — the ggml * strategy and the hand-rolled-Metal strategy (Apple/Metal only; see * eg_cosine_batch.h/eg_cosine_batch_strategy.h) — and reports each one's * latency + a correctness check against the CPU oracle side-by-side with the * existing CPU-vs-HNSW numbers. This harness deliberately reaches past the * single-selection Factory (eg_cosine_batch.c) to instantiate every * compiled-in strategy directly, so it can compare all of them against the * SAME dataset in one run — that is the harness's whole job; a real call * site (el_runtime.c) never does this, it only ever calls the plain * eg_cosine_batch()/eg_cosine_batch_multi() adapter functions. * EL_METAL_COSINE=0 forces CPU-only (skips every strategy comparison). * * Build (macOS, ggml + hand-rolled Metal): see build_vindex_bench.sh. * Build (Linux / no Metal): omit every eg_cosine_batch_strategy_*.{c,m} file * except eg_cosine_batch_strategy_cpu.c — this file never references * ggml/Metal directly except through the plain-C strategy header, guarded * by the same EG_HAVE_STRATEGY_* build macros the Factory itself uses. * Usage: vindex_bench store [nqueries] [k] [ef_csv] * vindex_bench synth [dim] [clusters] [nqueries] [k] [ef_csv] */ #include "engram_vindex.h" #include "eg_cosine_batch_strategy.h" #include #include #include #include #include #include /* ── deterministic PRNG (splitmix64) so runs are reproducible ─────────────── */ static uint64_t g_seed = 0xD1B54A32D192ED03ULL; static uint64_t sm(void){ uint64_t z = (g_seed += 0x9E3779B97F4A7C15ULL); z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9ULL; z = (z ^ (z >> 27)) * 0x94D049BB133111EBULL; return z ^ (z >> 31); } static double urand(void){ return (double)((sm() >> 11) + 1) * (1.0/9007199254740993.0); } static double grand(void){ /* Box-Muller */ double u1 = urand(), u2 = urand(); return sqrt(-2.0*log(u1)) * cos(2.0*M_PI*u2); } static double now_s(void){ struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); return (double)ts.tv_sec + (double)ts.tv_nsec*1e-9; } /* L2-normalise a row in place. */ static void l2norm(float* v, int dim){ double ss = 0; for (int i=0;i 0){ float inv = (float)(1.0/sqrt(ss)); for (int i=0;i= out_d[k-1]) continue; int p = k-1; while (p>0 && out_d[p-1] > d){ out_d[p]=out_d[p-1]; out_ids[p]=out_ids[p-1]; p--; } out_d[p]=d; out_ids[p]=i; } } /* recall@k: |brute_topk ∩ hnsw_topk| / k. Both are id arrays of length k. */ static double recall_at_k(const int* gt, const uint64_t* ann, int nann, int k){ int hit = 0; for (int i=0;ibatch_multi() call, * uploading/preparing the node population exactly once instead of once per * query. out_ids/out_d are nq*k, row-major (query i's results at * out_ids+i*k / out_d+i*k). Returns false (nothing written) on any * failure/unavailability; caller treats that as "skip this strategy in the * report", never as a hard error. */ static bool batch_topk_strategy(const EgCosineBatchStrategy* strat, const float* data, int n, int dim, const float* queries, int nq, int k, int* out_ids, float* out_d){ if (!strat || !strat->available()) return false; const float** row_ptrs = malloc((size_t)n * sizeof(float*)); int32_t* dims = malloc((size_t)n * sizeof(int32_t)); double* scores = malloc((size_t)nq * (size_t)n * sizeof(double)); if (!row_ptrs || !dims || !scores) { free(row_ptrs); free(dims); free(scores); return false; } for (int i = 0; i < n; i++) { row_ptrs[i] = data + (size_t)i * dim; dims[i] = dim; } bool ok = strat->batch_multi(queries, dim, nq, row_ptrs, dims, n, scores); free(row_ptrs); free(dims); if (!ok) { free(scores); return false; } for (int qi = 0; qi < nq; qi++) { int* ids = out_ids + (size_t)qi * k; float* ds = out_d + (size_t)qi * k; const double* srow = scores + (size_t)qi * n; for (int i = 0; i < k; i++) { ids[i] = -1; ds[i] = 3.0f; } for (int i = 0; i < n; i++) { float d = 1.0f - (float)srow[i]; /* same distance convention as brute_topk */ if (d >= ds[k-1]) continue; int p = k - 1; while (p > 0 && ds[p-1] > d) { ds[p] = ds[p-1]; ids[p] = ids[p-1]; p--; } ds[p] = d; ids[p] = i; } } free(scores); return true; } /* Runs batch_topk_strategy for one named strategy over ALL nq queries, diffs * against the CPU ground truth (gt/gd, both nq*k), and prints a report line * in the same shape PR #114 established for BRUTE-METAL — id-recall over * every query plus the actual max/mean same-rank distance delta across * every (query,rank) pair that was compared, never fabricated or assumed. */ static void report_strategy_vs_oracle(const char* label, const EgCosineBatchStrategy* strat, const float* data, int n, int dim, const float* qv, int nq, int k, const int* gt, const float* gd, double brute_ms){ if (g_strategy_disabled_by_env) { printf("%-13s: disabled via EL_METAL_COSINE\n", label); return; } if (!strat || !strat->available()) { printf("%-13s: not available on this build/host — skipped\n", label); return; } int* gtm = malloc((size_t)nq*k*sizeof(int)); float* gdm = malloc((size_t)nq*k*sizeof(float)); double tm0 = now_s(); bool ok = batch_topk_strategy(strat, data, n, dim, qv, nq, k, gtm, gdm); double strat_ms = (now_s()-tm0)*1000.0/nq; if (ok) { double rec_sum = 0; double max_ddiff = 0; double sum_ddiff = 0; int compared = 0; for (int i=0;imax_ddiff) max_ddiff=diff; sum_ddiff += diff; compared++; } } } printf("%-13s: %8.3f ms/query (%.1fx vs CPU brute; id-recall %.4f vs CPU oracle over %d queries; same-rank |Δdist|: max %.2e, mean %.2e over %d compared)\n", label, strat_ms, brute_ms/strat_ms, rec_sum/nq, nq, max_ddiff, compared?sum_ddiff/compared:0.0, compared); } else { printf("%-13s: batch call failed mid-run — skipped\n", label); } free(gtm); free(gdm); } /* Parse "64,128,256" into an int array; returns count. */ static int parse_csv(const char* s, int* out, int maxo){ int n=0; if(!s||!*s) return 0; const char* p=s; while(*p && n %.3f s (%.1f k nodes/s)\n", bM?bM:VINDEX_DEFAULT_M, bEFC?bEFC:VINDEX_DEFAULT_EF_CONSTRUCTION, bt, n/1000.0/bt); /* choose query vectors: perturb random dataset rows (near-but-not-identical). */ int* qidx = malloc((size_t)nq*sizeof(int)); float* qv = malloc((size_t)nq*dim*sizeof(float)); for (int i=0;i [nq] [k] [ef_csv] | synth [dim] [clusters] [nq] [k] [ef_csv] | sweep [nq] [k] [ef_csv]\n", argv[0]); return 2; } int defef[8]; int ndef; if (strcmp(argv[1],"sweep")==0){ if (argc < 4){ fprintf(stderr,"sweep needs \n"); return 2; } int dim = atoi(argv[2]); int Ns[16]; int nN = parse_csv(argv[3], Ns, 16); int nq = (argc>4)?atoi(argv[4]):200; int k = (argc>5)?atoi(argv[5]):10; ndef = (argc>6)?parse_csv(argv[6],defef,8):parse_csv("64,128,200",defef,8); for (int s=0;s \n"); return 2; } const char* path = argv[2]; int dim = atoi(argv[3]); int nq = (argc>4)?atoi(argv[4]):500; int k = (argc>5)?atoi(argv[5]):10; ndef = (argc>6)?parse_csv(argv[6],defef,8):parse_csv("32,64,128,200,400",defef,8); printf("Harvesting emb vectors from %s (dim=%d) ...\n", path, dim); float* data=NULL; int n=0; double t0=now_s(); int h = vindex_harvest_from_store(path, dim, &data, NULL, &n); double harvest_s = now_s()-t0; if (h < 0 || n == 0){ fprintf(stderr,"harvest failed (h=%d n=%d) — wrong dim or path?\n", h, n); return 1; } printf("Harvested %d live embedded nodes in %.2f s\n", n, harvest_s); for (int i=0;i n) nq = n; run_bench("REAL STORE", data, n, dim, nq, k, defef, ndef, 0.0); free(data); return 0; } if (strcmp(argv[1],"synth")==0){ if (argc < 3){ fprintf(stderr,"synth needs \n"); return 2; } int N = atoi(argv[2]); int dim = (argc>3)?atoi(argv[3]):768; int clusters = (argc>4)?atoi(argv[4]):200; int nq = (argc>5)?atoi(argv[5]):500; int k = (argc>6)?atoi(argv[6]):10; ndef = (argc>7)?parse_csv(argv[7],defef,8):parse_csv("64,128,200",defef,8); printf("Generating %d synthetic clustered vectors (dim=%d clusters=%d) ...\n", N, dim, clusters); float* data = malloc((size_t)N*dim*sizeof(float)); if (!data){ fprintf(stderr,"OOM allocating %zu bytes\n", (size_t)N*dim*sizeof(float)); return 1; } gen_synth(data, N, dim, clusters, 0.35); char lbl[64]; snprintf(lbl,sizeof lbl,"SYNTH"); run_bench(lbl, data, N, dim, nq, k, defef, ndef, 0.0); free(data); return 0; } fprintf(stderr,"unknown mode '%s'\n", argv[1]); return 2; }