728207aabf
The original brief targeted engram_activate's O(N*D) cosq prescan, but #109 (this branch) already retires that loop algorithmically (HNSW seed selection + lazy memoized cosine) — GPU-accelerating a loop being deleted isn't real work, so that target was dropped rather than forced. Re-investigated for a genuine remaining GPU-shaped call site (not a manufactured one): HNSW insert's candidate-list distance work is bounded- degree (M=24-48) and sequential/adaptive — too fine-grained for a GPU dispatch to pay off. No O(N^2) pairwise cosine pass exists (dedup only checks the K=8 already-selected seed slots). No concurrent multi-query traffic exists (server.el: the soul's curiosity loop is a single in-process caller). vindex_bench.c's brute_topk — the correctness oracle this same PR adds to validate HNSW recall — is the one real, unforced fit: genuine 1-query-vs-N-vectors, embarrassingly parallel, no adaptivity. Adds: - eg_cosine_batch.metal: batched cosine kernel, single- and multi-query variants, same -2.0 dim-mismatch/zero-norm sentinel as eg_cosine(). - eg_metal_cosine.h/.m: C-callable Objective-C bridge. Lazy one-time device/pipeline init, MTLResourceStorageModeShared buffers, returns false on ANY failure so callers fall back to the scalar CPU loop unconditionally — never partial, never throws. - eg_metal_cosine_stub.c: zero-dependency CPU-only implementation for non-Darwin builds (Linux CI) — same symbols, always returns false, no #ifdef needed at any call site. - build_vindex_bench.sh: one-command build, real bridge + Metal frameworks on Darwin, stub everywhere else. vindex_bench.c: brute_topk_metal / brute_topk_metal_batch call the bridge, falling back to the existing CPU brute_topk on any failure or EL_METAL_COSINE=0. The multi-query batched path exists because the first version (one GPU call per query) measured SLOWER than CPU at N~13.7k — it re-uploaded the full N*D matrix every query. Fixed by uploading the matrix once per query batch. Measured against a real nsbx-sandboxed clone of the live store (never :8742/:7770), 13,671 real embedded nodes, dim=768, 300 real queries: BRUTE-FORCE (CPU): 2.013 ms/query BRUTE-METAL (GPU): 0.117 ms/query (17.2x) id-recall vs CPU oracle: 0.9990 over 300 queries same-rank |Δdist|: max 2.98e-07, mean 7.53e-08 (float32 rounding, not a bug) Synthetic scaling sweep (13k -> 50k nodes, same dim/queries) shows the GPU speedup holding (~11x) as N grows toward the mathematical-foundations doc's 1.3M-node target, with CPU brute-force cost growing linearly as expected. Not wired into engram_activate or the daemon build (nsbx's _build_binary) — vindex_bench is a standalone offline tool, not part of the request-serving binary, so no engram_activate/server-latency claim is made here. The bridge is a reusable primitive (single eg_cosine_batch_metal + batched eg_cosine_batch_metal_multi) other call sites can adopt later without re-deriving any of this. Based on feat/reframe-region-setop (PR #109), not dev directly: the only genuine batch-cosine call site (vindex_bench.c) exists solely on this branch. Flagged explicitly in the PR description as a deliberate deviation from the original "base off dev" instruction.
400 lines
19 KiB
C
400 lines
19 KiB
C
/* vindex_bench.c — standalone proof harness for the engram HNSW ANN index.
|
||
*
|
||
* Measures brute-force cosine top-k (the correctness ORACLE) vs vindex_search
|
||
* (HNSW) on: (a) the REAL paged store harvested read-only, and (b) synthetic
|
||
* clustered data at several sizes to trace the scaling curve. Reports build time,
|
||
* per-query latency (brute vs HNSW), and recall@k (HNSW top-k vs brute top-k).
|
||
*
|
||
* Read-only: never opens a socket, never writes the store. Safe on an nsbx clone.
|
||
*
|
||
* Also runs the brute-force oracle a second way, through
|
||
* eg_cosine_batch_metal() (Apple/Metal only — see eg_metal_cosine.h), and
|
||
* reports its latency + a correctness check against the CPU oracle
|
||
* side-by-side with the existing CPU-vs-HNSW numbers. EL_METAL_COSINE=0
|
||
* forces CPU-only.
|
||
*
|
||
* Build (macOS):
|
||
* cc -O2 -std=c11 -x objective-c -c eg_metal_cosine.m -o eg_metal_cosine.o \
|
||
* -framework Metal -framework Foundation
|
||
* cc -O2 -std=c11 vindex_bench.c engram_vindex.c eg_metal_cosine.o -lm \
|
||
* -framework Metal -framework Foundation -o vindex_bench
|
||
* Build (Linux / no Metal): omit eg_metal_cosine.o entirely and instead link
|
||
* a CPU-only stub translation unit that defines eg_cosine_batch_metal() /
|
||
* eg_cosine_batch_metal_available() returning false — this file never
|
||
* references Metal directly, only the plain-C header.
|
||
* Usage: vindex_bench store <neuron.egm> <dim> [nqueries] [k] [ef_csv]
|
||
* vindex_bench synth <N> [dim] [clusters] [nqueries] [k] [ef_csv]
|
||
*/
|
||
#include "engram_vindex.h"
|
||
#include "eg_metal_cosine.h"
|
||
#include <stdio.h>
|
||
#include <stdlib.h>
|
||
#include <string.h>
|
||
#include <math.h>
|
||
#include <stdint.h>
|
||
#include <time.h>
|
||
|
||
/* ── deterministic PRNG (splitmix64) so runs are reproducible ─────────────── */
|
||
static uint64_t g_seed = 0xD1B54A32D192ED03ULL;
|
||
static uint64_t sm(void){
|
||
uint64_t z = (g_seed += 0x9E3779B97F4A7C15ULL);
|
||
z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9ULL;
|
||
z = (z ^ (z >> 27)) * 0x94D049BB133111EBULL;
|
||
return z ^ (z >> 31);
|
||
}
|
||
static double urand(void){ return (double)((sm() >> 11) + 1) * (1.0/9007199254740993.0); }
|
||
static double grand(void){ /* Box-Muller */
|
||
double u1 = urand(), u2 = urand();
|
||
return sqrt(-2.0*log(u1)) * cos(2.0*M_PI*u2);
|
||
}
|
||
|
||
static double now_s(void){
|
||
struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts);
|
||
return (double)ts.tv_sec + (double)ts.tv_nsec*1e-9;
|
||
}
|
||
|
||
/* L2-normalise a row in place. */
|
||
static void l2norm(float* v, int dim){
|
||
double ss = 0; for (int i=0;i<dim;i++) ss += (double)v[i]*v[i];
|
||
if (ss > 0){ float inv = (float)(1.0/sqrt(ss)); for (int i=0;i<dim;i++) v[i]*=inv; }
|
||
}
|
||
|
||
/* Brute-force top-k by cosine distance (1 - dot on normalised vecs).
|
||
* data is n*dim, already L2-normalised. Writes k node ids (row indices) into
|
||
* out_ids ascending by distance. Returns nothing; assumes k<=n. */
|
||
static void brute_topk(const float* data, int n, int dim, const float* q,
|
||
int k, int* out_ids, float* out_d){
|
||
/* maintain a small sorted array of the k best (ascending distance). */
|
||
for (int i=0;i<k;i++){ out_ids[i]=-1; out_d[i]=2.0f+1.0f; }
|
||
for (int i=0;i<n;i++){
|
||
const float* r = data + (size_t)i*dim;
|
||
float s0=0,s1=0,s2=0,s3=0; int j=0;
|
||
for (; j+4<=dim; j+=4){ s0+=q[j]*r[j]; s1+=q[j+1]*r[j+1]; s2+=q[j+2]*r[j+2]; s3+=q[j+3]*r[j+3]; }
|
||
float dot=(s0+s1)+(s2+s3); for (; j<dim; j++) dot+=q[j]*r[j];
|
||
float d = 1.0f - dot;
|
||
if (d >= out_d[k-1]) continue;
|
||
int p = k-1;
|
||
while (p>0 && out_d[p-1] > d){ out_d[p]=out_d[p-1]; out_ids[p]=out_ids[p-1]; p--; }
|
||
out_d[p]=d; out_ids[p]=i;
|
||
}
|
||
}
|
||
|
||
/* GPU-accelerated variant of brute_topk: same oracle, same contract, same
|
||
* output — computes all n distances via eg_cosine_batch_metal() instead of
|
||
* one C loop, then does the identical top-k selection over the result.
|
||
*
|
||
* data is already L2-normalised (vindex_bench's convention throughout), so
|
||
* eg_cosine()'s general unnormalised cosine and this file's "distance =
|
||
* 1 - dot" both reduce to the same number here (a unit vector's norm is 1,
|
||
* so cosine == dot). Passing pre-normalised rows through the general-purpose
|
||
* batch kernel is deliberate: it proves the SAME primitive that would serve
|
||
* el_runtime.c's raw/unnormalised embeddings also serves this oracle without
|
||
* a second code path.
|
||
*
|
||
* Returns false (out_ids/out_d untouched) if the GPU path is unavailable or
|
||
* fails for any reason — caller must fall back to brute_topk(). Never
|
||
* partial: either the full top-k was computed on GPU, or nothing was. */
|
||
/* EL_METAL_COSINE: 0/off/false disables the GPU path outright (falls back to
|
||
* brute_topk() every time), matching el_runtime.c's own gate for the same
|
||
* env var. Unset or any other value = auto (try Metal, fall back on failure). */
|
||
static bool g_metal_env_checked = false;
|
||
static bool g_metal_disabled_by_env = false;
|
||
static void eg_metal_check_env_once(void){
|
||
if (g_metal_env_checked) return;
|
||
g_metal_env_checked = true;
|
||
const char* v = getenv("EL_METAL_COSINE");
|
||
if (v && (v[0]=='0' || v[0]=='n' || v[0]=='N' || v[0]=='f' || v[0]=='F'))
|
||
g_metal_disabled_by_env = true;
|
||
}
|
||
static bool brute_topk_metal(const float* data, int n, int dim, const float* q,
|
||
int k, int* out_ids, float* out_d){
|
||
eg_metal_check_env_once();
|
||
if (g_metal_disabled_by_env) return false;
|
||
|
||
const float** row_ptrs = malloc((size_t)n * sizeof(float*));
|
||
int32_t* dims = malloc((size_t)n * sizeof(int32_t));
|
||
double* scores = malloc((size_t)n * sizeof(double));
|
||
if (!row_ptrs || !dims || !scores) { free(row_ptrs); free(dims); free(scores); return false; }
|
||
|
||
for (int i = 0; i < n; i++) {
|
||
row_ptrs[i] = data + (size_t)i * dim;
|
||
dims[i] = dim;
|
||
}
|
||
|
||
bool ok = eg_cosine_batch_metal(q, dim, row_ptrs, dims, n, scores);
|
||
free(row_ptrs); free(dims);
|
||
if (!ok) { free(scores); return false; }
|
||
|
||
for (int i = 0; i < k; i++) { out_ids[i] = -1; out_d[i] = 3.0f; }
|
||
for (int i = 0; i < n; i++) {
|
||
float d = 1.0f - (float)scores[i]; /* same distance convention as brute_topk */
|
||
if (d >= out_d[k-1]) continue;
|
||
int p = k - 1;
|
||
while (p > 0 && out_d[p-1] > d) { out_d[p] = out_d[p-1]; out_ids[p] = out_ids[p-1]; p--; }
|
||
out_d[p] = d; out_ids[p] = i;
|
||
}
|
||
free(scores);
|
||
return true;
|
||
}
|
||
|
||
/* Batched sibling of brute_topk_metal: computes top-k for ALL nq queries in
|
||
* ONE eg_cosine_batch_metal_multi() call, uploading node_matrix exactly
|
||
* once instead of once per query. out_ids/out_d are nq*k, row-major
|
||
* (query i's results at out_ids+i*k / out_d+i*k) — same layout run_bench
|
||
* already uses for `gt`/per-query scratch. Returns false (nothing written)
|
||
* on any failure; caller falls back to the per-query CPU brute_topk loop. */
|
||
static bool brute_topk_metal_batch(const float* data, int n, int dim,
|
||
const float* queries, int nq,
|
||
int k, int* out_ids, float* out_d){
|
||
eg_metal_check_env_once();
|
||
if (g_metal_disabled_by_env) return false;
|
||
|
||
const float** row_ptrs = malloc((size_t)n * sizeof(float*));
|
||
int32_t* dims = malloc((size_t)n * sizeof(int32_t));
|
||
double* scores = malloc((size_t)nq * (size_t)n * sizeof(double));
|
||
if (!row_ptrs || !dims || !scores) { free(row_ptrs); free(dims); free(scores); return false; }
|
||
|
||
for (int i = 0; i < n; i++) { row_ptrs[i] = data + (size_t)i * dim; dims[i] = dim; }
|
||
|
||
bool ok = eg_cosine_batch_metal_multi(queries, dim, nq, row_ptrs, dims, n, scores);
|
||
free(row_ptrs); free(dims);
|
||
if (!ok) { free(scores); return false; }
|
||
|
||
for (int qi = 0; qi < nq; qi++) {
|
||
int* ids = out_ids + (size_t)qi * k;
|
||
float* ds = out_d + (size_t)qi * k;
|
||
const double* srow = scores + (size_t)qi * n;
|
||
for (int i = 0; i < k; i++) { ids[i] = -1; ds[i] = 3.0f; }
|
||
for (int i = 0; i < n; i++) {
|
||
float d = 1.0f - (float)srow[i];
|
||
if (d >= ds[k-1]) continue;
|
||
int p = k - 1;
|
||
while (p > 0 && ds[p-1] > d) { ds[p] = ds[p-1]; ids[p] = ids[p-1]; p--; }
|
||
ds[p] = d; ids[p] = i;
|
||
}
|
||
}
|
||
free(scores);
|
||
return true;
|
||
}
|
||
|
||
/* recall@k: |brute_topk ∩ hnsw_topk| / k. Both are id arrays of length k. */
|
||
static double recall_at_k(const int* gt, const uint64_t* ann, int nann, int k){
|
||
int hit = 0;
|
||
for (int i=0;i<k;i++){
|
||
if (gt[i] < 0) continue;
|
||
for (int j=0;j<nann;j++){ if ((int)ann[j] == gt[i]){ hit++; break; } }
|
||
}
|
||
return (double)hit / (double)k;
|
||
}
|
||
|
||
/* Parse "64,128,256" into an int array; returns count. */
|
||
static int parse_csv(const char* s, int* out, int maxo){
|
||
int n=0; if(!s||!*s) return 0;
|
||
const char* p=s;
|
||
while(*p && n<maxo){ out[n++]=atoi(p); while(*p && *p!=',') p++; if(*p==',') p++; }
|
||
return n;
|
||
}
|
||
|
||
/* Generate n unit vectors on a LOW-DIMENSIONAL MANIFOLD, the property that makes
|
||
* real text embeddings tractable for ANN: each vector is a fixed random linear map
|
||
* A (dim × LATENT) applied to a latent gaussian z ∈ R^LATENT, plus small ambient
|
||
* noise, then L2-normalised. Points therefore lie near a `latent`-dim subspace, so
|
||
* every point has a well-defined tight neighbourhood (high recall) and the HNSW
|
||
* graph is cheap to build — unlike near-isotropic 768-d gaussians, where the curse
|
||
* of dimensionality makes all points near-equidistant (no structure → slow build,
|
||
* low recall) and unlike tight clusters (near-duplicates → artificial top-k ties).
|
||
* `sigma` is the ambient-noise scale. This reproduces the intrinsic-dimensionality
|
||
* regime of nomic embeddings, so the scaling curve reflects real-corpus behaviour. */
|
||
#define SYNTH_LATENT 48
|
||
static void gen_synth(float* data, int n, int dim, int clusters, double sigma){
|
||
(void)clusters;
|
||
float* A = malloc((size_t)dim*SYNTH_LATENT*sizeof(float)); /* fixed random basis */
|
||
for (size_t i=0;i<(size_t)dim*SYNTH_LATENT;i++) A[i]=(float)grand();
|
||
float z[SYNTH_LATENT];
|
||
for (int i=0;i<n;i++){
|
||
for (int l=0;l<SYNTH_LATENT;l++) z[l]=(float)grand();
|
||
float* v = data+(size_t)i*dim;
|
||
for (int j=0;j<dim;j++){
|
||
float acc = (float)(sigma*grand());
|
||
const float* row = A + (size_t)j*SYNTH_LATENT;
|
||
for (int l=0;l<SYNTH_LATENT;l++) acc += row[l]*z[l];
|
||
v[j]=acc;
|
||
}
|
||
l2norm(v, dim);
|
||
}
|
||
free(A);
|
||
}
|
||
|
||
/* Build M / ef_construction come from env (VIDX_M / VIDX_EFC) so the scaling
|
||
* sweep can trade build cost against graph quality without a recompile. 0 = default. */
|
||
static int env_int(const char* k, int dflt){ const char* s=getenv(k); return (s&&*s)?atoi(s):dflt; }
|
||
|
||
/* Run the full brute-vs-HNSW comparison over an already-normalised dataset. */
|
||
static void run_bench(const char* label, float* data, int n, int dim,
|
||
int nq, int k, int* efs, int nef, double build_s){
|
||
(void)build_s;
|
||
int bM = env_int("VIDX_M", 0), bEFC = env_int("VIDX_EFC", 0);
|
||
printf("\n=== %s : N=%d dim=%d k=%d queries=%d ===\n", label, n, dim, k, nq);
|
||
|
||
/* build the index once (shared across ef settings). */
|
||
double t0 = now_s();
|
||
VIndex* ix = vindex_create(dim, bM, bEFC);
|
||
for (int i=0;i<n;i++) vindex_insert(ix, (uint64_t)i, data + (size_t)i*dim);
|
||
double bt = now_s()-t0;
|
||
printf("HNSW build: M=%d ef_construction=%d -> %.3f s (%.1f k nodes/s)\n",
|
||
bM?bM:VINDEX_DEFAULT_M, bEFC?bEFC:VINDEX_DEFAULT_EF_CONSTRUCTION, bt, n/1000.0/bt);
|
||
|
||
/* choose query vectors: perturb random dataset rows (near-but-not-identical). */
|
||
int* qidx = malloc((size_t)nq*sizeof(int));
|
||
float* qv = malloc((size_t)nq*dim*sizeof(float));
|
||
for (int i=0;i<nq;i++){
|
||
int r = (int)(sm() % (uint64_t)n);
|
||
qidx[i]=r;
|
||
float* dst = qv+(size_t)i*dim; const float* src = data+(size_t)r*dim;
|
||
for (int j=0;j<dim;j++) dst[j] = src[j] + (float)(0.01*grand());
|
||
l2norm(dst, dim);
|
||
}
|
||
|
||
/* ground truth: brute-force top-k for every query (also the oracle latency).
|
||
* gd is nq*k (one real slot per query, not a shared scratch buffer) so the
|
||
* GPU comparison below can diff against every query's actual distances,
|
||
* not just whichever query happened to run last. */
|
||
int* gt = malloc((size_t)nq*k*sizeof(int));
|
||
float* gd = malloc((size_t)nq*k*sizeof(float));
|
||
double tb0 = now_s();
|
||
for (int i=0;i<nq;i++) brute_topk(data, n, dim, qv+(size_t)i*dim, k, gt+(size_t)i*k, gd+(size_t)i*k);
|
||
double brute_ms = (now_s()-tb0)*1000.0/nq;
|
||
printf("BRUTE-FORCE : %8.3f ms/query (oracle; O(N*D), CPU)\n", brute_ms);
|
||
|
||
/* GPU-accelerated oracle: SAME nq queries, SAME top-k contract, via ONE
|
||
* eg_cosine_batch_metal_multi() call (uploads node_matrix once, not once
|
||
* per query — see brute_topk_metal_batch). Run only if the GPU path is
|
||
* actually available (checked once) — never fabricated, never assumed.
|
||
* Verified against the CPU ground truth computed above: id-recall across
|
||
* ALL nq queries, plus the actual max distance delta across every
|
||
* (query, rank) pair that was compared — not a single spot check. */
|
||
eg_metal_check_env_once();
|
||
if (!g_metal_disabled_by_env && eg_cosine_batch_metal_available()) {
|
||
int* gtm = malloc((size_t)nq*k*sizeof(int));
|
||
float* gdm = malloc((size_t)nq*k*sizeof(float));
|
||
double tm0 = now_s();
|
||
bool ok = brute_topk_metal_batch(data, n, dim, qv, nq, k, gtm, gdm);
|
||
double metal_ms = (now_s()-tm0)*1000.0/nq;
|
||
if (ok) {
|
||
double rec_sum = 0; double max_ddiff = 0; double sum_ddiff = 0; int compared = 0;
|
||
for (int i=0;i<nq;i++) {
|
||
const int* ids_gt = gt+(size_t)i*k;
|
||
const float* d_gt = gd+(size_t)i*k;
|
||
const int* ids_m = gtm+(size_t)i*k;
|
||
const float* d_m = gdm+(size_t)i*k;
|
||
uint64_t idset[512]; int m = (k<512)?k:512;
|
||
for (int j=0;j<m;j++) idset[j] = (uint64_t)ids_m[j];
|
||
rec_sum += recall_at_k(ids_gt, idset, m, k);
|
||
/* same-rank distance delta — valid whenever both sides agree on
|
||
* the id at that rank (true almost always, given ~100% recall;
|
||
* a rank where they disagree isn't a meaningful delta to diff). */
|
||
for (int j=0;j<k;j++) {
|
||
if (ids_gt[j] == ids_m[j]) {
|
||
double diff = fabs((double)d_gt[j]-(double)d_m[j]);
|
||
if (diff>max_ddiff) max_ddiff=diff;
|
||
sum_ddiff += diff; compared++;
|
||
}
|
||
}
|
||
}
|
||
printf("BRUTE-METAL : %8.3f ms/query (%.1fx vs CPU brute; id-recall %.4f vs CPU oracle over %d queries; same-rank |Δdist|: max %.2e, mean %.2e over %d compared)\n",
|
||
metal_ms, brute_ms/metal_ms, rec_sum/nq, nq, max_ddiff, compared?sum_ddiff/compared:0.0, compared);
|
||
} else {
|
||
printf("BRUTE-METAL : GPU batch call failed/unavailable mid-run — skipped\n");
|
||
}
|
||
free(gtm); free(gdm);
|
||
} else {
|
||
printf("BRUTE-METAL : no Metal device/pipeline available — CPU-only\n");
|
||
}
|
||
|
||
/* HNSW at each ef. */
|
||
uint64_t* aid = malloc((size_t)k*sizeof(uint64_t));
|
||
float* ad = malloc((size_t)k*sizeof(float));
|
||
printf("%-6s %14s %12s %10s\n", "ef", "HNSW ms/query", "speedup", "recall@k");
|
||
for (int e=0;e<nef;e++){
|
||
int ef = efs[e];
|
||
double th0 = now_s();
|
||
double rec_sum = 0;
|
||
for (int i=0;i<nq;i++){
|
||
int m = vindex_search(ix, qv+(size_t)i*dim, k, ef, aid, ad);
|
||
rec_sum += recall_at_k(gt+(size_t)i*k, aid, m, k);
|
||
}
|
||
double hnsw_ms = (now_s()-th0)*1000.0/nq;
|
||
printf("%-6d %14.4f %11.1fx %10.4f\n", ef, hnsw_ms, brute_ms/hnsw_ms, rec_sum/nq);
|
||
}
|
||
|
||
free(qidx); free(qv); free(gt); free(gd); free(aid); free(ad);
|
||
vindex_free(ix);
|
||
}
|
||
|
||
int main(int argc, char** argv){
|
||
setvbuf(stdout, NULL, _IOLBF, 0); /* line-buffered so progress streams to a log */
|
||
if (argc < 2){ fprintf(stderr,"usage: %s store <path> <dim> [nq] [k] [ef_csv] | synth <N> [dim] [clusters] [nq] [k] [ef_csv] | sweep <dim> <N_csv> [nq] [k] [ef_csv]\n", argv[0]); return 2; }
|
||
int defef[8]; int ndef;
|
||
|
||
if (strcmp(argv[1],"sweep")==0){
|
||
if (argc < 4){ fprintf(stderr,"sweep needs <dim> <N_csv>\n"); return 2; }
|
||
int dim = atoi(argv[2]);
|
||
int Ns[16]; int nN = parse_csv(argv[3], Ns, 16);
|
||
int nq = (argc>4)?atoi(argv[4]):200;
|
||
int k = (argc>5)?atoi(argv[5]):10;
|
||
ndef = (argc>6)?parse_csv(argv[6],defef,8):parse_csv("64,128,200",defef,8);
|
||
for (int s=0;s<nN;s++){
|
||
int N = Ns[s];
|
||
float* data = malloc((size_t)N*dim*sizeof(float));
|
||
if (!data){ fprintf(stderr,"OOM at N=%d\n",N); continue; }
|
||
int clusters = N/100; if (clusters < 64) clusters = 64;
|
||
gen_synth(data, N, dim, clusters, 1.0);
|
||
char lbl[64]; snprintf(lbl,sizeof lbl,"SYNTH N=%d", N);
|
||
run_bench(lbl, data, N, dim, nq, k, defef, ndef, 0.0);
|
||
free(data);
|
||
}
|
||
return 0;
|
||
}
|
||
|
||
if (strcmp(argv[1],"store")==0){
|
||
if (argc < 4){ fprintf(stderr,"store needs <path> <dim>\n"); return 2; }
|
||
const char* path = argv[2]; int dim = atoi(argv[3]);
|
||
int nq = (argc>4)?atoi(argv[4]):500;
|
||
int k = (argc>5)?atoi(argv[5]):10;
|
||
ndef = (argc>6)?parse_csv(argv[6],defef,8):parse_csv("32,64,128,200,400",defef,8);
|
||
printf("Harvesting emb vectors from %s (dim=%d) ...\n", path, dim);
|
||
float* data=NULL; int n=0;
|
||
double t0=now_s();
|
||
int h = vindex_harvest_from_store(path, dim, &data, NULL, &n);
|
||
double harvest_s = now_s()-t0;
|
||
if (h < 0 || n == 0){ fprintf(stderr,"harvest failed (h=%d n=%d) — wrong dim or path?\n", h, n); return 1; }
|
||
printf("Harvested %d live embedded nodes in %.2f s\n", n, harvest_s);
|
||
for (int i=0;i<n;i++) l2norm(data+(size_t)i*dim, dim); /* oracle needs normalised */
|
||
if (nq > n) nq = n;
|
||
run_bench("REAL STORE", data, n, dim, nq, k, defef, ndef, 0.0);
|
||
free(data);
|
||
return 0;
|
||
}
|
||
|
||
if (strcmp(argv[1],"synth")==0){
|
||
if (argc < 3){ fprintf(stderr,"synth needs <N>\n"); return 2; }
|
||
int N = atoi(argv[2]);
|
||
int dim = (argc>3)?atoi(argv[3]):768;
|
||
int clusters = (argc>4)?atoi(argv[4]):200;
|
||
int nq = (argc>5)?atoi(argv[5]):500;
|
||
int k = (argc>6)?atoi(argv[6]):10;
|
||
ndef = (argc>7)?parse_csv(argv[7],defef,8):parse_csv("64,128,200",defef,8);
|
||
printf("Generating %d synthetic clustered vectors (dim=%d clusters=%d) ...\n", N, dim, clusters);
|
||
float* data = malloc((size_t)N*dim*sizeof(float));
|
||
if (!data){ fprintf(stderr,"OOM allocating %zu bytes\n", (size_t)N*dim*sizeof(float)); return 1; }
|
||
gen_synth(data, N, dim, clusters, 0.35);
|
||
char lbl[64]; snprintf(lbl,sizeof lbl,"SYNTH");
|
||
run_bench(lbl, data, N, dim, nq, k, defef, ndef, 0.0);
|
||
free(data);
|
||
return 0;
|
||
}
|
||
|
||
fprintf(stderr,"unknown mode '%s'\n", argv[1]);
|
||
return 2;
|
||
}
|