635453b936
Replaces the score-fusion first cut with rank fusion, which is what the data called for. nomic's cosine scale is compressed (true matches 0.55-0.70, unrelated pairs 0.35-0.50), so an additive blend of cosine onto token-coverage is dominated by whichever leg has the wider spread. Alternation is invariant to both scales: L1, S1, L2, S2, ... deduped, capped at limit Lexical ranking is left byte-identical; the semantic ranking is computed beside it and admitted only above ENGRAM_EMBED_SEED_MIN (0.60) — Will's existing seed floor, no new tuning constant. That floor is what keeps the nonsense controls clean: a query with no real match must not be answered with its neighbours. embed-corpus.py / merge-corpus.py produce the derived corpus the semantic leg needs (76,986 vectors, nomic-embed-text, 0 failures, 11 min). Zero of 78,791 nodes carried an embedding before this; the field round-tripped through the snapshot but nothing ever wrote it. MEASURED, 38-query gold set, paired against the SAME derived corpus so the comparison isolates the code change: hit@5 34.3% -> 51.4% paraphrase 0.0% -> 38.5% MRR@10 0.294 -> 0.387 superseded 1/3 -> 2/3 outranks recall@10 33.3% -> 50.5% latency p50 1146 -> 1220ms (1.06x) exact_rare 100% -> 100% phrase 85.7% -> 85.7% nonsense 2/3 -> 2/3 6 queries fixed, 0 broken, McNemar exact p=0.0312, 0 drift across repeats. Regression guards all held. Contrast PR #135, which swapped the read path to spreading activation wholesale: phrase 85.7 -> 28.6, latency 2.81x. Correct mechanism, wrong substrate. The substrate is now present. Restores engram claim 24 (previously 0% honoured). Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
145 lines
3.0 KiB
JSON
145 lines
3.0 KiB
JSON
{
|
|
"baseline": "baseline-embcorpus",
|
|
"candidate": "hybrid-semantic",
|
|
"n_shared_queries": 38,
|
|
"fixed_by_candidate": [
|
|
"q15",
|
|
"q16",
|
|
"q20",
|
|
"q21",
|
|
"q26",
|
|
"q37"
|
|
],
|
|
"broken_by_candidate": [],
|
|
"discordant": 6,
|
|
"net_queries": 6,
|
|
"mcnemar_exact_p": 0.03125,
|
|
"min_detectable_swing_queries": 6,
|
|
"observed_run_to_run_drift_queries": 0,
|
|
"noise_floor_queries": 6,
|
|
"verdict": "candidate better",
|
|
"baseline_aggregate": {
|
|
"n_queries": 38,
|
|
"n_scored": 35,
|
|
"hit@5": 0.34285714285714286,
|
|
"recall@5": 0.26947278911564626,
|
|
"recall@10": 0.3333333333333333,
|
|
"precision@5": 0.12000000000000001,
|
|
"mrr@10": 0.2943197278911564,
|
|
"nonsense_clean": "2/3",
|
|
"superseded_outranks": "1/3",
|
|
"latency_ms_p50": 1145.9,
|
|
"latency_ms_p95": 1574.3,
|
|
"latency_ms_max": 1634.2,
|
|
"errors": 0,
|
|
"by_category": {
|
|
"associative": {
|
|
"n": 6,
|
|
"hit@5": 0.0,
|
|
"recall@5": 0.0,
|
|
"recall@10": 0.0,
|
|
"mrr@10": 0.0
|
|
},
|
|
"exact_rare": {
|
|
"n": 6,
|
|
"hit@5": 1.0,
|
|
"recall@5": 1.0,
|
|
"recall@10": 1.0,
|
|
"mrr@10": 1.0
|
|
},
|
|
"nonsense": {
|
|
"n": 3,
|
|
"clean": 2,
|
|
"avg_false_positives": 3.3333333333333335
|
|
},
|
|
"paraphrase": {
|
|
"n": 13,
|
|
"hit@5": 0.0,
|
|
"recall@5": 0.0,
|
|
"recall@10": 0.0,
|
|
"mrr@10": 0.0
|
|
},
|
|
"phrase": {
|
|
"n": 7,
|
|
"hit@5": 0.8571428571428571,
|
|
"recall@5": 0.4902210884353741,
|
|
"recall@10": 0.6666666666666666,
|
|
"mrr@10": 0.5965986394557822
|
|
},
|
|
"superseded": {
|
|
"n": 3,
|
|
"hit@5": 0.0,
|
|
"recall@5": 0.0,
|
|
"recall@10": 0.3333333333333333,
|
|
"mrr@10": 0.041666666666666664,
|
|
"outranks": 1
|
|
}
|
|
}
|
|
},
|
|
"candidate_aggregate": {
|
|
"n_queries": 38,
|
|
"n_scored": 35,
|
|
"hit@5": 0.5142857142857142,
|
|
"recall@5": 0.4409013605442177,
|
|
"recall@10": 0.5047619047619047,
|
|
"precision@5": 0.15428571428571433,
|
|
"mrr@10": 0.38746031746031745,
|
|
"nonsense_clean": "2/3",
|
|
"superseded_outranks": "2/3",
|
|
"latency_ms_p50": 1219.7,
|
|
"latency_ms_p95": 1667.1,
|
|
"latency_ms_max": 1720.2,
|
|
"errors": 0,
|
|
"by_category": {
|
|
"associative": {
|
|
"n": 6,
|
|
"hit@5": 0.0,
|
|
"recall@5": 0.0,
|
|
"recall@10": 0.0,
|
|
"mrr@10": 0.0
|
|
},
|
|
"exact_rare": {
|
|
"n": 6,
|
|
"hit@5": 1.0,
|
|
"recall@5": 1.0,
|
|
"recall@10": 1.0,
|
|
"mrr@10": 1.0
|
|
},
|
|
"nonsense": {
|
|
"n": 3,
|
|
"clean": 2,
|
|
"avg_false_positives": 3.3333333333333335
|
|
},
|
|
"paraphrase": {
|
|
"n": 13,
|
|
"hit@5": 0.38461538461538464,
|
|
"recall@5": 0.38461538461538464,
|
|
"recall@10": 0.38461538461538464,
|
|
"mrr@10": 0.17307692307692307
|
|
},
|
|
"phrase": {
|
|
"n": 7,
|
|
"hit@5": 0.8571428571428571,
|
|
"recall@5": 0.4902210884353741,
|
|
"recall@10": 0.6666666666666666,
|
|
"mrr@10": 0.6634920634920636
|
|
},
|
|
"superseded": {
|
|
"n": 3,
|
|
"hit@5": 0.3333333333333333,
|
|
"recall@5": 0.3333333333333333,
|
|
"recall@10": 0.6666666666666666,
|
|
"mrr@10": 0.2222222222222222,
|
|
"outranks": 2
|
|
}
|
|
}
|
|
},
|
|
"repeat_variance": {
|
|
"candidate": {
|
|
"runs": 2,
|
|
"hit@5_min": 0.5142857142857142,
|
|
"hit@5_max": 0.5142857142857142,
|
|
"spread_queries": 0
|
|
}
|
|
}
|
|
} |