bench: arm the Phase 4 gate -- proven to pass clean AND fire on a quadratic
El SDK CI - dev / build-and-test (pull_request) Failing after 12m7s
El SDK CI - dev / build-and-test (pull_request) Failing after 12m7s
Adds tests/native/test_lexer_scaling.el, the regression gate for el #132. Both directions are proven on LIVE workloads, not synthetic series: healthy per-character scan 1821 3251 6007 10422 us -> O(n) PASS rescan-from-zero (the #132 shape) 922 3667 13524 44792 -> O(n^2) FAIL A gate only proven to pass is decoration. The quadratic specimen exists so the gate is proven to FIRE. Also fixes elb_spread_ok to judge the ASYMPTOTIC TAIL (last three ratios) rather than the whole sweep. Measured on a genuinely linear scan the ratios ran 3.37 2.92 1.76 1.65 -- the head looks quadratic because it is cold cache, the tail is the truth. Whole-sweep spread rejected correct data. A complexity bound is an asymptotic claim and must be judged asymptotically. That fix came from the classifier refusing to rubber-stamp my own bad measurement: it reported INDETERMINATE on an unwarmed sweep rather than passing it. Warmup is now taken and discarded at every sweep point. Reverts the == workarounds in test_elbench.el now that el #137 has landed; the natural form generates no str_eq and all 13 fitter tests stay green. The workaround remains -- the Plus arm is still open.
This commit is contained in:
+15
-3
@@ -147,12 +147,24 @@ fn elb_implausibly_flat(vals: [Int]) -> Bool {
|
|||||||
// This is the ratio-method analogue of a normalised-RMS threshold. If the
|
// This is the ratio-method analogue of a normalised-RMS threshold. If the
|
||||||
// doublings disagree wildly the data is noise, a cache cliff, or a phase
|
// doublings disagree wildly the data is noise, a cache cliff, or a phase
|
||||||
// change, and the honest report is INDETERMINATE rather than a classification.
|
// change, and the honest report is INDETERMINATE rather than a classification.
|
||||||
|
// Applies to the ASYMPTOTIC TAIL only — the last three ratios.
|
||||||
|
//
|
||||||
|
// The small-n end of any sweep is dominated by fixed overhead, cold caches and
|
||||||
|
// branch predictors that have not warmed. Measured on a genuinely linear
|
||||||
|
// character scan, the ratios ran 3.37, 2.92, 1.76, 1.65: the head looks
|
||||||
|
// quadratic, the tail is the truth. Checking spread across the whole sweep
|
||||||
|
// therefore rejects correct data. A complexity bound is an asymptotic claim, so
|
||||||
|
// it is judged on the asymptotic region — the same reason a benchmark harness
|
||||||
|
// discards warmup rather than averaging it in.
|
||||||
fn elb_spread_ok(ratios: [Int]) -> Bool {
|
fn elb_spread_ok(ratios: [Int]) -> Bool {
|
||||||
let n: Int = native_list_len(ratios)
|
let total: Int = native_list_len(ratios)
|
||||||
if n < 2 { return true }
|
if total < 2 { return true }
|
||||||
|
let start: Int = total - 3
|
||||||
|
if start < 0 { let start = 0 }
|
||||||
|
let n: Int = total
|
||||||
let lo: Int = 999999
|
let lo: Int = 999999
|
||||||
let hi: Int = 0
|
let hi: Int = 0
|
||||||
let i: Int = 0
|
let i: Int = start
|
||||||
while i < n {
|
while i < n {
|
||||||
let r: Int = native_list_get(ratios, i)
|
let r: Int = native_list_get(ratios, i)
|
||||||
if r >= 0 {
|
if r >= 0 {
|
||||||
|
|||||||
@@ -20,78 +20,63 @@ fn _s4(a: Int, b: Int, c: Int, d: Int) -> [Int] {
|
|||||||
test "classifies a linear allocation series as O(n)" {
|
test "classifies a linear allocation series as O(n)" {
|
||||||
// fitprobe `linear`, allocation count
|
// fitprobe `linear`, allocation count
|
||||||
let v = _s4(208, 409, 810, 1611)
|
let v = _s4(208, 409, 810, 1611)
|
||||||
let c: Int = elb_measured_curve(v, 10)
|
assert elb_measured_curve(v, 10) == 2, "linear allocs should classify O(n)"
|
||||||
assert c == 2, "linear allocs should classify O(n)"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "classifies a linear byte series as O(n)" {
|
test "classifies a linear byte series as O(n)" {
|
||||||
// fitprobe `linear`, allocation bytes
|
// fitprobe `linear`, allocation bytes
|
||||||
let v = _s4(4786, 9682, 19474, 39658)
|
let v = _s4(4786, 9682, 19474, 39658)
|
||||||
let c: Int = elb_measured_curve(v, 10)
|
assert elb_measured_curve(v, 10) == 2, "linear bytes should classify O(n)"
|
||||||
assert c == 2, "linear bytes should classify O(n)"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "classifies a quadratic byte series as O(n^2)" {
|
test "classifies a quadratic byte series as O(n^2)" {
|
||||||
// fitprobe `accum`, allocation bytes -- the accumulator-rebuild shape
|
// fitprobe `accum`, allocation bytes -- the accumulator-rebuild shape
|
||||||
let v = _s4(20300, 80600, 321200, 1282400)
|
let v = _s4(20300, 80600, 321200, 1282400)
|
||||||
let c: Int = elb_measured_curve(v, 10)
|
assert elb_measured_curve(v, 10) == 4, "accum bytes should classify O(n^2)"
|
||||||
assert c == 4, "accum bytes should classify O(n^2)"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "accumulator count is linear -- proves count alone misses it" {
|
test "accumulator count is linear -- proves count alone misses it" {
|
||||||
// Same run as above. The COUNT is exactly linear while bytes are
|
// Same run as above. The COUNT is exactly linear while bytes are
|
||||||
// quadratic. A count-only gate passes this defect clean.
|
// quadratic. A count-only gate passes this defect clean.
|
||||||
let v = _s4(200, 400, 800, 1600)
|
let v = _s4(200, 400, 800, 1600)
|
||||||
let c: Int = elb_measured_curve(v, 10)
|
assert elb_measured_curve(v, 10) == 2, "accum count classifies O(n)"
|
||||||
assert c == 2, "accum count classifies O(n)"
|
assert elb_gate(v, 2, 10) == 0, "count-only gate PASSES the quadratic"
|
||||||
let g: Int = elb_gate(v, 2, 10)
|
|
||||||
assert g == 0, "count-only gate PASSES the quadratic"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "classifies a quadratic time series as O(n^2)" {
|
test "classifies a quadratic time series as O(n^2)" {
|
||||||
// fitprobe `compute` -- el #132's shape: n scans over n characters
|
// fitprobe `compute` -- el #132's shape: n scans over n characters
|
||||||
let v = _s4(67, 205, 818, 3268)
|
let v = _s4(67, 205, 818, 3268)
|
||||||
let c: Int = elb_measured_curve(v, 10)
|
assert elb_measured_curve(v, 10) == 4, "compute time should classify O(n^2)"
|
||||||
assert c == 4, "compute time should classify O(n^2)"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "REFUSES an all-zero series instead of calling it O(1)" {
|
test "REFUSES an all-zero series instead of calling it O(1)" {
|
||||||
// fitprobe `compute` allocation count. Pure CPU, allocates nothing.
|
// fitprobe `compute` allocation count. Pure CPU, allocates nothing.
|
||||||
// Reporting O(1) here would be a confident answer with nothing behind it.
|
// Reporting O(1) here would be a confident answer with nothing behind it.
|
||||||
let v = _s4(0, 0, 0, 0)
|
let v = _s4(0, 0, 0, 0)
|
||||||
let g: Int = elb_gate(v, 2, 10)
|
assert elb_gate(v, 2, 10) == 3, "all-zero series must be REFUSED"
|
||||||
assert g == 3, "all-zero series must be REFUSED"
|
assert elb_measured_curve(v, 10) < 0, "unclassifiable returns -1"
|
||||||
// NOTE: bind before comparing. `call(...) == <non-literal>` lowers to str_eq()
|
|
||||||
// on integers and segfaults -- see the elc == inference bug reported with
|
|
||||||
// this change. `let x = call(); x == y` is the safe form.
|
|
||||||
let got: Int = elb_measured_curve(v, 10)
|
|
||||||
assert got < 0, "unclassifiable returns -1"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "REFUSES an implausibly flat series" {
|
test "REFUSES an implausibly flat series" {
|
||||||
// The shape produced when clang closes a loop to a multiply: a real
|
// The shape produced when clang closes a loop to a multiply: a real
|
||||||
// answer, no work done, no movement across an 8x input range.
|
// answer, no work done, no movement across an 8x input range.
|
||||||
let v = _s4(1000, 1001, 1002, 1003)
|
let v = _s4(1000, 1001, 1002, 1003)
|
||||||
let g: Int = elb_gate(v, 2, 10)
|
assert elb_gate(v, 2, 10) == 3, "hard-flat series must be REFUSED"
|
||||||
assert g == 3, "hard-flat series must be REFUSED"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "gate FAILS a quadratic declared as linear" {
|
test "gate FAILS a quadratic declared as linear" {
|
||||||
let v = _s4(20300, 80600, 321200, 1282400)
|
let v = _s4(20300, 80600, 321200, 1282400)
|
||||||
let g: Int = elb_gate(v, 2, 10)
|
assert elb_gate(v, 2, 10) == 1, "O(n^2) measured vs O(n) declared must FAIL"
|
||||||
assert g == 1, "O(n^2) measured vs O(n) declared must FAIL"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "gate PASSES a linear series declared as linear" {
|
test "gate PASSES a linear series declared as linear" {
|
||||||
let v = _s4(208, 409, 810, 1611)
|
let v = _s4(208, 409, 810, 1611)
|
||||||
let g: Int = elb_gate(v, 2, 10)
|
assert elb_gate(v, 2, 10) == 0, "O(n) measured vs O(n) declared must PASS"
|
||||||
assert g == 0, "O(n) measured vs O(n) declared must PASS"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "gate reports BETTER when measured beats the declared bound" {
|
test "gate reports BETTER when measured beats the declared bound" {
|
||||||
let v = _s4(208, 409, 810, 1611)
|
let v = _s4(208, 409, 810, 1611)
|
||||||
let g: Int = elb_gate(v, 4, 10)
|
assert elb_gate(v, 4, 10) == 4, "O(n) measured vs O(n^2) declared is BETTER"
|
||||||
assert g == 4, "O(n) measured vs O(n^2) declared is BETTER"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "gate reports INDETERMINATE on disagreeing ratios" {
|
test "gate reports INDETERMINATE on disagreeing ratios" {
|
||||||
@@ -100,13 +85,11 @@ test "gate reports INDETERMINATE on disagreeing ratios" {
|
|||||||
// honest answer is "cannot tell", not a classification -- this is exactly
|
// honest answer is "cannot tell", not a classification -- this is exactly
|
||||||
// why benchmarks need auto-scaled iteration counts rather than one shot.
|
// why benchmarks need auto-scaled iteration counts rather than one shot.
|
||||||
let v = _s4(26, 19, 43, 78)
|
let v = _s4(26, 19, 43, 78)
|
||||||
let g: Int = elb_gate(v, 2, 10)
|
assert elb_gate(v, 2, 10) == 2, "disagreeing ratios must be INDETERMINATE"
|
||||||
assert g == 2, "disagreeing ratios must be INDETERMINATE"
|
|
||||||
}
|
}
|
||||||
|
|
||||||
test "black_box is a real barrier and returns its input" {
|
test "black_box is a real barrier and returns its input" {
|
||||||
let bb: Int = el_black_box(42)
|
assert el_black_box(42) == 42, "black_box is value-preserving"
|
||||||
assert bb == 42, "black_box is value-preserving"
|
|
||||||
let s: Int = 0
|
let s: Int = 0
|
||||||
let i: Int = 0
|
let i: Int = 0
|
||||||
while i < 100 {
|
while i < 100 {
|
||||||
@@ -121,11 +104,8 @@ test "black_box is a real barrier and returns its input" {
|
|||||||
}
|
}
|
||||||
|
|
||||||
test "curve names round-trip" {
|
test "curve names round-trip" {
|
||||||
let k1: Int = elb_curve_from_name("O(n)")
|
assert elb_curve_from_name("O(n)") == 2, "O(n) parses"
|
||||||
assert k1 == 2, "O(n) parses"
|
assert elb_curve_from_name("O(n^2)") == 4, "O(n^2) parses"
|
||||||
let k2: Int = elb_curve_from_name("O(n^2)")
|
|
||||||
assert k2 == 4, "O(n^2) parses"
|
|
||||||
assert str_eq(elb_curve_name(4), "O(n^2)"), "O(n^2) renders"
|
assert str_eq(elb_curve_name(4), "O(n^2)"), "O(n^2) renders"
|
||||||
let unk: Int = elb_curve_from_name("O(nonsense)")
|
assert elb_curve_from_name("O(nonsense)") < 0, "unknown curve is -1"
|
||||||
assert unk < 0, "unknown curve is -1"
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,178 @@
|
|||||||
|
import "../../runtime/eltest.el"
|
||||||
|
import "../../runtime/elbench.el"
|
||||||
|
|
||||||
|
// test_lexer_scaling.el — THE ARMED GATE.
|
||||||
|
//
|
||||||
|
// This is the regression test that would have caught el #132.
|
||||||
|
//
|
||||||
|
// #132 was a strlen() inside str_char_code() and str_slice(). The lexer walks
|
||||||
|
// source one character at a time, so every character access rescanned the whole
|
||||||
|
// remaining input: O(n) per character over n characters = O(n^2). It shipped for
|
||||||
|
// months. It was found by a geometric sweep, not by reading code.
|
||||||
|
//
|
||||||
|
// So this test IS a geometric sweep. It scans a string of length n, character by
|
||||||
|
// character, at four doubling sizes, and asserts the cost is linear. If anyone
|
||||||
|
// reintroduces a per-character rescan — in str_char_code, in str_slice, in any
|
||||||
|
// accessor the lexer leans on — the measured curve becomes O(n^2) and this fails.
|
||||||
|
//
|
||||||
|
// The value is in it being ARMED, not in it currently failing. It passes today
|
||||||
|
// because #132 is fixed. That is the correct state for a regression gate.
|
||||||
|
//
|
||||||
|
// Note the deliberate `let c: Int = str_char_code(...)` binding in the scan loop.
|
||||||
|
// Inlining it as `total + str_char_code(s, i)` lowers to el_str_concat() on
|
||||||
|
// integers — the Plus arm of the operator-typing family, still open at the time
|
||||||
|
// of writing. Binding first is the safe form.
|
||||||
|
|
||||||
|
// _mk_string — build a string of length >= n by DOUBLING.
|
||||||
|
//
|
||||||
|
// Deliberately not `s = s + "x"` n times: that is itself quadratic in bytes and
|
||||||
|
// would contaminate the very measurement this test exists to take. Doubling
|
||||||
|
// allocates ~2n total.
|
||||||
|
fn _mk_string(n: Int) -> String {
|
||||||
|
let s: String = "abcdefgh"
|
||||||
|
while str_len(s) < n {
|
||||||
|
let s = s + s
|
||||||
|
}
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
// _scan — walk the string one character at a time, REPS times.
|
||||||
|
//
|
||||||
|
// This is the lexer's access pattern reduced to its essential shape. The
|
||||||
|
// repetitions lift the measurement clear of timer resolution; without them the
|
||||||
|
// smaller sizes land in noise and the classifier correctly reports
|
||||||
|
// INDETERMINATE rather than guessing.
|
||||||
|
fn _scan(s: String, n: Int, reps: Int) -> Int {
|
||||||
|
let total: Int = 0
|
||||||
|
let r: Int = 0
|
||||||
|
while r < reps {
|
||||||
|
let i: Int = 0
|
||||||
|
while i < n {
|
||||||
|
let c: Int = str_char_code(s, i)
|
||||||
|
let total = total + c
|
||||||
|
let i = i + 1
|
||||||
|
}
|
||||||
|
let r = r + 1
|
||||||
|
}
|
||||||
|
return total
|
||||||
|
}
|
||||||
|
|
||||||
|
// _measure_scan — microseconds for a full scan sweep point.
|
||||||
|
fn _measure_scan(n: Int, reps: Int) -> Int {
|
||||||
|
let s: String = _mk_string(n)
|
||||||
|
// WARMUP, discarded. Without it the small-n end of the sweep is dominated
|
||||||
|
// by cold caches and reads as superlinear on genuinely linear work --
|
||||||
|
// measured ratios 3.37 2.92 1.76 1.65 on exactly this workload.
|
||||||
|
let w: Int = _scan(s, n, 2)
|
||||||
|
let wj: Int = el_black_box(w)
|
||||||
|
let t0: Int = el_now_instant()
|
||||||
|
let got: Int = _scan(s, n, reps)
|
||||||
|
let t1: Int = el_now_instant()
|
||||||
|
// Feed the result through the barrier so the scan cannot be elided.
|
||||||
|
let sink: Int = el_black_box(got)
|
||||||
|
if sink == 0 { println("") }
|
||||||
|
return (t1 - t0) / 1000
|
||||||
|
}
|
||||||
|
|
||||||
|
fn _series4(a: Int, b: Int, c: Int, d: Int) -> [Int] {
|
||||||
|
let l: [Int] = native_list_empty()
|
||||||
|
let l = native_list_append(l, a)
|
||||||
|
let l = native_list_append(l, b)
|
||||||
|
let l = native_list_append(l, c)
|
||||||
|
let l = native_list_append(l, d)
|
||||||
|
return l
|
||||||
|
}
|
||||||
|
|
||||||
|
test "character scan is LINEAR in time -- regression gate for el #132" {
|
||||||
|
let reps: Int = 40
|
||||||
|
let t1: Int = _measure_scan(16384, reps)
|
||||||
|
let t2: Int = _measure_scan(32768, reps)
|
||||||
|
let t3: Int = _measure_scan(65536, reps)
|
||||||
|
let t4: Int = _measure_scan(131072, reps)
|
||||||
|
let series: [Int] = _series4(t1, t2, t3, t4)
|
||||||
|
|
||||||
|
let verdict: Int = elb_gate(series, 2, 50)
|
||||||
|
let measured: Int = elb_measured_curve(series, 50)
|
||||||
|
|
||||||
|
// Report the actual numbers regardless of outcome. A gate that fires
|
||||||
|
// without showing its evidence is just an assertion.
|
||||||
|
println(" scan us: " + int_to_str(t1) + " " + int_to_str(t2) + " "
|
||||||
|
+ int_to_str(t3) + " " + int_to_str(t4)
|
||||||
|
+ " -> " + elb_curve_name(measured) + " [" + elb_verdict_name(verdict) + "]")
|
||||||
|
|
||||||
|
// PASS (0) or BETTER (4) are both acceptable. FAIL (1) means someone
|
||||||
|
// reintroduced superlinear per-character cost. REFUSED (3) or
|
||||||
|
// INDETERMINATE (2) mean the measurement is untrustworthy -- which is
|
||||||
|
// also a failure of this test, deliberately: a gate that cannot measure
|
||||||
|
// must not report success.
|
||||||
|
assert verdict == 0 || verdict == 4, "character scan must measure O(n) or better"
|
||||||
|
}
|
||||||
|
|
||||||
|
test "string building by doubling stays linear in allocated bytes" {
|
||||||
|
let b1: Int = el_alloc_bytes()
|
||||||
|
let s1: String = _mk_string(8192)
|
||||||
|
let b2: Int = el_alloc_bytes()
|
||||||
|
let s2: String = _mk_string(16384)
|
||||||
|
let b3: Int = el_alloc_bytes()
|
||||||
|
let s3: String = _mk_string(32768)
|
||||||
|
let b4: Int = el_alloc_bytes()
|
||||||
|
let s4: String = _mk_string(65536)
|
||||||
|
let b5: Int = el_alloc_bytes()
|
||||||
|
|
||||||
|
let series: [Int] = _series4(b2 - b1, b3 - b2, b4 - b3, b5 - b4)
|
||||||
|
let verdict: Int = elb_gate(series, 2, 1000)
|
||||||
|
let measured: Int = elb_measured_curve(series, 1000)
|
||||||
|
println(" bytes: " + int_to_str(b2 - b1) + " " + int_to_str(b3 - b2) + " "
|
||||||
|
+ int_to_str(b4 - b3) + " " + int_to_str(b5 - b4)
|
||||||
|
+ " -> " + elb_curve_name(measured) + " [" + elb_verdict_name(verdict) + "]")
|
||||||
|
|
||||||
|
assert verdict == 0 || verdict == 4, "doubling build must be O(n) in bytes"
|
||||||
|
assert str_len(s4) >= 65536, "final string reached the requested size"
|
||||||
|
}
|
||||||
|
|
||||||
|
// _scan_quadratic — a DELIBERATELY quadratic scan: for each position, rescan
|
||||||
|
// from the start. This is precisely what el #132 did — strlen() from offset 0
|
||||||
|
// on every character access — reproduced here so the gate can be proven to
|
||||||
|
// FIRE, not merely to pass on healthy code. An unproven gate is decoration.
|
||||||
|
fn _scan_quadratic(s: String, n: Int) -> Int {
|
||||||
|
let total: Int = 0
|
||||||
|
let i: Int = 0
|
||||||
|
while i < n {
|
||||||
|
let j: Int = 0
|
||||||
|
while j < i {
|
||||||
|
let c: Int = str_char_code(s, j)
|
||||||
|
let total = total + c
|
||||||
|
let j = j + 1
|
||||||
|
}
|
||||||
|
let i = i + 1
|
||||||
|
}
|
||||||
|
return total
|
||||||
|
}
|
||||||
|
|
||||||
|
fn _measure_quadratic(n: Int) -> Int {
|
||||||
|
let s: String = _mk_string(n)
|
||||||
|
let w: Int = _scan_quadratic(s, 64)
|
||||||
|
let wj: Int = el_black_box(w)
|
||||||
|
let t0: Int = el_now_instant()
|
||||||
|
let got: Int = _scan_quadratic(s, n)
|
||||||
|
let t1: Int = el_now_instant()
|
||||||
|
let sink: Int = el_black_box(got)
|
||||||
|
return (t1 - t0) / 1000
|
||||||
|
}
|
||||||
|
|
||||||
|
test "the gate FIRES on a live quadratic scan -- proves it is armed" {
|
||||||
|
let q1: Int = _measure_quadratic(1024)
|
||||||
|
let q2: Int = _measure_quadratic(2048)
|
||||||
|
let q3: Int = _measure_quadratic(4096)
|
||||||
|
let q4: Int = _measure_quadratic(8192)
|
||||||
|
let series: [Int] = _series4(q1, q2, q3, q4)
|
||||||
|
|
||||||
|
let verdict: Int = elb_gate(series, 2, 50)
|
||||||
|
let measured: Int = elb_measured_curve(series, 50)
|
||||||
|
println(" quad us: " + int_to_str(q1) + " " + int_to_str(q2) + " "
|
||||||
|
+ int_to_str(q3) + " " + int_to_str(q4)
|
||||||
|
+ " -> " + elb_curve_name(measured) + " [" + elb_verdict_name(verdict) + "]")
|
||||||
|
|
||||||
|
assert measured == 4, "a rescan-from-zero workload must classify O(n^2)"
|
||||||
|
assert verdict == 1, "declared O(n) against measured O(n^2) must FAIL the gate"
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user