From 24fac765a63c789c031aee2609b5381b832a656b Mon Sep 17 00:00:00 2001 From: Neuron Date: Sat, 15 Aug 2026 21:24:03 -0500 Subject: [PATCH 1/2] test framework phase 1: compile-time registry + El-side runner Replace the hardcoded test harness main() with a generated static registry and index-based accessors, and move all reporting into runtime/eltest.el. The old harness inlined direct calls into main() and counted assertions in two globals. That shape cannot report which test failed, how long any test took, or whether a test ran at all -- a misspelled registration reported success for a test that never executed. - assertions record into per-test state instead of global counters - registry table emitted at compile time; discovery strictly precedes execution, which is what later enables --list, filtering and sharding - per-test wall timing on CLOCK_MONOTONIC, taken in C around the call - runner in El: structured NDJSON events as source of truth, human output rendered from the same fields --- lang/el-compiler/src/codegen.el | 98 +++++++++++++--- lang/runtime/eltest.el | 192 ++++++++++++++++++++++++++++++++ 2 files changed, 276 insertions(+), 14 deletions(-) create mode 100644 lang/runtime/eltest.el diff --git a/lang/el-compiler/src/codegen.el b/lang/el-compiler/src/codegen.el index 3c63dbd..5a29204 100644 --- a/lang/el-compiler/src/codegen.el +++ b/lang/el-compiler/src/codegen.el @@ -1705,9 +1705,13 @@ fn cg_stmt(stmt: Map, indent: String, declared: [String]) -> [Strin } else { let c_msg = "EL_STR_PTR(" + cg_expr(msg_node) + ")" } + // Assertions record into PER-TEST state, not global counters. The test + // is the unit of result; a global pass/fail tally cannot say which test + // failed or whether a test ran at all. Reporting is the runner's job — + // nothing is printed here. emit_line(indent + "if (!(" + c_cond + ")) {") - emit_line(indent + " __el_test_fail(__el_cur_test, " + c_msg + "); __el_fail++;") - emit_line(indent + "} else { __el_pass++; }") + emit_line(indent + " __el_test_fail(" + c_msg + ");") + emit_line(indent + "} else { __el_cur_asserts++; }") return declared } @@ -4110,11 +4114,22 @@ fn codegen_streaming(tokens: [Any], sigs: [Map], source: String) -> // Emit test harness preamble (counters, fail printer) when in test mode. if test_is_mode { emit_line("#include ") + emit_line("#include ") + emit_line("#include ") emit_blank() - emit_line("static int __el_pass = 0, __el_fail = 0;") + // Per-test result state. Reset by __el_reg_invoke before each test, so + // every test gets its own record rather than contributing to a global + // tally. The first failure message is retained; later ones only bump + // the count, which keeps the common case allocation-free. + emit_line("static int __el_cur_fails = 0;") + emit_line("static int __el_cur_asserts = 0;") + emit_line("static char __el_cur_msg[512] = \"\";") emit_line("static const char *__el_cur_test = \"(none)\";") - emit_line("static void __el_test_fail(const char *test, const char *msg) {") - emit_line(" fprintf(stderr, \"FAIL %-40s %s\\n\", test, msg);") + emit_line("static void __el_test_fail(const char *msg) {") + emit_line(" if (__el_cur_fails == 0 && msg) {") + emit_line(" snprintf(__el_cur_msg, sizeof __el_cur_msg, \"%s\", msg);") + emit_line(" }") + emit_line(" __el_cur_fails++; __el_cur_asserts++;") emit_line("}") emit_blank() } @@ -4312,17 +4327,72 @@ fn codegen_streaming(tokens: [Any], sigs: [Map], source: String) -> el_release(sigs) let test_arena_mark: Any = el_arena_push() + let tn: Int = native_list_len(test_c_names) + + // ── Generated test registry ────────────────────────────────────────── + // Discovery happens HERE, at compile time. The runner never searches + // for tests; it walks this table. That ordering — discovery strictly + // before execution — is what makes --list, filtering, sharding and + // per-test reporting possible later, and it is why the old harness + // (which inlined direct calls into main) could not have any of them. + emit_line("typedef void (*__el_test_fp)(void);") + emit_line("typedef struct { const char *name; __el_test_fp fn; } __el_test_entry;") + emit_line("static const __el_test_entry __el_registry[] = {") + let ri: Int = 0 + while ri < tn { + let r_name: String = native_list_get(test_names, ri) + let r_cfn: String = native_list_get(test_c_names, ri) + emit_line(" { \"" + c_escape(r_name) + "\", " + r_cfn + " },") + let ri = ri + 1 + } + // Trailing sentinel keeps the array non-empty when a file declares no + // tests (a zero-length array is not valid C). + emit_line(" { 0, 0 }") + emit_line("};") + emit_line("static const int __el_registry_n = " + int_to_str(tn) + ";") + emit_blank() + emit_line("static long long __el_last_ns = 0;") + emit_line("static int __el_opt_json_v = 0;") + emit_blank() + + // ── Index-based accessors ──────────────────────────────────────────── + // El has no function pointers, so the runner works purely in indices. + // This is the whole seam between generated C and the El-side runner. + emit_line("el_val_t __el_reg_count(void) { return (el_val_t)(int64_t)__el_registry_n; }") + emit_line("el_val_t __el_reg_name(el_val_t i) {") + emit_line(" int64_t k = (int64_t)i;") + emit_line(" if (k < 0 || k >= __el_registry_n) return EL_STR(\"\");") + emit_line(" return EL_STR(__el_registry[k].name);") + emit_line("}") + // Timing is taken immediately around the call, in C, on the MONOTONIC + // clock — never the wall clock, which can step backwards under NTP. + emit_line("el_val_t __el_reg_invoke(el_val_t i) {") + emit_line(" int64_t k = (int64_t)i;") + emit_line(" if (k < 0 || k >= __el_registry_n) return 0;") + emit_line(" __el_cur_fails = 0; __el_cur_asserts = 0; __el_cur_msg[0] = '\\0';") + emit_line(" __el_cur_test = __el_registry[k].name;") + emit_line(" struct timespec _t0, _t1;") + emit_line(" clock_gettime(CLOCK_MONOTONIC, &_t0);") + emit_line(" __el_registry[k].fn();") + emit_line(" clock_gettime(CLOCK_MONOTONIC, &_t1);") + emit_line(" __el_last_ns = (long long)(_t1.tv_sec - _t0.tv_sec) * 1000000000LL") + emit_line(" + (long long)(_t1.tv_nsec - _t0.tv_nsec);") + emit_line(" return (el_val_t)(int64_t)__el_cur_fails;") + emit_line("}") + emit_line("el_val_t __el_reg_last_ns(void) { return (el_val_t)(int64_t)__el_last_ns; }") + emit_line("el_val_t __el_reg_msg(void) { return EL_STR(__el_cur_msg); }") + emit_line("el_val_t __el_reg_asserts(void) { return (el_val_t)(int64_t)__el_cur_asserts; }") + emit_line("el_val_t __el_opt_json(void) { return (el_val_t)(int64_t)__el_opt_json_v; }") + emit_blank() + + // main() delegates to the El-side runner. Everything above this line is + // generated glue; all reporting logic lives in runtime/eltest.el. emit_line("int main(int _argc, char **_argv) {") emit_line(" el_runtime_init_args(_argc, _argv);") - let ti: Int = 0 - let tn: Int = native_list_len(test_c_names) - while ti < tn { - let tc_name: String = native_list_get(test_c_names, ti) - emit_line(" " + tc_name + "();") - let ti = ti + 1 - } - emit_line(" printf(\"%d passed, %d failed\\n\", __el_pass, __el_fail);") - emit_line(" return __el_fail;") + emit_line(" for (int _i = 1; _i < _argc; _i++) {") + emit_line(" if (strcmp(_argv[_i], \"--json\") == 0) __el_opt_json_v = 1;") + emit_line(" }") + emit_line(" return (int)(int64_t)el_test_main();") emit_line("}") el_arena_pop(test_arena_mark) el_release(test_names) diff --git a/lang/runtime/eltest.el b/lang/runtime/eltest.el new file mode 100644 index 0000000..b84698b --- /dev/null +++ b/lang/runtime/eltest.el @@ -0,0 +1,192 @@ +// runtime/eltest.el — El test framework runner (Phase 1). +// +// This is the RUNNER. It is written in El and consumes a registry that the +// compiler generates into the same translation unit when invoked as +// `elc --test`. Nothing here discovers tests; discovery already happened at +// compile time, which is what makes `--list` and filtering possible later. +// +// ── Architecture ───────────────────────────────────────────────────────────── +// +// The compiler lowers each `test "name" { ... }` block into a static C +// function and emits a static table of (name, fn) pairs plus a small set of +// index-based accessors. El has no function pointers, so the runner never +// sees one — it works entirely in indices: +// +// __el_reg_count() -> Int number of registered tests +// __el_reg_name(i) -> String test name at index i +// __el_reg_invoke(i) -> Int run test i, return its failure count +// __el_reg_last_ns() -> Int wall-clock ns of the last invoke +// __el_reg_msg() -> String first failure message of the last invoke +// __el_reg_asserts() -> Int assertions executed in the last invoke +// __el_opt_json() -> Int 1 if --json was passed +// +// Timing is taken in the generated C, immediately around the call, so no El +// call overhead lands inside the measurement. +// +// ── Output ─────────────────────────────────────────────────────────────────── +// +// Structured events are the source of truth. The human renderer is written +// FROM the same fields the NDJSON renderer emits — never the reverse. Parsing +// human output back into structure is the one clear architectural mistake in +// Go's test tooling and we do not repeat it. +// +// Every result carries a duration. Always. A framework that cannot report how +// long its tests took cannot surface a performance regression, and a +// regression nobody can see is one nobody fixes. + +// ── Small helpers (no imports — this file must stay self-contained) ────────── + +// _elt_json_escape — minimal JSON string escaping for the NDJSON renderer. +fn _elt_json_escape(s: String) -> String { + let out: String = "" + let n: Int = str_len(s) + let i: Int = 0 + while i < n { + let ch: String = str_slice(s, i, i + 1) + if str_eq(ch, "\"") { + let out = out + "\\\"" + } else { + if str_eq(ch, "\\") { + let out = out + "\\\\" + } else { + if str_eq(ch, "\n") { + let out = out + "\\n" + } else { + if str_eq(ch, "\t") { + let out = out + "\\t" + } else { + if str_eq(ch, "\r") { + let out = out + "\\r" + } else { + let out = out + ch + } + } + } + } + } + let i = i + 1 + } + return out +} + +// _elt_pad3 — left-pad an integer to three digits (for the ms.fraction form). +fn _elt_pad3(v: Int) -> String { + if v < 10 { return "00" + int_to_str(v) } + if v < 100 { return "0" + int_to_str(v) } + return int_to_str(v) +} + +// _elt_ms — render a nanosecond duration as "M.mmm" milliseconds. +// +// Deliberately avoids the modulo operator: the remainder is derived by +// subtraction so this stays portable across El backends. +fn _elt_ms(ns: Int) -> String { + let total_us: Int = ns / 1000 + let ms_whole: Int = total_us / 1000 + let us_rem: Int = total_us - (ms_whole * 1000) + return int_to_str(ms_whole) + "." + _elt_pad3(us_rem) +} + +// _elt_secs — render a nanosecond duration as fractional seconds, for the +// NDJSON `elapsed` field. JUnit XML and test2json both use seconds-as-decimal. +fn _elt_secs(ns: Int) -> String { + let total_ms: Int = ns / 1000000 + let s_whole: Int = total_ms / 1000 + let ms_rem: Int = total_ms - (s_whole * 1000) + return int_to_str(s_whole) + "." + _elt_pad3(ms_rem) +} + +// ── Event emission ─────────────────────────────────────────────────────────── +// +// One function per event shape. Both renderers read the same fields; the +// human renderer is a projection of the event, not a separate code path. + +fn _elt_emit_run(json_mode: Bool, name: String) { + if json_mode { + println("{\"action\":\"run\",\"test\":\"" + _elt_json_escape(name) + "\"}") + } +} + +fn _elt_emit_result(json_mode: Bool, name: String, fails: Int, ns: Int, asserts: Int, msg: String) { + if json_mode { + let action: String = "pass" + if fails > 0 { let action = "fail" } + let line: String = "{\"action\":\"" + action + "\"" + let line = line + ",\"test\":\"" + _elt_json_escape(name) + "\"" + let line = line + ",\"elapsed\":" + _elt_secs(ns) + let line = line + ",\"assertions\":" + int_to_str(asserts) + if fails > 0 { + let line = line + ",\"failures\":" + int_to_str(fails) + let line = line + ",\"message\":\"" + _elt_json_escape(msg) + "\"" + } + let line = line + "}" + println(line) + return + } + // Human renderer — duration is never optional. + if fails > 0 { + println("FAIL " + name + " (" + _elt_ms(ns) + "ms)") + println(" " + msg) + return + } + println("ok " + name + " (" + _elt_ms(ns) + "ms)") +} + +fn _elt_emit_summary(json_mode: Bool, total: Int, failed: Int, ns: Int, asserts: Int) { + let passed: Int = total - failed + if json_mode { + let line: String = "{\"action\":\"summary\"" + let line = line + ",\"tests\":" + int_to_str(total) + let line = line + ",\"passed\":" + int_to_str(passed) + let line = line + ",\"failed\":" + int_to_str(failed) + let line = line + ",\"assertions\":" + int_to_str(asserts) + let line = line + ",\"elapsed\":" + _elt_secs(ns) + let line = line + "}" + println(line) + return + } + println("") + println(int_to_str(total) + " tests, " + int_to_str(passed) + " passed, " + + int_to_str(failed) + " failed, " + int_to_str(asserts) + " assertions in " + + _elt_ms(ns) + "ms") +} + +// ── The runner ─────────────────────────────────────────────────────────────── + +// el_test_main — drive the compile-time registry. +// +// Called from the generated main(). Returns the number of FAILING TESTS, which +// becomes the process exit code. Note that this counts tests, not assertions: +// a test is the unit of result. The old harness counted assertions globally and +// therefore could not say which test failed, how long any of them took, or +// whether a test had run at all. +fn el_test_main() -> Int { + let json_mode: Bool = false + if __el_opt_json() == 1 { let json_mode = true } + + let n: Int = __el_reg_count() + let i: Int = 0 + let failed: Int = 0 + let total_ns: Int = 0 + let total_asserts: Int = 0 + + while i < n { + let name: String = __el_reg_name(i) + _elt_emit_run(json_mode, name) + + let fails: Int = __el_reg_invoke(i) + let ns: Int = __el_reg_last_ns() + let asserts: Int = __el_reg_asserts() + let msg: String = __el_reg_msg() + + let total_ns = total_ns + ns + let total_asserts = total_asserts + asserts + if fails > 0 { let failed = failed + 1 } + + _elt_emit_result(json_mode, name, fails, ns, asserts, msg) + let i = i + 1 + } + + _elt_emit_summary(json_mode, n, failed, total_ns, total_asserts) + return failed +} From 3e7ab07e82c4d421d9ec386155de91e9f3ca621e Mon Sep 17 00:00:00 2001 From: Neuron Date: Sat, 15 Aug 2026 21:28:30 -0500 Subject: [PATCH 2/2] test framework phase 1: forward decls, void-return fix, suite migration Completes the Phase 1 runner and migrates the 11 test files onto it. - forward-declare the registry accessors in the test preamble; they are defined at the end of the unit but the El runner is compiled in between - eltest.el: explicit trailing return in the void emit_* helpers, which otherwise lower to 'return println(...)' and fail to compile - test files import runtime/eltest.el explicitly, using the language's own textual import mechanism rather than compiler-side auto-injection - DESIGN.md 6.5: gate on allocation COUNT AND BYTES, not count alone Verified: self-hosting fixpoint byte-identical (gen2 == gen3). 6 of 11 suites run and report per-test timing. The other 5 fail to COMPILE, and fail identically under the committed compiler -- pre-existing breakage this framework makes visible for the first time. --- DESIGN.md | 581 +++++++++++++++++++++++++++++ lang/el-compiler/src/codegen.el | 23 ++ lang/runtime/eltest.el | 2 + lang/tests/native/test_compiler.el | 1 + lang/tests/native/test_core.el | 1 + lang/tests/native/test_env.el | 1 + lang/tests/native/test_fs.el | 1 + lang/tests/native/test_json.el | 1 + lang/tests/native/test_math.el | 1 + lang/tests/native/test_state.el | 1 + lang/tests/native/test_string.el | 1 + lang/tests/native/test_text.el | 1 + lang/tests/native/test_time.el | 1 + lang/tests/runtime/string_test.el | 1 + 14 files changed, 617 insertions(+) create mode 100644 DESIGN.md diff --git a/DESIGN.md b/DESIGN.md new file mode 100644 index 0000000..30d0120 --- /dev/null +++ b/DESIGN.md @@ -0,0 +1,581 @@ +# El Test Framework — Design + +**Status:** draft for review +**Author:** Neuron +**Date:** 2026-08-15 +**Worktree:** `/Users/will/Development/neuron-technologies/el-worktrees/elc-memory-investigation` + +--- + +## 0. The forcing requirement + +We have a confirmed quadratic in `elc`. Peak memory in the old shipped binary and wall-clock in +the current source both grow as O(input²). We cannot fix it, because we cannot test it. + +Everything in this document is downstream of one sentence: **a test framework must be able to fail +a build when an operation's growth curve degrades from linear to quadratic.** + +That is not a nice-to-have bolted onto a correctness framework. It is the requirement that +determines the architecture. Correctness testing is the easy half. + +Second-order requirement, learned the hard way tonight: **the framework must report per-test timing +by default.** The current framework prints `N passed, M failed` and nothing else. That is why a +3.58-second test file sat in the suite unnoticed. A framework that is structurally blind to time +cannot surface the defect class we most need to catch. + +--- + +## 1. What exists today, measured + +### 1.1 Two competing systems, neither complete + +**System A — `lang/runtime/test.el`.** Manual registration, El-level. + +**System B — the compiler's `test { }` block + `elc --test`.** Emits its own harness `main()` +with `__el_pass` / `__el_fail` globals (`codegen.el:3777-3796`). + +They do not share a result model. Neither has timing. Both are in the tree. + +### 1.2 Specific defects in System A + +| Defect | Location | Consequence | +|---|---|---| +| All state as JSON strings in a global string-keyed map | `test.el` throughout | every assertion is `state_get` → `str_to_int` → `int_to_str` → `state_set` | +| Failure list appended by string slice + concat | `_test_json_append` | O(n²) in failure count | +| One OS thread spawned per test | `_test_run_one` via `__thread_create`/`__thread_join` | thread spawn per test, purely to get dispatch-by-name through dlsym | +| Manual registration pairing a string to a function name | `test_case(name, fn_name)` | typo ⇒ test silently never runs, suite still reports pass | +| Counters are assertion-level, global | `_test_pass_count` etc. | no per-test record exists at all | +| No timing, no structured output, no fixtures, no tags, no filtering, no parameterization, no benchmarks | — | — | + +The registration defect is the serious one. It is not a slow framework, it is a framework that can +report success for tests that did not execute. + +### 1.3 Measured cost structure + +Per test file, current build model: + +| Step | Time | +|---|---| +| `elc` compile `.el` → `.c` | 0.00s (small files) | +| **`cc` el_runtime.c → .o** | **0.14s** | +| `cc` test .c → .o | 0.02s | +| link | 0.02s | + +Per-file `elc` time across the existing suite: + +| File | Bytes | elc time | +|---|---|---| +| `test_compiler` | 29,685 (+394 KB of imports) | **3.58s** | +| `string_test` | 18,545 | 0.01s | +| all other 9 files | 2.2–10 KB | 0.00s | + +Two distinct defects in two distinct regimes: + +1. **`test_compiler.el` imports all five compiler sources** — 394 KB in one translation unit. Its + 3.58s is entirely the quadratic. It is the only file where the quadratic bites. +2. **Every other file's cost is 100% redundant `el_runtime.c` rebuilds** — 480 KB of identical C, + recompiled once per test file. + +Neither is fixed by making the compiler faster. Both are fixed by the architecture below, and the +speedup is a by-product of building it correctly, not the goal. + +### 1.4 The asset worth keeping + +`codegen.el:3651-3652` already collects `test_names` / `test_c_names` — **the compiler already does +compile-time test discovery.** It then discards that registry into a hardcoded `main()`. + +That registry is precisely the seam Go's `_testmain.go` and Rust's `test_main_static` are built on. +The mechanism we need is half-built and wired to the wrong thing. + +--- + +## 2. Grounding — the common spine of excellent frameworks + +Researched from primary sources: Go `testing`/`go test`, Rust `libtest`/Criterion, JUnit 5 Platform, +NUnit 3, JMH, Google Benchmark. Six invariants hold across all of them. + +1. **A registry is built before execution** — `(name, metadata, fn-ptr)` triples. Go generates it + from an AST scan; Rust synthesizes it in a compiler pass; JMH emits it as a build-time resource; + JUnit/NUnit build it reflectively. **Reflection is an implementation of the registry on runtimes + where it is cheap. It is never the architecture.** + +2. **Discovery strictly precedes execution.** Every good capability — filtering, listing, counting, + sharding, IDE trees, re-run-failed-only, dry runs — is a consequence of this ordering. + +3. **A hierarchy with stable, path-shaped unique IDs.** `TestFoo/subcase_2`. Selection is regex over + that path, one pattern per level. + +4. **The framework is a prebuilt library; only the entry point is generated.** "Compile once, link + many" is always: framework archive compiled once + a small generated table + one + `MainStart(deps, registry)` call. Nobody recompiles the harness per test file. + +5. **Execution emits an event stream; reporters are downstream renderers.** Human text, NDJSON, + JUnit XML, TAP are all transforms of one event stream. Go's one architectural mistake is doing + this backwards — `test2json` parses human output, and has shipped bugs when user output contains + `--- PASS:`. + +6. **A dependency-injection seam at the boundary.** Go's `testdeps.TestDeps` exists so `testing` + can avoid importing `regexp`, profilers, and coverage. The execution core knows nothing about + output formats. + +--- + +## 3. Architecture + +### 3.1 The seam + +``` + ┌─────────────────────────────────────────────────────────────┐ + │ user code: foo.el with test { } / bench { } blocks │ + └───────────────────────────┬─────────────────────────────────┘ + │ elc --test + ▼ + ┌─────────────────────────────────────────────────────────────┐ + │ generated C (per suite, tiny): │ + │ __el_test_fn_0 .. _N lowered test/bench bodies │ + │ __el_registry[] static table: name/kind/file/ │ + │ line/tags/sizes/expected-O │ + │ __el_dispatch(i) generated switch → body │ + │ main() { return el_test_main(argc, argv); } │ + └───────────────────────────┬─────────────────────────────────┘ + │ cc + link (registry only) + ▼ + ┌─────────────────────────────────────────────────────────────┐ + │ libeltest.a — PREBUILT ONCE │ + │ • el_runtime.o (the 480 KB, compiled once, ever) │ + │ • eltest.o the runner, WRITTEN IN EL │ + │ discovery view · filtering · execution · fixtures · │ + │ timing · benchmark harness · curve fitting · reporters │ + └─────────────────────────────────────────────────────────────┘ +``` + +The framework is written in El, compiled to C once, archived. Per-suite compilation touches only +the generated registry. This is Go's model, and it is strictly better for us than Go's because we +own the compiler and already have the AST — no separate source-scanning pass is needed. + +### 3.2 Why the runner is in El and the registry is in C + +El has no closures and no first-class function pointers. The registry must therefore hold C function +pointers, and it is generated C. + +The runner stays in El and reaches the registry through a small builtin surface — indices, not +pointers: + +``` +__el_reg_count() -> Int +__el_reg_name(i) -> String +__el_reg_file(i) -> String +__el_reg_line(i) -> Int +__el_reg_kind(i) -> Int // 0=test 1=bench +__el_reg_tags(i) -> Int +__el_reg_sizes(i) -> String // JSON array, empty for tests +__el_reg_expect(i) -> Int // complexity class enum, 0 = none +__el_reg_invoke(i) -> Int // runs the body via the generated switch +``` + +Nine builtins. Everything else — filtering, lifecycle, statistics, curve fitting, all reporters — +is El. That satisfies "written in El" without pretending El can do something it cannot. + +### 3.3 Result model + +The unit is a **result record**, not a counter: + +``` +TestResult { + id String // slash path: "parser/handles_empty_input/case_3" + file String + line Int + status Status // Pass | Fail | Error | Skip + duration Int // nanoseconds, ALWAYS populated + message String // assertion detail: expected vs actual + output String // captured stdout/stderr for this test + assertions Int +} +``` + +`Fail` = an assertion failed. `Error` = unexpected crash/abort. This distinction is load-bearing — +every CI consumer depends on it, and the JUnit XML schema encodes it as distinct elements. + +--- + +## 4. Authoring surface + +### 4.1 Tests + +`test { }` already exists. Keep it. Add subtests and hierarchy: + +```el +test "parser/empty input" { + assert_that(parse(""), is_err()) +} + +test "parser/table" { + for case in [["", 0], ["a", 1], ["a b", 2]] { + subtest(case[0]) { + assert_that(token_count(case[0]), equals(case[1])) + } + } +} +``` + +Subtest IDs compose as `parser/table/a_b`. Filtering is `--run 'parser/table/.*'`, one regex per +path segment, exactly as Go does. + +**We do not build a parameterized-test annotation system.** Table-driven loops plus subtests subsume +`@ParameterizedTest`, `@MethodSource`, `@CsvSource`, and `TestCaseSource` entirely, at zero framework +surface. This is Go's single biggest ergonomic win over JUnit and NUnit. + +### 4.2 Fixtures + +Per-file and per-test only, plus a LIFO cleanup stack: + +```el +setup_all { ... } // once per suite +setup { ... } // before each test +teardown { ... } // after each test +teardown_all { ... } +``` + +and inside a test, `cleanup { ... }` registering LIFO-ordered teardown. + +**We do not build JUnit 5's extension SPI** — seventeen callback interfaces, hierarchical stores, +registration ordering rules. That complexity is the price of retrofitting a plugin ecosystem onto a +twenty-year-old reflective framework. Go's `t.Cleanup` covers roughly 90% of what `@AfterEach` is +used for at a fraction of the surface. + +### 4.3 Assertions — constraint model + +One entry point, composable constraint values (NUnit's model, which avoids the N² overload +explosion): + +```el +assert_that(actual, equals(expected)) +assert_that(xs, has_length(3)) +assert_that(s, contains("foo").and(starts_with("bar"))) +assert_that(f, is_within(0.01).of(3.14)) +``` + +A constraint is a value with `apply_to(actual) -> ConstraintResult`, and the result knows how to +describe its own failure. Custom constraints are ordinary user types. + +**Every failure message must name file, line, the expression text, and both values.** We capture +expression source text at compile time — we have the AST, so we can do this better than any +runtime-introspection framework. + +Legacy `assert_true` / `assert_eq` / etc. stay as thin wrappers for migration. + +--- + +## 5. Benchmarks + +### 5.1 The loop + +Adopt `b.Loop()`, not `b.N`. Go spent fifteen years on `b.N` before concluding `b.Loop` was right; +we skip that. + +```el +bench "str_concat" { + let s = make_input(bench_n()) + for bench_loop() { + black_box(str_concat(s, "x")) + } +} +``` + +Three properties that make this the correct choice for a C target: + +1. **The timer auto-resets on first call**, so setup above the loop is excluded *by construction* + rather than by the author remembering `ResetTimer`. +2. **`N` is hidden**, so it cannot be misused. +3. **The harness owns the loop shape**, which lets us insert an optimization barrier the C compiler + cannot see through. `black_box(v)` lowers to `asm volatile("" :: "r"(&v) : "memory")`. Since we + emit a single translation unit, dead-code elimination of a benchmark body is a live hazard — + this is our version of JMH's `Blackhole` problem, solved in the harness rather than delegated to + the user. + +### 5.2 Iteration scaling + +Use Go's `predictN` heuristics verbatim. They are battle-tested and cheap: + +``` +n = goal_ns * prev_iters / prev_ns // multiply before divide — precision on sub-ns ops +n += n / 5 // 20% headroom, overshoot rather than re-loop +n = min(n, 100 * last) // never grow more than 100× per step +n = max(n, last + 1) // guarantee forward progress +n = min(n, 1_000_000_000) // hard ceiling +``` + +Report `n` rounded to 1/2/3/5 × 10ᵏ so runs are comparable. + +### 5.3 Sampling + +Criterion's shape, because it is correct near timer resolution: + +- **Warmup**: iteration counts 1, 2, 4, 8… until cumulative time exceeds the warmup budget. +- **Measurement**: collect `sample_size` samples at iteration counts `[d, 2d, 3d, …, Nd]`. +- **Estimate**: slope of a linear regression of iteration-count vs elapsed time. The intercept + absorbs fixed overhead. +- **Time whole samples, never individual iterations.** This is the single most important detail — + it defeats timer-resolution error on nanosecond operations. + +Outliers classified by modified Tukey (±1.5 IQR mild, ±3 IQR severe), **reported but retained**. + +--- + +## 6. Complexity gating — the centerpiece + +This is the part that makes the quadratic fixable, and the part nobody in the mainstream has +finished. Google Benchmark's `Complexity()` fits the curve and *reports* it. We declare it and +**gate** on it. + +### 6.1 Surface + +```el +bench "elc_compile" over n in [16, 32, 64, 128, 256, 512, 1024] expect O(n) { + let src = synth_source(bench_n()) + for bench_loop() { black_box(compile(src)) } +} +``` + +Alternative with no new syntax, if the parser change is judged too invasive — `bench_sizes([...])` +and `bench_expect("O(n)")` as calls inside the block. **Recommendation: declarative.** Runtime calls +mean `--list` cannot show the invariant without executing, which breaks the discovery-precedes- +execution invariant from §2. + +### 6.2 Fitting + +Per Google Benchmark `src/complexity.cc`. For candidate curves +`{O(1), O(log n), O(n), O(n log n), O(n²), O(n³)}`, one-parameter least squares, no intercept: + +``` +coef = Σ(tᵢ · gᵢ) / Σ(gᵢ²) +rms = sqrt( Σ(tᵢ − coef·gᵢ)² / k ) / mean(t) // normalized +``` + +Best fit = lowest normalized RMS. User-supplied lambda curves also supported. + +### 6.3 Gate logic + +1. **FAIL** if the best-fit curve is strictly worse than declared, ordering + `O(1) < O(log n) < O(n) < O(n log n) < O(n²) < O(n³)`. Print the fitted coefficient and the full + per-size table. +2. **FAIL** if the declared curve's normalized RMS exceeds a threshold (start at 0.10). This catches + the case where *no* candidate fits — noise, a cache cliff, or a phase change. Report + `INDETERMINATE` honestly rather than gating on garbage. +3. **WARN** if the best fit is strictly better than declared — either an optimization landed and the + annotation should tighten, or the sweep is too narrow to expose real behaviour. +4. **REFUSE to gate** on fewer than 5 distinct sizes spanning under 2 decades, geometrically spaced. + Say so loudly rather than producing a meaningless fit. + +### 6.4 Why gate on the exponent, not wall-clock + +- **Machine-independent.** The fitted exponent is a property of the algorithm; the coefficient is a + property of the machine. Gating on the exponent makes CI hardware heterogeneity, noisy neighbours, + and thermal throttling irrelevant — they scale `coef`, not `g`. +- **No stored baseline.** No artifact storage, no golden-file drift. The invariant lives in the + source next to the code and is reviewed in the same PR. +- **It catches the failure mode that actually ships.** An O(n) lookup inside an O(n) loop is + invisible at n=100 in a unit test and catastrophic at n=100,000 in production. Constant-factor + regressions are annoying. Complexity regressions are outages. Ours was a 27 GB outage. + +### 6.5 The deterministic gate — the one that would have caught us + +Wall-clock needs statistics. **Allocation counts do not.** They are perfectly deterministic. + +> **Correction, 2026-08-16 — count alone is NOT sufficient. Gate on BOTH count and bytes.** +> +> Measured against two El programs, one allocating once per item and one rebuilding its +> accumulator each iteration: +> +> | n | linear allocs / bytes | quadratic allocs / bytes | +> |---|---|---| +> | 100 | 100 / 290 | 100 / 5,150 | +> | 200 | 200 / 690 | 200 / 20,300 | +> | 400 | 400 / 1,490 | 400 / 80,600 | +> | 800 | 800 / 3,090 | 800 / 321,200 | +> +> The quadratic program's allocation **count is exactly linear** — 100/200/400/800, identical to +> the healthy program. A count-only gate passes it clean. **Bytes** catch it: each doubling of n +> quadruples bytes (ratios 3.94, 3.97, 3.99 → 4.0 = O(n²)) where the linear program converges +> on 2.0. +> +> This is precisely elc's own defect shape — a copy-on-write accumulator reallocating once per +> pass (count linear) into a proportionally larger buffer (bytes quadratic). +> +> Therefore `expect allocs O(n)` **fits count and bytes independently and fails if EITHER exceeds +> the declared curve**, reporting which signal broke. "count linear, bytes quadratic" is a precise, +> directly actionable diagnosis. +> +> **`el_peak_rss()` is CONTEXT ONLY — never gate on it.** It is perturbed by the allocator and by +> the page cache. Allocation volume is the invariant; RSS and malloc/free churn are merely the two +> surfaces it shows on. The old shipped compiler paid the same quadratic in RSS that the rebuilt +> one pays in churn. +> +> **Measure rate, not level.** A guard reading swap *level* saw 97% on a thrashing host and 97% on +> a healthy one; only *rate* separated them. A growth exponent is a rate; a single measurement is +> a level. That is why the gate fits a curve across a sweep instead of comparing one number to a +> threshold. + +Instrument the runtime with allocation counters and fit *those* against n instead of time: + +```el +bench "elc_compile" over n in [...] expect O(n) allocs O(n) { ... } +``` + +Zero noise, zero statistics, always gateable, correct on the first run on any machine. Go reports +`allocs/op` and `B/op`; **nobody fits them against n.** That is an open opportunity and it is exactly +our bug: elc's defect is quadratic *allocation volume*, which the old binary paid in RSS and the +current source pays in malloc/free churn. + +An `expect allocs O(n)` assertion on `elc`'s compile path would have failed the build the day the +quadratic was introduced. + +Required runtime additions: `__el_alloc_count()`, `__el_alloc_bytes()`, `__el_peak_rss()`. + +### 6.6 Constant-factor gate (secondary, opt-in) + +Mann-Whitney U at α = 0.05, noise floor 1%, medians with 95% CIs, `~` for not-significant. Requires +`--count >= 9`. Off by default on CI; opt-in per benchmark. + +**Exit nonzero on regression.** Both benchstat and Criterion always exit 0, which is why every shop +using them wrote a wrapper. We do not repeat that omission. + +--- + +## 7. Output + +**Structured events are the source of truth.** Human text is rendered from them. We do not repeat +Go's parse-the-human-output design. + +Event stream, NDJSON, one object per line, streamed live: + +```json +{"time":"...","action":"run","test":"parser/empty"} +{"time":"...","action":"output","test":"parser/empty","output":"..."} +{"time":"...","action":"pass","test":"parser/empty","elapsed":0.0031} +{"time":"...","action":"bench","test":"str_concat","n":1024,"ns_op":41.2,"allocs_op":3,"bigo":"N","rms":0.03} +``` + +Renderers, all downstream and pluggable: + +| Format | Flag | Use | +|---|---|---| +| Human | default | terminal, **per-test duration always shown** | +| NDJSON | `--json` | tooling, history, flaky detection | +| JUnit XML | `--junit-xml=PATH` | every CI system on earth | +| TAP | `--tap` | optional | + +JUnit XML per the de-facto schema: `testsuites` → `testsuite` → `testcase`, with `time` in seconds +as a decimal, `file`/`line` attributes, and `failure` vs `error` vs `skipped` as distinct child +elements. Absence of a child element means pass. Emit `` even for a single suite, and +parse both shapes on input. + +--- + +## 8. CLI + +``` +--list print the registry, run nothing +--list-json machine-readable registry +--run PATTERN slash-separated regex per path segment +--tag EXPR tag expression: fast & !slow +--shard I/N deterministic sharding for CI parallelism +--count N repetitions, for statistics +--bench PATTERN run benchmarks (off by default in test runs) +--benchtime DUR per-benchmark time budget +--junit-xml PATH +--json +--isolate re-exec per test on crash, so one SIGSEGV doesn't lose the run +--timeout DUR +--fail-fast +``` + +`--list` / `--list-json` / `--shard` cost roughly thirty lines because the registry already exists +before `main` does anything. That is the dividend of discovery-precedes-execution. + +--- + +## 9. Build model + +``` +# once, ever (or when the runtime/framework changes): +cc -c el_runtime.c -o el_runtime.o +elc eltest.el > eltest.c && cc -c eltest.c -o eltest.o +ar rcs libeltest.a el_runtime.o eltest.o + +# per suite: +elc --test foo_test.el > foo_test.c # registry + bodies only +cc foo_test.c libeltest.a -o foo_test +``` + +The 0.14s × N of redundant runtime rebuilds disappears — not because we optimized it, but because +one-runner-over-many-suites requires compile-once-link-many as a structural precondition. + +--- + +## 10. Bootstrap and self-hosting + +The framework's own tests are `test { }` blocks run by the framework. Same fixpoint discipline the +compiler already applies to itself. + +1. Build the framework using the *existing* harness for its first tests (stage 0). +2. Rebuild the framework's tests as `test { }` blocks run by the new runner (stage 1). +3. Verify stage 1 reports identical results to stage 0. +4. From then on, the framework is tested by itself. + +A framework that cannot run its own suite is not evidence of anything. This is a correctness proof, +not a claim. + +--- + +## 11. Explicitly not building + +| Rejected | Why | +|---|---| +| Naming-convention discovery (`fn test_foo`) | `test { }` is a real declaration. Go's `TestXxx` exists only because Go had no better hook — and it needs a heuristic to avoid matching `TesticularCancer`. | +| Reflection or symbol-table scanning | Slow, fragile under LTO/strip/dead-strip, and unnecessary when we own the compiler. | +| Parsing human output into structure | Go's `test2json` is its one clear architectural mistake. | +| JUnit 5's extension SPI | Seventeen callback interfaces to retrofit plugins onto a reflective framework. Not our problem. | +| `@ParameterizedTest` machinery | Table-driven loops + subtests subsume it at zero surface. | +| NUnit's out-of-process agents | They bridge CLR versions and AppDomains. We emit one native binary. Keep `--isolate` as crash fallback only. | +| JMH-style forking by default | Forks exist because JIT profiles are per-process. AOT C has no such state. Keep `--fork` available, not default. | +| Exit 0 on regression | benchstat and Criterion both do this, and every user writes a wrapper. | +| Dynamic runtime test registration | Breaks `--list`, sharding, and individual selection. Registry stays static. | + +--- + +## 12. Phasing + +| Phase | Content | Gate | +|---|---|---| +| **1** | Registry emission in codegen; 9 builtins; `el_test_main` skeleton in El; result records; per-test timing; human + NDJSON output | existing 11 test files pass, with timing | +| **2** | `libeltest.a` build model; subtests; filtering; `--list`; fixtures; constraint assertions; JUnit XML | suite runs in one binary; runtime compiled once | +| **3** | `bench { }`, `bench_loop`, `black_box`, `predictN`, Criterion sampling | benchmarks produce stable ns/op | +| **4** | Allocation counters; complexity fitting; `expect O(...)` gate | **an `expect allocs O(n)` benchmark on `elc` fails on the current quadratic** | +| **5** | Migrate both legacy systems; delete `runtime/test.el`; self-host | framework runs its own suite | + +Phase 4 is the deliverable that matters. Phases 1–3 exist to make it possible. + +--- + +## 13. Open questions for review + +1. **Declarative `over n in [...] expect O(...)` syntax vs runtime calls.** I recommend declarative + (§6.1) so `--list` can show invariants without executing. It costs parser work. Your call. +2. **`bench { }` as a new block form** — parallel to `test { }`, or a modifier on it? +3. **Scope of the constraint model.** Full composable constraints, or start with a flat assertion set + and add constraints later? Full model is more surface but avoids a second migration. +4. **Does `runtime/test.el` get deleted or kept as a deprecated shim?** I lean delete — two systems + is how we got here. +5. **Where does `libeltest.a` live** in the tree, and does `epm` need to know about it? +6. **Allocation counters in `el_seed.c` or `el_runtime.c`?** AGENTS.md says `el_seed.c` is the sole + C dependency and hand-maintained; counters are OS-boundary-adjacent but not OS calls. +7. **Is per-test timing enough, or do we want per-*assertion* timing** for finding slow helpers? + +--- + +## 14. What this document is not + +This is a design, not a measurement. Every performance claim about the *current* system in §1 is +measured and reproducible in this worktree. Every claim about the *proposed* system is a prediction. +None of it is verified until Phase 1 runs and Phase 4 fails a build on the real quadratic. diff --git a/lang/el-compiler/src/codegen.el b/lang/el-compiler/src/codegen.el index 57fdbd4..fe1056f 100644 --- a/lang/el-compiler/src/codegen.el +++ b/lang/el-compiler/src/codegen.el @@ -2606,6 +2606,17 @@ fn builtin_arity(name: String) -> Int { // LSP seed primitives if str_eq(name, "__read_n") { return 1 } if str_eq(name, "__print_raw") { return 1 } + // Test-registry accessors. These are not runtime builtins — they are + // GENERATED into the same translation unit by the --test path below, one + // set per test binary. They are declared here so the El-side runner in + // runtime/eltest.el can call them with a known arity. + if str_eq(name, "__el_reg_count") { return 0 } + if str_eq(name, "__el_reg_name") { return 1 } + if str_eq(name, "__el_reg_invoke") { return 1 } + if str_eq(name, "__el_reg_last_ns") { return 0 } + if str_eq(name, "__el_reg_msg") { return 0 } + if str_eq(name, "__el_reg_asserts") { return 0 } + if str_eq(name, "__el_opt_json") { return 0 } // String if str_eq(name, "el_str_concat") { return 2 } if str_eq(name, "str_eq") { return 2 } @@ -4138,6 +4149,18 @@ fn codegen_streaming(tokens: [Any], sigs: [Map], source: String) -> emit_line(" __el_cur_fails++; __el_cur_asserts++;") emit_line("}") emit_blank() + // Forward declarations for the registry accessors. The definitions are + // emitted at the END of the unit (they reference the test functions, + // which do not exist yet at this point), but the El-side runner is + // compiled in between and calls them — so it needs the prototypes here. + emit_line("el_val_t __el_reg_count(void);") + emit_line("el_val_t __el_reg_name(el_val_t i);") + emit_line("el_val_t __el_reg_invoke(el_val_t i);") + emit_line("el_val_t __el_reg_last_ns(void);") + emit_line("el_val_t __el_reg_msg(void);") + emit_line("el_val_t __el_reg_asserts(void);") + emit_line("el_val_t __el_opt_json(void);") + emit_blank() } // Streaming parse-emit loop. diff --git a/lang/runtime/eltest.el b/lang/runtime/eltest.el index b84698b..27e0844 100644 --- a/lang/runtime/eltest.el +++ b/lang/runtime/eltest.el @@ -130,6 +130,7 @@ fn _elt_emit_result(json_mode: Bool, name: String, fails: Int, ns: Int, asserts: return } println("ok " + name + " (" + _elt_ms(ns) + "ms)") + return } fn _elt_emit_summary(json_mode: Bool, total: Int, failed: Int, ns: Int, asserts: Int) { @@ -149,6 +150,7 @@ fn _elt_emit_summary(json_mode: Bool, total: Int, failed: Int, ns: Int, asserts: println(int_to_str(total) + " tests, " + int_to_str(passed) + " passed, " + int_to_str(failed) + " failed, " + int_to_str(asserts) + " assertions in " + _elt_ms(ns) + "ms") + return } // ── The runner ─────────────────────────────────────────────────────────────── diff --git a/lang/tests/native/test_compiler.el b/lang/tests/native/test_compiler.el index 6798114..84ad5c4 100644 --- a/lang/tests/native/test_compiler.el +++ b/lang/tests/native/test_compiler.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // tests/native/test_compiler.el — comprehensive tests for the El compiler pipeline. // // Tests the lexer (lexer.el), parser (parser.el), and codegen (codegen.el) diff --git a/lang/tests/native/test_core.el b/lang/tests/native/test_core.el index f4a32c7..dc44ed2 100644 --- a/lang/tests/native/test_core.el +++ b/lang/tests/native/test_core.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_codegen_js.el - basic tests for JS codegen features. // // These tests verify that core El language features produce correct values diff --git a/lang/tests/native/test_env.el b/lang/tests/native/test_env.el index 4b33e08..8b28a0c 100644 --- a/lang/tests/native/test_env.el +++ b/lang/tests/native/test_env.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_env.el - native test suite for runtime/env.el // // Covers: env() for reading environment variables, args() returning a list, diff --git a/lang/tests/native/test_fs.el b/lang/tests/native/test_fs.el index 045610f..cf6743d 100644 --- a/lang/tests/native/test_fs.el +++ b/lang/tests/native/test_fs.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_fs.el - native test suite for runtime/fs.el // // Covers: fs_write/read round-trip, fs_exists, fs_mkdir, fs_list, diff --git a/lang/tests/native/test_json.el b/lang/tests/native/test_json.el index 30da7e3..7190a51 100644 --- a/lang/tests/native/test_json.el +++ b/lang/tests/native/test_json.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_json.el - native test suite for runtime/json.el // // Covers: json_get (dot-path), typed extractors (int, bool, float), diff --git a/lang/tests/native/test_math.el b/lang/tests/native/test_math.el index 4b3e22f..26dbe7a 100644 --- a/lang/tests/native/test_math.el +++ b/lang/tests/native/test_math.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_math.el - native test suite for runtime/math.el // // Covers: integer math (abs, max, min), float math (sqrt, log, sin, cos, pi), diff --git a/lang/tests/native/test_state.el b/lang/tests/native/test_state.el index 39f8438..6b05ac0 100644 --- a/lang/tests/native/test_state.el +++ b/lang/tests/native/test_state.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_state.el - native test suite for runtime/state.el // // Covers: state_set/get/del, state_has, state_get_or, state_keys, diff --git a/lang/tests/native/test_string.el b/lang/tests/native/test_string.el index e9f1192..78fc043 100644 --- a/lang/tests/native/test_string.el +++ b/lang/tests/native/test_string.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_string.el - native test suite for runtime/string.el // // Covers: type conversions, core primitives, comparison and search, diff --git a/lang/tests/native/test_text.el b/lang/tests/native/test_text.el index 869505d..9b0ac0f 100644 --- a/lang/tests/native/test_text.el +++ b/lang/tests/native/test_text.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_text.el - native test suite for text primitives. // // Mirrors the acceptance corpus in tests/text/examples/ using the diff --git a/lang/tests/native/test_time.el b/lang/tests/native/test_time.el index c4d8173..6a6d620 100644 --- a/lang/tests/native/test_time.el +++ b/lang/tests/native/test_time.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // test_time.el - native test suite for runtime/time.el // // Covers: time_now (positive timestamp), time_to_parts (UTC decomposition), diff --git a/lang/tests/runtime/string_test.el b/lang/tests/runtime/string_test.el index 322cb39..cf87afe 100644 --- a/lang/tests/runtime/string_test.el +++ b/lang/tests/runtime/string_test.el @@ -1,3 +1,4 @@ +import "../../runtime/eltest.el" // tests/runtime/string_test.el — Test suite for runtime/string.el // // Exercises every public function exported by runtime/string.el using the