cheatah
Source

docs/gen-cheatah/gen_bench_compare.purr

1# Copyright (c) 2026 BigBrain LLC. MIT-licensed (see LICENSE).
2# Original work; see ACKNOWLEDGMENTS.md for the open-source ideas we build upon.
3# gen_bench_compare.purr — the STRIATED driver for the doc-render comparison.
4#
5# The three programs it drives do the same work three ways: gen_bench.py (CPython with the
6# C-accelerated xml.etree), gen_bench.purr (single-threaded, from-scratch parsers.xml), and
7# gen_bench_parallel.purr (the same work over four threads sharing one memory.Owner). Each
8# now performs exactly ONE timed pass per invocation and prints `sample_ms=`.
9#
10# Why a driver rather than a loop inside each program. Previously each program ran 25 passes
11# internally and printed the MINIMUM, and the three were run one after another. That gives a
12# number with no dispersion — a rock-steady case and one that swings 40% print the same
13# thing — and it measures each program in its own block, so any clock or thermal drift over
14# the session lands on whichever ran last rather than cancelling out. Here one ROUND invokes
15# all three back to back, and the ratios are formed WITHIN a round before being aggregated,
16# which is the only estimator that stays unbiased while the machine moves under all three.
18# purrc docs/gen-cheatah/gen_bench_compare.purr -o /tmp/genbc.so
19# cheatah /tmp/genbc.so # run from the repo root; needs docs/xml
20import io
21import os
22import stamp
23import statistics
24import string
25import sys
27fn rounds() { return 9 }
29# Pull every `sample_ms=<x>` line out of a run's stdout. Returns -1.0 when the program
30# produced none, which the caller reports rather than silently averaging away.
31fn sample_of(path : str) -> float {
32 let text = io.read_file(path)
33 for line in string.splitlines(text) {
34 if string.startswith(line, "sample_ms=") {
35 return float(string.strip(string.replace(line, "sample_ms=", "")))
36 }
37 }
38 return -1.0
41fn run_once(cmd : str, out : str) -> float {
42 os.system(cmd + " > " + out + " 2>/dev/null")
43 return sample_of(out)
46fn spread(xs : list<float>) -> float {
47 let lo = xs[0]
48 let hi = xs[0]
49 for v in xs {
50 if v < lo { lo = v }
51 if v > hi { hi = v }
52 }
53 return hi - lo
56# One published row. `ratio` of 0 marks the baseline itself, which shows 1.0x.
57fn trow(label : str, xs : list<float>, ratio : float) -> str {
58 let r = "1.0×"
59 if ratio > 0.0 { r = "**" + io.fixed(ratio, 2) + "×**" }
60 let out = "| " + label + " | " + io.fixed(statistics.median(xs), 1) + " ms | ±"
61 out = out + io.fixed(statistics.stdev(xs), 1) + " ms | " + r + " |\n"
62 return out
65fn report(tag : str, xs : list<float>) {
66 io.print(tag, statistics.median(xs), statistics.stdev(xs), spread(xs))
69fn main() {
70 let purrc = "build/release/bin/purrc"
71 let cheatah = "build/release/bin/cheatah"
73 # Compile both cheatah sides ONCE. Compilation is not what is being measured, and doing
74 # it between rounds would put a multi-second gap between the sides of a comparison.
75 if os.system(purrc + " docs/gen-cheatah/gen_bench.purr -o /tmp/gb_single.so > /tmp/gb_c1.log 2>&1") != 0 {
76 io.print("gen_bench.purr failed to compile - see /tmp/gb_c1.log")
77 return
78 }
79 if os.system(purrc + " docs/gen-cheatah/gen_bench_parallel.purr -o /tmp/gb_par.so > /tmp/gb_c2.log 2>&1") != 0 {
80 io.print("gen_bench_parallel.purr failed to compile - see /tmp/gb_c2.log")
81 return
82 }
84 let py: list<float> = []
85 let single: list<float> = []
86 let par: list<float> = []
87 let r_single: list<float> = []
88 let r_par: list<float> = []
90 for k in range(0, rounds()) {
91 # One round touches all three before any is repeated.
92 let p = run_once("python3 docs/gen-cheatah/gen_bench.py", "/tmp/gb_py.out")
93 let s = run_once(cheatah + " /tmp/gb_single.so", "/tmp/gb_s.out")
94 let q = run_once(cheatah + " /tmp/gb_par.so", "/tmp/gb_p.out")
95 if p < 0.0 or s < 0.0 or q < 0.0 {
96 io.print("round", k, "produced no sample - aborting")
97 return
98 }
99 py.append(p)
100 single.append(s)
101 par.append(q)
102 r_single.append(p / s) # PAIRED within the round, not a ratio of two medians
103 r_par.append(p / q)
104 }
106 io.print("# doc-render kernel, " + io.str(rounds()) + " striated rounds")
107 io.print("# columns: median_ms stdev_ms range_ms")
108 report("cpython ", py)
109 report("cheatah-single ", single)
110 report("cheatah-parallel ", par)
111 io.print("# speedup = median of per-round PAIRED ratios, with [min,max]")
112 io.print("single vs cpython ", statistics.median(r_single), spread(r_single))
113 io.print("parallel vs cpython", statistics.median(r_par), spread(r_par))
114 io.print("parallel vs single ", statistics.median(single) / statistics.median(par))
116 # A published table is GENERATED or it is not published: sys.argv[1] is where
117 # scripts/bench_table.purr expects the artifact (docs/bench/render-kernel.md).
118 if len(sys.argv) > 1 {
119 let t = "| generator | median | spread (σ) | vs CPython |\n"
120 t = t + "|-----------|-------:|-----------:|-----------:|\n"
121 t = t + trow("CPython (`xml.etree`, native `expat`)", py, 0.0)
122 t = t + trow("cheatah, single-threaded (`gen_bench.purr`)", single,
123 statistics.median(r_single))
124 t = t + trow("cheatah, 4 threads over a shared `memory.Owner` (`gen_bench_parallel.purr`)",
125 par, statistics.median(r_par))
126 let st = stamp.build("render-kernel",
127 "stdlib/parsers/xml/, docs/gen-cheatah/gen_bench.purr, docs/gen-cheatah/gen_bench_parallel.purr, docs/gen-cheatah/gen_bench_compare.purr",
128 "CPython 3.12 (xml.etree / expat)",
129 rounds(),
130 "median wall clock; speedup = median of per-round PAIRED ratios",
131 "bash scripts/bench/build-harness.sh docs/gen-cheatah/gen_bench_compare.purr /tmp/gbc.so && cheatah /tmp/gbc.so docs/bench/render-kernel.md")
132 stamp.write_region(sys.argv[1], st, t)
133 }
136main()