cheatah
Source

scripts/app_compare.purr

1# Copyright (c) 2026 BigBrain LLC. MIT-licensed (see LICENSE).
2# Original work; see ACKNOWLEDGMENTS.md for the open-source ideas we build upon.
3# app_compare.purr — application-scale cheatah-vs-CPython benchmarks, written in
4# PURE CHEATAH (the benchmark harness dogfoods the language). The QA gate stays in
5# trusted Python/C++ tooling; benchmarks only measure already-verified code, so it is
6# safe — and fitting — to run them in cheatah itself.
7#
8# For each whole-program workload it compiles the cheatah version with purrc, runs it
9# (best of 3, the program self-times its compute), runs the CPython twin the same way,
10# and reports the speedup + cross-checks the two results. Run from the repo root after
11# a `release` build:
12# purrc scripts/app_compare.purr -o /tmp/app_compare.so
13# cheatah /tmp/app_compare.so
15import io
16import stamp
17import statistics
18import sys
19import os
20import string
22# A program prints `<result>` then `<elapsed seconds>`; capture both (best elapsed).
23struct Timing {
24 secs: float
25 result: str
28# ONE timed run. The round loop below calls this alternately for the two languages, so a
29# language's samples are never taken as one consecutive block.
30fn time_once(runcmd, outpath) {
31 os.system(runcmd + " > " + outpath + " 2>/dev/null")
32 let out = string.strip(io.read_file(outpath))
33 let lines = string.splitlines(out)
34 let n = len(lines)
35 if n >= 2 {
36 return Timing(float(lines[n - 1]), lines[n - 2])
37 }
38 return Timing(-1.0, "")
41fn spread_of(samples : list<float>) -> float {
42 let lo = samples[0]
43 let hi = samples[0]
44 for v in samples {
45 if v < lo { lo = v }
46 if v > hi { hi = v }
47 }
48 return hi - lo
51# |a - b| <= 1e-4 * max(1, |b|) — cheatah's io.print emits ~6 significant figures.
52fn close_enough(a, b) {
53 let cf = float(a)
54 let pf = float(b)
55 let diff = cf - pf
56 if diff < 0.0 { diff = -diff }
57 let scale = pf
58 if scale < 0.0 { scale = -scale }
59 if scale < 1.0 { scale = 1.0 }
60 return diff <= 0.0001 * scale
63# ---- top-level program (runs as purr_main) --------------------------------
64let PURRC = "build/release/bin/purrc"
65let CHEATAH = "build/release/bin/cheatah"
66let apps = ["mandelbrot", "integral", "oscillator", "nbody"]
68# The published table names these for a reader, not for the filesystem. Kept next to `apps`
69# so adding a workload without a description is visible here rather than silently blank.
70fn describe(a : str) -> str {
71 if a == "mandelbrot" { return "Mandelbrot" }
72 if a == "integral" { return "Numerical integral" }
73 if a == "oscillator" { return "RK4 ODE" }
74 if a == "nbody" { return "N-body" }
75 return a
78fn what(a : str) -> str {
79 if a == "mandelbrot" { return "escape-time over an 800×600 grid (≤256 iters)" }
80 if a == "integral" { return "trapezoid ∫ sin(x)·e^(−x/100), 20M points" }
81 if a == "oscillator" { return "4th-order Runge–Kutta, 4 000 000 steps" }
82 if a == "nbody" { return "direct O(N²) gravity, 256 bodies × 200 leapfrog steps" }
83 return ""
85# 7 rounds. The old harness took 3 and reported the MINIMUM of each side independently —
86# an optimistic estimator with no dispersion, and one that pairs a lucky cheatah run with an
87# unlucky CPython one. Here each round produces one PAIRED ratio, and the headline is the
88# median of those ratios, which stays unbiased when the machine drifts under both sides.
89let ROUNDS = 7
91io.print("# cheatah (compiled) vs CPython — whole-program workloads")
92io.print("# pure-cheatah harness; compute time only; results cross-checked")
93io.print("# " + io.str(ROUNDS) + " striated rounds: each round times cheatah then CPython before")
94io.print("# either is repeated, so drift across the session cannot favour one side.")
95io.print("# Reported: MEDIAN of per-round ratios (not the ratio of two medians) + [min,max].")
96io.print("")
97let table = "| program | what it does | cheatah | CPython | speedup |\n"
98table = table + "|---------|--------------|--------:|--------:|--------:|\n"
99for a in apps {
100 let purr = "scripts/bench/" + a + ".purr"
101 let so = "/tmp/bench_" + a + ".so"
102 let crc = os.system(PURRC + " " + purr + " -o " + so + " > /tmp/bench_compile.log 2>&1")
103 if crc != 0 {
104 io.print(a + " - cheatah compile FAILED")
105 } else {
106 let chs: list<float> = []
107 let pys: list<float> = []
108 let ratios: list<float> = []
109 let chk = "ok"
110 for r in range(0, ROUNDS) {
111 let ch = time_once(CHEATAH + " " + so, "/tmp/bench_ch.out")
112 let py = time_once("python3 scripts/bench/" + a + ".py", "/tmp/bench_py.out")
113 if ch.result != py.result and not close_enough(ch.result, py.result) {
114 chk = "DIFFER " + ch.result + " vs " + py.result
115 }
116 chs.append(ch.secs)
117 pys.append(py.secs)
118 ratios.append(py.secs / ch.secs)
119 }
120 let lo = ratios[0]
121 let hi = ratios[0]
122 for v in ratios {
123 if v < lo { lo = v }
124 if v > hi { hi = v }
125 }
126 let line = a + " | " + io.str(statistics.median(chs) * 1000.0) + " ms cheatah (range "
127 line = line + io.str(spread_of(chs) * 1000.0) + ") | "
128 line = line + io.str(statistics.median(pys) * 1000.0) + " ms cpython (range "
129 line = line + io.str(spread_of(pys) * 1000.0) + ") | "
130 line = line + io.str(statistics.median(ratios)) + "x faster [" + io.str(lo) + ", "
131 line = line + io.str(hi) + "] | " + chk
132 io.print(line)
134 # The published row. The bracketed band is the SPREAD OF THE PER-ROUND RATIOS, and it
135 # earns its place: on this workload CPython's own variance is large enough that a bare
136 # headline number hides which side of its range you are being shown.
137 let r = "| **" + describe(a) + "** | " + what(a) + " | "
138 r = r + io.fixed(statistics.median(chs) * 1000.0, 0) + " ms | "
139 r = r + io.fixed(statistics.median(pys) * 1000.0, 0) + " ms | **"
140 r = r + io.fixed(statistics.median(ratios), 0) + "×** [" + io.fixed(lo, 0) + "–"
141 r = r + io.fixed(hi, 0) + "] |\n"
142 table = table + r
143 }
145io.print("")
146io.print("Same algorithm both sides; cheatah compiles to native, no GC pauses.")
148# A published table is GENERATED or it is not published: sys.argv[1] is where
149# scripts/bench_table.purr expects the artifact (docs/bench/whole-programs.md).
150if len(sys.argv) > 1 {
151 let st = stamp.build("whole-programs",
152 "compiler/, runtime/, scripts/bench/, scripts/app_compare.purr",
153 "CPython 3.12",
154 ROUNDS,
155 "median of per-round PAIRED ratios; [lo–hi] is the range of those ratios",
156 "bash scripts/bench/build-harness.sh scripts/app_compare.purr /tmp/ac.so && cheatah /tmp/ac.so docs/bench/whole-programs.md")
157 stamp.write_region(sys.argv[1], st, table)