c897d2eef0
Closes the question CLAUDE.md has carried since day one ("LLVM-
linkable, performance is extremely important") with data. Hand-C
variants of the four bench fixtures, compiled with clang -O2,
each carefully matching the AILang algorithm and explicitly
documenting representation differences (cell width, leak policy)
that affect the ratio.
Three substantive findings:
1. Pure-compute parity with C: bench_compute_collatz runs at
AILang/C = 0.99x across both allocators. AILang's IR composes
with LLVM's optimizer at the same level a hand-C source does.
This is the LLVM-linkable performance claim, backed by data
for the first time. bench_compute_intsum (1.05-1.18x) confirms.
2. AILang bump beats glibc malloc 2x on linear allocation:
bench_list_sum.bump/c = 0.50x. Bump's two-instruction inline
fastpath outperforms glibc's free-list-managed malloc on
no-free workloads. Quantitatively measured for the first time.
3. RC overhead vs C malloc quantified: bench_list_sum.rc/c =
1.49x, bench_tree_walk.rc/c = 2.61x. The 8-byte refcount
header + zero-init + libc backing add 50-160% over glibc
malloc on these implicit-mode workloads. Explicit-mode + a
free()-adding C variant (21'f, queued) will close the
apples-to-apples gap on dec-cost.
CLAUDE.md updated to list bench/cross_lang.py as the third
tidy-iter gate alongside bench/check.py and bench/compile_check.py.
20 new metrics in bench/baseline_cross_lang.json with 12-15%
tolerances (cross-language ratios are inherently noisier than
within-AILang ratios — two compiler stacks contribute variance).
242 lines
8.7 KiB
Python
Executable File
242 lines
8.7 KiB
Python
Executable File
#!/usr/bin/env python3
|
||
# Cross-language reference bench.
|
||
#
|
||
# For each fixture in CORPUS, builds:
|
||
# - AILang binary at -O2 --alloc=rc (canonical default)
|
||
# - AILang binary at -O2 --alloc=bump (no-free upper bound)
|
||
# - hand-C binary via clang -O2 from bench/reference/<fixture>.c
|
||
# Times each, drops the slowest of N runs, takes the median, computes
|
||
# AILang/C ratios. Diffs against bench/baseline_cross_lang.json or, with
|
||
# --update-baseline, captures fresh numbers as the new baseline.
|
||
#
|
||
# What this answers (and what it does not): the AILang/C ratio is the
|
||
# honest answer to CLAUDE.md's "LLVM-linkable, performance is extremely
|
||
# important" claim. Ratios near 1.0 mean AILang's IR composes with
|
||
# LLVM's optimizer at the same level as a hand-C source; ratios of
|
||
# 2-3x are JIT-quality territory; ratios of 10x+ are a real problem.
|
||
# The bench DOES NOT distinguish "AILang's IR is slow" from "the
|
||
# discriminated-union representation is wider than C's hand-tuned
|
||
# struct"; that's a representation question, not an optimizer one.
|
||
# Each fixture's .c file documents its representation choices so the
|
||
# orchestrator can read the ratio with that context in mind.
|
||
#
|
||
# Usage:
|
||
# bench/cross_lang.py # run + diff
|
||
# bench/cross_lang.py -n 10 # tighter median
|
||
# bench/cross_lang.py --update-baseline
|
||
# bench/cross_lang.py --baseline path
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import os
|
||
import resource
|
||
import subprocess
|
||
import sys
|
||
import time
|
||
from pathlib import Path
|
||
|
||
ROOT = Path(__file__).resolve().parent.parent
|
||
DEFAULT_BASELINE = ROOT / "bench" / "baseline_cross_lang.json"
|
||
AIL = ROOT / "target" / "release" / "ail"
|
||
EXAMPLES = ROOT / "examples"
|
||
REFERENCE = ROOT / "bench" / "reference"
|
||
OUTDIR = ROOT / "target" / "cross_lang"
|
||
|
||
# (ailang fixture stem, C reference stem) — both must exist as
|
||
# `examples/<ail>.ail.json` and `bench/reference/<c>.c`.
|
||
CORPUS = [
|
||
("bench_list_sum", "list_sum"),
|
||
("bench_tree_walk", "tree_walk"),
|
||
("bench_compute_intsum", "compute_intsum"),
|
||
("bench_compute_collatz", "compute_collatz"),
|
||
]
|
||
|
||
|
||
def time_one(bin_path: Path) -> float:
|
||
"""Run binary, return wall-time in seconds. Stdout discarded."""
|
||
t0 = time.monotonic()
|
||
proc = subprocess.run([str(bin_path)], stdout=subprocess.DEVNULL,
|
||
stderr=subprocess.DEVNULL)
|
||
t1 = time.monotonic()
|
||
if proc.returncode != 0:
|
||
raise RuntimeError(f"{bin_path} exited {proc.returncode}")
|
||
return t1 - t0
|
||
|
||
|
||
def median_drop_slowest(runs: list[float]) -> float:
|
||
if len(runs) < 2:
|
||
return runs[0] if runs else 0.0
|
||
kept = sorted(runs)[:-1]
|
||
n = len(kept)
|
||
if n % 2 == 1:
|
||
return kept[n // 2]
|
||
return 0.5 * (kept[n // 2 - 1] + kept[n // 2])
|
||
|
||
|
||
def build_ailang(stem: str, alloc: str) -> Path:
|
||
src = EXAMPLES / f"{stem}.ail.json"
|
||
if not src.is_file():
|
||
print(f"missing AILang fixture: {src}", file=sys.stderr)
|
||
sys.exit(2)
|
||
bin_path = OUTDIR / f"{stem}_{alloc}"
|
||
subprocess.run(
|
||
[str(AIL), "build", "--opt=-O2", f"--alloc={alloc}", str(src), "-o", str(bin_path)],
|
||
check=True, capture_output=True,
|
||
)
|
||
return bin_path
|
||
|
||
|
||
def build_c(stem: str) -> Path:
|
||
src = REFERENCE / f"{stem}.c"
|
||
if not src.is_file():
|
||
print(f"missing C reference: {src}", file=sys.stderr)
|
||
sys.exit(2)
|
||
bin_path = OUTDIR / f"{stem}_c"
|
||
subprocess.run(["clang", "-O2", "-o", str(bin_path), str(src)],
|
||
check=True, capture_output=True)
|
||
return bin_path
|
||
|
||
|
||
def measure(num_runs: int) -> dict[str, dict[str, float]]:
|
||
if not AIL.is_file():
|
||
subprocess.run(["cargo", "build", "--release", "-p", "ail"],
|
||
cwd=str(ROOT), check=True, capture_output=True)
|
||
OUTDIR.mkdir(parents=True, exist_ok=True)
|
||
|
||
out: dict[str, dict[str, float]] = {}
|
||
for ail_stem, c_stem in CORPUS:
|
||
print(f">>> {ail_stem}", file=sys.stderr)
|
||
ail_rc = build_ailang(ail_stem, "rc")
|
||
ail_bump = build_ailang(ail_stem, "bump")
|
||
c_bin = build_c(c_stem)
|
||
|
||
rc_runs = [time_one(ail_rc) for _ in range(num_runs)]
|
||
bump_runs = [time_one(ail_bump) for _ in range(num_runs)]
|
||
c_runs = [time_one(c_bin) for _ in range(num_runs)]
|
||
|
||
rc_s = median_drop_slowest(rc_runs)
|
||
bump_s = median_drop_slowest(bump_runs)
|
||
c_s = median_drop_slowest(c_runs)
|
||
|
||
out[ail_stem] = {
|
||
"ail_rc_s": rc_s,
|
||
"ail_bump_s": bump_s,
|
||
"c_s": c_s,
|
||
"rc_over_c": (rc_s / c_s) if c_s > 0 else 0.0,
|
||
"bump_over_c": (bump_s / c_s) if c_s > 0 else 0.0,
|
||
}
|
||
print(f" ail_rc={rc_s*1000:7.2f}ms ail_bump={bump_s*1000:7.2f}ms "
|
||
f"c={c_s*1000:7.2f}ms rc/c={out[ail_stem]['rc_over_c']:5.2f}× "
|
||
f"bump/c={out[ail_stem]['bump_over_c']:5.2f}×", file=sys.stderr)
|
||
return out
|
||
|
||
|
||
def diff_report(measured: dict, baseline: dict) -> tuple[str, bool]:
|
||
rows = []
|
||
has_regression = False
|
||
for fixture, spec_dict in baseline.get("fixtures", {}).items():
|
||
actual = measured.get(fixture)
|
||
if actual is None:
|
||
continue
|
||
for metric, spec in spec_dict.items():
|
||
base = spec["baseline"]
|
||
tol = spec["tolerance_pct"]
|
||
a = actual.get(metric)
|
||
if a is None:
|
||
continue
|
||
diff = 100.0 * (a - base) / base if base else 0.0
|
||
if diff > tol:
|
||
status = "REGRESSION"
|
||
has_regression = True
|
||
elif diff < -tol:
|
||
status = "improvement"
|
||
else:
|
||
status = "ok"
|
||
rows.append((f"{fixture}.{metric}", base, a, diff, tol, status))
|
||
|
||
lines = []
|
||
lines.append(f"{'metric':<48} {'baseline':>10} {'actual':>10} {'diff':>9} {'tol':>6} status")
|
||
lines.append("-" * 100)
|
||
for m, b, a, d, t, s in rows:
|
||
lines.append(f"{m:<48} {b:>10.4f} {a:>10.4f} {d:>+8.2f}% {t:>5.1f}% {s}")
|
||
regressed = sum(1 for r in rows if r[5] == "REGRESSION")
|
||
improved = sum(1 for r in rows if r[5] == "improvement")
|
||
stable = len(rows) - regressed - improved
|
||
lines.append("")
|
||
lines.append(f"summary: {len(rows)} metrics; "
|
||
f"{regressed} regressed, {improved} improved beyond tolerance, "
|
||
f"{stable} stable")
|
||
return "\n".join(lines), has_regression
|
||
|
||
|
||
def write_baseline(measured: dict, path: Path) -> None:
|
||
today = subprocess.check_output(["date", "+%Y-%m-%d"], text=True).strip()
|
||
if path.exists():
|
||
existing = json.loads(path.read_text())
|
||
existing_tols = existing.get("fixtures", {})
|
||
else:
|
||
existing_tols = {}
|
||
|
||
DEFAULT_TOLS = {
|
||
"ail_rc_s": 15,
|
||
"ail_bump_s": 15,
|
||
"c_s": 15,
|
||
"rc_over_c": 12,
|
||
"bump_over_c": 12,
|
||
}
|
||
|
||
new = {
|
||
"version": 1,
|
||
"captured": today,
|
||
"captured_via": "bench/cross_lang.py",
|
||
"note": "Cross-language wall-time baseline. Per-fixture: AILang at --alloc=rc, AILang at --alloc=bump, hand-C at clang -O2. Ratios rc/c and bump/c are the headline answer to CLAUDE.md's LLVM-linkable performance claim. The C reference uses malloc-and-leak to mirror AILang's implicit-mode RC; an explicit-mode + free() variant is queued.",
|
||
"fixtures": {},
|
||
}
|
||
for fixture, metrics in measured.items():
|
||
existing_fix = existing_tols.get(fixture, {})
|
||
new["fixtures"][fixture] = {
|
||
metric: {
|
||
"baseline": round(value, 6),
|
||
"tolerance_pct": existing_fix.get(metric, {}).get(
|
||
"tolerance_pct", DEFAULT_TOLS.get(metric, 15)
|
||
),
|
||
}
|
||
for metric, value in metrics.items()
|
||
}
|
||
path.write_text(json.dumps(new, indent=2) + "\n")
|
||
print(f">>> wrote new baseline to {path}", file=sys.stderr)
|
||
|
||
|
||
def main() -> int:
|
||
ap = argparse.ArgumentParser(description=__doc__)
|
||
ap.add_argument("-n", "--runs", type=int, default=5,
|
||
help="runs per binary; min 2, default 5")
|
||
ap.add_argument("--baseline", type=Path, default=DEFAULT_BASELINE)
|
||
ap.add_argument("--update-baseline", action="store_true")
|
||
args = ap.parse_args()
|
||
if args.runs < 2:
|
||
print("--runs must be >= 2", file=sys.stderr)
|
||
return 2
|
||
|
||
measured = measure(args.runs)
|
||
|
||
if args.update_baseline:
|
||
write_baseline(measured, args.baseline)
|
||
return 0
|
||
|
||
if not args.baseline.exists():
|
||
print(f"no baseline at {args.baseline}; create one with --update-baseline",
|
||
file=sys.stderr)
|
||
return 2
|
||
|
||
baseline = json.loads(args.baseline.read_text())
|
||
report, has_regression = diff_report(measured, baseline)
|
||
print(report)
|
||
return 1 if has_regression else 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|