#!/usr/bin/env python3 # Mono-vs-virtual-dispatch micro-bench (one-off, hypothesis-driven). # # Hypothesis (H1): Monomorphised class-method calls in AILang are # measurably faster than the equivalent function-pointer-indirected # variant on a tight hot loop. # # What this harness measures: # - AILang mono'd binary at -O2 --alloc=rc on # examples/bench_mono_dispatch.ail.json # - Hand-C reference, three variants: # direct-inlinable — foo() inlinable, direct call # direct-noinline — foo() noinline, direct call # indirect — foo() noinline, called via volatile fnptr # # All four binaries run the same algorithm: a 100M-iter tail-recursive # loop accumulating `acc + foo(acc + i)` where `foo(x) = x*1103515245 # + 12345`. The loop has a serial data dependency on `acc` so clang # cannot algebraically close-form-fold it. # # Ratios reported (N runs, slowest dropped, median of the rest): # # ail / direct-inlinable — codegen-quality gap. Should be ~1.0x # if mono produces an equivalent IR. # direct-noinline / direct-inlinable # — inlining benefit when callee is # statically known. # indirect / direct-noinline # — THE actual mono-vs-vdisp delta on this # hardware when both arms deny inlining. # ail / indirect — end-to-end win: real workload (mono'd # AILang, where LLVM CAN inline) vs # hypothetical fnptr-dispatched AILang. # # What this bench does NOT show: # - It does not measure dictionary-passing in any literal sense — # AILang has no dict-passing implementation. The fnptr indirection # stands in for "dispatch through an opaque target", which is the # load-bearing optimiser barrier any vdisp scheme imposes. # - It does not measure RC traffic. The fixture uses Int args (no # heap allocation), so RC is a no-op. This is intentional: we # are measuring DISPATCH cost, not RC cost. Dictionary RC traffic # under a hypothetical vdisp implementation would be additional # overhead on top of what `indirect` measures here. # - It does not exercise polymorphic call sites with multiple # instances (megamorphic dispatch). The fnptr is monomorphic # in the C indirect variant; a multi-instance bench would # produce a larger indirect/direct gap due to predictor misses. # # Usage: bench/mono_dispatch.py [-n RUNS] # -n RUNS number of timed runs per binary (default 7; min 3). from __future__ import annotations import argparse import statistics import subprocess import sys import time from pathlib import Path ROOT = Path(__file__).resolve().parent.parent AIL = ROOT / "target" / "release" / "ail" EXAMPLES = ROOT / "examples" REFERENCE = ROOT / "bench" / "reference" OUTDIR = ROOT / "target" / "bench_mono" EXPECTED = "-2551317978420243992" def time_one(bin_path: Path) -> tuple[float, str]: """Run binary, return (wall_seconds, stdout). Stdout captured for correctness check (all four binaries must print the same value).""" t0 = time.monotonic() proc = subprocess.run([str(bin_path)], capture_output=True, text=True) t1 = time.monotonic() if proc.returncode != 0: raise RuntimeError(f"{bin_path} exited {proc.returncode}: {proc.stderr}") return (t1 - t0, proc.stdout.strip()) def median_drop_slowest(runs: list[float]) -> dict[str, float]: if len(runs) < 2: return {"min": runs[0], "median": runs[0], "max": runs[0]} kept = sorted(runs)[:-1] return { "min": min(kept), "median": statistics.median(kept), "max": max(kept), } def main() -> int: ap = argparse.ArgumentParser(description=__doc__) ap.add_argument("-n", "--runs", type=int, default=7, help="runs per binary; min 3, default 7") args = ap.parse_args() if args.runs < 3: print("--runs must be >= 3", file=sys.stderr) return 2 OUTDIR.mkdir(parents=True, exist_ok=True) # Build AILang fixture if needed. if not AIL.is_file(): subprocess.run(["cargo", "build", "--release", "-p", "ail"], cwd=str(ROOT), check=True) print(">>> building AILang fixture (-O2, --alloc=rc)", file=sys.stderr) ail_bin = OUTDIR / "bench_mono_dispatch_ail" subprocess.run( [str(AIL), "build", "--opt=-O2", "--alloc=rc", str(EXAMPLES / "bench_mono_dispatch.ail.json"), "-o", str(ail_bin)], check=True, capture_output=True, ) # Build the three C references. print(">>> building C references (clang -O2)", file=sys.stderr) # `indirect-polymorphic` produces a DIFFERENT stdout (4 distinct # foo bodies → different acc); the harness skips the equality # check for it but still times it. c_specs = [ ("direct-inlinable", "bench_mono_direct.c", True), ("direct-noinline", "bench_mono_direct_noinline.c", True), ("indirect", "bench_mono_indirect.c", True), ("indirect-polymorphic", "bench_mono_indirect_polymorphic.c", False), ] c_bins: dict[str, tuple[Path, bool]] = {} for label, src, equality_check in c_specs: bin_path = OUTDIR / f"bench_mono_{label.replace('-', '_')}" subprocess.run( ["clang", "-O2", "-o", str(bin_path), str(REFERENCE / src)], check=True, capture_output=True, ) c_bins[label] = (bin_path, equality_check) # Equality check: ail-mono + the three monomorphic-foo C variants # must all print the EXPECTED value. The polymorphic variant has # different `foo` semantics → different acc; we skip the equality # check for it. print(">>> correctness check", file=sys.stderr) binaries: list[tuple[str, Path, bool]] = [("ail-mono", ail_bin, True)] binaries.extend((label, path, eq) for label, (path, eq) in c_bins.items()) for label, path, eq in binaries: _, out = time_one(path) if eq and out != EXPECTED: print(f"correctness fail: {label} produced {out!r}, expected {EXPECTED!r}", file=sys.stderr) return 1 if not eq: print(f" {label} stdout = {out} (equality check skipped)", file=sys.stderr) print(f" monomorphic-foo binaries produce {EXPECTED}", file=sys.stderr) # Time each binary RUNS times. print(f">>> timing (runs={args.runs}, slowest dropped)", file=sys.stderr) timings: dict[str, dict[str, float]] = {} for label, path, _eq in binaries: runs = [time_one(path)[0] for _ in range(args.runs)] timings[label] = median_drop_slowest(runs) timings[label]["raw"] = runs # full raw data for the report # Report. print() print(f"=== bench_mono_dispatch ({args.runs} runs each, slowest dropped, all times in seconds) ===") print() print(f"{'binary':<24} {'min':>8} {'median':>8} {'max':>8} raw runs") print("-" * 110) for label, _, _eq in binaries: t = timings[label] raw_str = " ".join(f"{r:.3f}" for r in t["raw"]) print(f"{label:<24} {t['min']:8.4f} {t['median']:8.4f} {t['max']:8.4f} [{raw_str}]") # Ratios (always median-over-median). print() print("=== ratios (median over median) ===") print() ail_med = timings["ail-mono"]["median"] di_med = timings["direct-inlinable"]["median"] dn_med = timings["direct-noinline"]["median"] ind_med = timings["indirect"]["median"] poly_med = timings["indirect-polymorphic"]["median"] print(f" ail-mono / direct-inlinable = {ail_med / di_med:6.3f}x (codegen-quality gap)") print(f" direct-noinline / direct-inlinable = {dn_med / di_med:6.3f}x (inlining benefit)") print(f" indirect / direct-noinline = {ind_med / dn_med:6.3f}x (monomorphic vdisp delta)") print(f" indirect-polymorphic / direct-noinline = {poly_med / dn_med:6.3f}x (polymorphic vdisp delta)") print(f" indirect-polymorphic / indirect = {poly_med / ind_med:6.3f}x (predictor-miss penalty)") print(f" ail-mono / indirect = {ail_med / ind_med:6.3f}x (end-to-end vs mono indirect)") print(f" ail-mono / indirect-polymorphic = {ail_med / poly_med:6.3f}x (end-to-end vs poly indirect)") print() return 0 if __name__ == "__main__": sys.exit(main())