4cacfcbdac
Hypothesis-driven measurement of "did monomorphisation actually buy us performance?" on a 100M-iter LCG hot loop, AILang mono'd code vs. four C reference variants (direct-inlinable, direct- noinline, indirect-monomorphic, indirect-polymorphic). Zen 3, clang -O2, median-of-15. Headline: H1 supported, but the mechanism is inlining, not dispatch shape. AILang mono = hand-C direct (1.000x). Indirect- monomorphic = direct-noinline (1.000x) — saturating branch predictor makes the indirect-call cost vanish on this hardware. Inlining is the actual 3.31x win; polymorphic indirect adds another 21% predictor-miss penalty. DESIGN.md Decision 11 gains a rationale paragraph reframing mono as inlining-enabler rather than indirect-call-eliminator, with explicit pointer to the bench. JOURNAL entry records the full methodology, ratios, limitations, and the side-effect mono-pass env.globals-seeding bug surfaced while building the AILang fixture (separate RED-first debug iter to follow).
195 lines
8.4 KiB
Python
Executable File
195 lines
8.4 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
# Mono-vs-virtual-dispatch micro-bench (one-off, hypothesis-driven).
|
|
#
|
|
# Hypothesis (H1): Monomorphised class-method calls in AILang are
|
|
# measurably faster than the equivalent function-pointer-indirected
|
|
# variant on a tight hot loop.
|
|
#
|
|
# What this harness measures:
|
|
# - AILang mono'd binary at -O2 --alloc=rc on
|
|
# examples/bench_mono_dispatch.ail.json
|
|
# - Hand-C reference, three variants:
|
|
# direct-inlinable — foo() inlinable, direct call
|
|
# direct-noinline — foo() noinline, direct call
|
|
# indirect — foo() noinline, called via volatile fnptr
|
|
#
|
|
# All four binaries run the same algorithm: a 100M-iter tail-recursive
|
|
# loop accumulating `acc + foo(acc + i)` where `foo(x) = x*1103515245
|
|
# + 12345`. The loop has a serial data dependency on `acc` so clang
|
|
# cannot algebraically close-form-fold it.
|
|
#
|
|
# Ratios reported (N runs, slowest dropped, median of the rest):
|
|
#
|
|
# ail / direct-inlinable — codegen-quality gap. Should be ~1.0x
|
|
# if mono produces an equivalent IR.
|
|
# direct-noinline / direct-inlinable
|
|
# — inlining benefit when callee is
|
|
# statically known.
|
|
# indirect / direct-noinline
|
|
# — THE actual mono-vs-vdisp delta on this
|
|
# hardware when both arms deny inlining.
|
|
# ail / indirect — end-to-end win: real workload (mono'd
|
|
# AILang, where LLVM CAN inline) vs
|
|
# hypothetical fnptr-dispatched AILang.
|
|
#
|
|
# What this bench does NOT show:
|
|
# - It does not measure dictionary-passing in any literal sense —
|
|
# AILang has no dict-passing implementation. The fnptr indirection
|
|
# stands in for "dispatch through an opaque target", which is the
|
|
# load-bearing optimiser barrier any vdisp scheme imposes.
|
|
# - It does not measure RC traffic. The fixture uses Int args (no
|
|
# heap allocation), so RC is a no-op. This is intentional: we
|
|
# are measuring DISPATCH cost, not RC cost. Dictionary RC traffic
|
|
# under a hypothetical vdisp implementation would be additional
|
|
# overhead on top of what `indirect` measures here.
|
|
# - It does not exercise polymorphic call sites with multiple
|
|
# instances (megamorphic dispatch). The fnptr is monomorphic
|
|
# in the C indirect variant; a multi-instance bench would
|
|
# produce a larger indirect/direct gap due to predictor misses.
|
|
#
|
|
# Usage: bench/mono_dispatch.py [-n RUNS]
|
|
# -n RUNS number of timed runs per binary (default 7; min 3).
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import statistics
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
from pathlib import Path
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
AIL = ROOT / "target" / "release" / "ail"
|
|
EXAMPLES = ROOT / "examples"
|
|
REFERENCE = ROOT / "bench" / "reference"
|
|
OUTDIR = ROOT / "target" / "bench_mono"
|
|
|
|
EXPECTED = "-2551317978420243992"
|
|
|
|
|
|
def time_one(bin_path: Path) -> tuple[float, str]:
|
|
"""Run binary, return (wall_seconds, stdout). Stdout captured for
|
|
correctness check (all four binaries must print the same value)."""
|
|
t0 = time.monotonic()
|
|
proc = subprocess.run([str(bin_path)], capture_output=True, text=True)
|
|
t1 = time.monotonic()
|
|
if proc.returncode != 0:
|
|
raise RuntimeError(f"{bin_path} exited {proc.returncode}: {proc.stderr}")
|
|
return (t1 - t0, proc.stdout.strip())
|
|
|
|
|
|
def median_drop_slowest(runs: list[float]) -> dict[str, float]:
|
|
if len(runs) < 2:
|
|
return {"min": runs[0], "median": runs[0], "max": runs[0]}
|
|
kept = sorted(runs)[:-1]
|
|
return {
|
|
"min": min(kept),
|
|
"median": statistics.median(kept),
|
|
"max": max(kept),
|
|
}
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser(description=__doc__)
|
|
ap.add_argument("-n", "--runs", type=int, default=7,
|
|
help="runs per binary; min 3, default 7")
|
|
args = ap.parse_args()
|
|
if args.runs < 3:
|
|
print("--runs must be >= 3", file=sys.stderr)
|
|
return 2
|
|
|
|
OUTDIR.mkdir(parents=True, exist_ok=True)
|
|
|
|
# Build AILang fixture if needed.
|
|
if not AIL.is_file():
|
|
subprocess.run(["cargo", "build", "--release", "-p", "ail"],
|
|
cwd=str(ROOT), check=True)
|
|
print(">>> building AILang fixture (-O2, --alloc=rc)", file=sys.stderr)
|
|
ail_bin = OUTDIR / "bench_mono_dispatch_ail"
|
|
subprocess.run(
|
|
[str(AIL), "build", "--opt=-O2", "--alloc=rc",
|
|
str(EXAMPLES / "bench_mono_dispatch.ail.json"), "-o", str(ail_bin)],
|
|
check=True, capture_output=True,
|
|
)
|
|
|
|
# Build the three C references.
|
|
print(">>> building C references (clang -O2)", file=sys.stderr)
|
|
# `indirect-polymorphic` produces a DIFFERENT stdout (4 distinct
|
|
# foo bodies → different acc); the harness skips the equality
|
|
# check for it but still times it.
|
|
c_specs = [
|
|
("direct-inlinable", "bench_mono_direct.c", True),
|
|
("direct-noinline", "bench_mono_direct_noinline.c", True),
|
|
("indirect", "bench_mono_indirect.c", True),
|
|
("indirect-polymorphic", "bench_mono_indirect_polymorphic.c", False),
|
|
]
|
|
c_bins: dict[str, tuple[Path, bool]] = {}
|
|
for label, src, equality_check in c_specs:
|
|
bin_path = OUTDIR / f"bench_mono_{label.replace('-', '_')}"
|
|
subprocess.run(
|
|
["clang", "-O2", "-o", str(bin_path), str(REFERENCE / src)],
|
|
check=True, capture_output=True,
|
|
)
|
|
c_bins[label] = (bin_path, equality_check)
|
|
|
|
# Equality check: ail-mono + the three monomorphic-foo C variants
|
|
# must all print the EXPECTED value. The polymorphic variant has
|
|
# different `foo` semantics → different acc; we skip the equality
|
|
# check for it.
|
|
print(">>> correctness check", file=sys.stderr)
|
|
binaries: list[tuple[str, Path, bool]] = [("ail-mono", ail_bin, True)]
|
|
binaries.extend((label, path, eq) for label, (path, eq) in c_bins.items())
|
|
for label, path, eq in binaries:
|
|
_, out = time_one(path)
|
|
if eq and out != EXPECTED:
|
|
print(f"correctness fail: {label} produced {out!r}, expected {EXPECTED!r}",
|
|
file=sys.stderr)
|
|
return 1
|
|
if not eq:
|
|
print(f" {label} stdout = {out} (equality check skipped)", file=sys.stderr)
|
|
print(f" monomorphic-foo binaries produce {EXPECTED}", file=sys.stderr)
|
|
|
|
# Time each binary RUNS times.
|
|
print(f">>> timing (runs={args.runs}, slowest dropped)", file=sys.stderr)
|
|
timings: dict[str, dict[str, float]] = {}
|
|
for label, path, _eq in binaries:
|
|
runs = [time_one(path)[0] for _ in range(args.runs)]
|
|
timings[label] = median_drop_slowest(runs)
|
|
timings[label]["raw"] = runs # full raw data for the report
|
|
|
|
# Report.
|
|
print()
|
|
print(f"=== bench_mono_dispatch ({args.runs} runs each, slowest dropped, all times in seconds) ===")
|
|
print()
|
|
print(f"{'binary':<24} {'min':>8} {'median':>8} {'max':>8} raw runs")
|
|
print("-" * 110)
|
|
for label, _, _eq in binaries:
|
|
t = timings[label]
|
|
raw_str = " ".join(f"{r:.3f}" for r in t["raw"])
|
|
print(f"{label:<24} {t['min']:8.4f} {t['median']:8.4f} {t['max']:8.4f} [{raw_str}]")
|
|
|
|
# Ratios (always median-over-median).
|
|
print()
|
|
print("=== ratios (median over median) ===")
|
|
print()
|
|
ail_med = timings["ail-mono"]["median"]
|
|
di_med = timings["direct-inlinable"]["median"]
|
|
dn_med = timings["direct-noinline"]["median"]
|
|
ind_med = timings["indirect"]["median"]
|
|
poly_med = timings["indirect-polymorphic"]["median"]
|
|
print(f" ail-mono / direct-inlinable = {ail_med / di_med:6.3f}x (codegen-quality gap)")
|
|
print(f" direct-noinline / direct-inlinable = {dn_med / di_med:6.3f}x (inlining benefit)")
|
|
print(f" indirect / direct-noinline = {ind_med / dn_med:6.3f}x (monomorphic vdisp delta)")
|
|
print(f" indirect-polymorphic / direct-noinline = {poly_med / dn_med:6.3f}x (polymorphic vdisp delta)")
|
|
print(f" indirect-polymorphic / indirect = {poly_med / ind_med:6.3f}x (predictor-miss penalty)")
|
|
print(f" ail-mono / indirect = {ail_med / ind_med:6.3f}x (end-to-end vs mono indirect)")
|
|
print(f" ail-mono / indirect-polymorphic = {ail_med / poly_med:6.3f}x (end-to-end vs poly indirect)")
|
|
print()
|
|
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|