#!/usr/bin/env bash # # RC-overhead bench harness. # # Builds each fixture twice — `--alloc=rc` (canonical RC runtime) and # `--alloc=bump` (no-free 256 MB arena from `runtime/bump.c`, the # raw-alloc bench-floor). Runs each binary N times, drops the slowest # run, takes the median wall time. The rc-over-bump ratio is the # bench-health regression gate. # # Output: a table with bump-median, rc-median, rc/bump ratio, and max # RSS for both modes. Designed to be captured verbatim into a commit # body. # # Requirements: bash, /usr/bin/time -v (GNU coreutils), bc, sort, awk, # a release-mode `ail` binary. # # Usage: bench/run.sh [-n RUNS] # -n RUNS number of timed runs per binary (default 5; min 3 so we # can drop the slowest and still take a median over 4). set -euo pipefail RUNS=5 while getopts "n:" opt; do case $opt in n) RUNS="$OPTARG" ;; *) echo "usage: $0 [-n RUNS]" >&2; exit 2 ;; esac done if (( RUNS < 3 )); then echo "RUNS must be >= 3" >&2 exit 2 fi # Anchor at the workspace root regardless of CWD. SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" cd "$ROOT" # We measure wall-clock + max RSS via a small Python helper that wraps # the binary, calls `time.monotonic()` around `os.waitpid`, and reads # `getrusage(RUSAGE_CHILDREN).ru_maxrss` (KB on Linux). This avoids a # /usr/bin/time dependency (Arch / minimal containers often don't have # the GNU coreutils `time` binary installed) without sacrificing # either signal: monotonic clocks for wall time, kernel-reported peak # resident set for RSS. PY="$(command -v python3 || true)" if [[ -z "$PY" ]]; then echo "error: python3 is required for the timing helper" >&2 exit 2 fi # Build the release `ail` binary if needed. echo ">>> ensuring release ail binary" cargo build --release -p ail >/dev/null AIL="$ROOT/target/release/ail" [[ -x "$AIL" ]] || { echo "ail binary missing: $AIL" >&2; exit 1; } OUTDIR="$ROOT/target/bench" mkdir -p "$OUTDIR" # Compile both modes for both fixtures up front so the bench loop only # measures runtime, not build time. fixtures=(bench_list_sum bench_tree_walk bench_closure_chain bench_hof_pipeline bench_compute_collatz bench_list_sum_explicit) modes=(bump rc) echo ">>> compiling fixtures (-O2)" for f in "${fixtures[@]}"; do src="$ROOT/examples/$f.ail" [[ -f "$src" ]] || { echo "missing fixture: $src" >&2; exit 1; } for m in "${modes[@]}"; do bin="$OUTDIR/${f}_${m}" echo " $f --alloc=$m -> $bin" "$AIL" build --opt=-O2 --alloc="$m" "$src" -o "$bin" >/dev/null done done # Time one binary one time. Wraps the binary in a Python helper that # measures wall-clock via time.monotonic() and max RSS (KB) via # getrusage(RUSAGE_CHILDREN).ru_maxrss after the child exits. Stdout # of the binary is discarded; we already verified correctness via a # smoke run earlier. Output: "wall_seconds rss_kb" on a single line. time_one() { local bin="$1" "$PY" -c ' import os, resource, subprocess, sys, time bin_path = sys.argv[1] t0 = time.monotonic() p = subprocess.Popen([bin_path], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) p.wait() t1 = time.monotonic() ru = resource.getrusage(resource.RUSAGE_CHILDREN) # ru_maxrss is in KB on Linux. We want the peak across only this child; # RUSAGE_CHILDREN is cumulative across all children of the helper, but # the helper only spawns this one child per invocation, so the value is # this run. print(f"{t1 - t0:.6f} {ru.ru_maxrss}") sys.exit(0 if p.returncode == 0 else 1) ' "$bin" } # Run a binary RUNS times, drop the slowest run by wall time, return # median of the rest as `wall rss` (rss = max across the kept runs). median_run() { local bin="$1" local times=() local rsses=() for ((i = 0; i < RUNS; i++)); do read -r w r < <(time_one "$bin") times+=("$w") rsses+=("$r") done # Compute index of slowest (largest wall) and drop it. local slowest_idx=0 for ((i = 1; i < ${#times[@]}; i++)); do if [[ $(awk -v a="${times[$i]}" -v b="${times[$slowest_idx]}" 'BEGIN { print (a > b) ? 1 : 0 }') == 1 ]]; then slowest_idx=$i fi done local kept_t=() local kept_r=() for ((i = 0; i < ${#times[@]}; i++)); do if [[ $i -ne $slowest_idx ]]; then kept_t+=("${times[$i]}") kept_r+=("${rsses[$i]}") fi done # Median wall over kept runs. local sorted_t sorted_t=$(printf "%s\n" "${kept_t[@]}" | sort -g) local n=${#kept_t[@]} local mid=$((n / 2)) local median_t if (( n % 2 == 1 )); then median_t=$(echo "$sorted_t" | sed -n "$((mid + 1))p") else local a b a=$(echo "$sorted_t" | sed -n "${mid}p") b=$(echo "$sorted_t" | sed -n "$((mid + 1))p") median_t=$(awk -v a="$a" -v b="$b" 'BEGIN { printf "%.6f", (a + b) / 2 }') fi # Max RSS across kept runs (peak memory is the natural per-run agg). local max_r=0 for r in "${kept_r[@]}"; do if (( r > max_r )); then max_r=$r; fi done printf "%s %s\n" "$median_t" "$max_r" } echo echo ">>> timing (RUNS=$RUNS, drop slowest, median of $((RUNS - 1)))" echo # Header. The rc/bump ratio is the RC-overhead-vs-bump bench-health # regression gate (1.3× ceiling on linear/tree corpus, ±15% on # closure-chain). printf "%-22s | %10s | %10s | %10s | %12s | %12s\n" \ "workload" "bump(s)" "rc(s)" "rc/bump" "bump RSS(KB)" "rc RSS(KB)" printf -- "-----------------------+------------+------------+------------+--------------+--------------\n" for f in "${fixtures[@]}"; do read -r bp_t bp_r < <(median_run "$OUTDIR/${f}_bump") read -r rc_t rc_r < <(median_run "$OUTDIR/${f}_rc") # Guard against bump_t == 0 (LLVM-folded sub-microsecond fixtures). rc_ratio=$(awk -v r="$rc_t" -v b="$bp_t" 'BEGIN { if (b+0 == 0) printf "n/a"; else printf "%.2fx", r / b }') printf "%-22s | %10s | %10s | %10s | %12s | %12s\n" \ "$f" "$bp_t" "$rc_t" "$rc_ratio" "$bp_r" "$rc_r" done # Latency bench. The throughput table above is wall-time-and-RSS; # `bench/latency_harness.py` measures per-operation tail latency # (median + p99 + p99.9 + max) on PTY-line-buffered stdout for the # `bench_latency_*` fixtures. We invoke it for the two RC arms # (RC-fair explicit @ rc, implicit-mode @ rc as control) and emit a # second table. # # Skipped if the harness / fixtures aren't present (the latency bench # was added in 18f.2 and may not exist on older branches that share # this script). LAT_HARNESS="$ROOT/bench/latency_harness.py" LAT_IMPL_SRC="$ROOT/examples/bench_latency_implicit.ail" LAT_EXPL_SRC="$ROOT/examples/bench_latency_explicit.ail" if [[ -x "$LAT_HARNESS" && -f "$LAT_IMPL_SRC" && -f "$LAT_EXPL_SRC" ]]; then echo echo ">>> latency bench (PTY inter-arrival, 1000 samples per arm)" echo # Build the two RC arms. -O2 to match the throughput table. "$AIL" build --opt=-O2 --alloc=rc "$LAT_EXPL_SRC" -o "$OUTDIR/bench_latency_explicit_rc" >/dev/null "$AIL" build --opt=-O2 --alloc=rc "$LAT_IMPL_SRC" -o "$OUTDIR/bench_latency_implicit_rc" >/dev/null # The harness prints a multi-line block per arm; we let it speak # for itself. The orchestrator captures the verbatim output into # the commit body like the throughput table above. `--runs 5` runs # each arm five times; the harness drops the slowest run and # reports median + range per cell, matching the throughput # table's drop-slowest convention. "$PY" "$LAT_HARNESS" "$OUTDIR/bench_latency_explicit_rc" --runs 5 --label "explicit @ rc (RC-fair)" echo "$PY" "$LAT_HARNESS" "$OUTDIR/bench_latency_implicit_rc" --runs 5 --label "implicit @ rc (control)" fi echo echo ">>> done"