#!/usr/bin/env bash # # GC-overhead bench harness (Bench iter). # # Builds each fixture twice — `--alloc=gc` (Boehm conservative GC) and # `--alloc=bump` (no-free 256 MB arena from `runtime/bump.c`). Runs each # binary N times, drops the slowest run, takes the median wall time. # The bump number minus the gc number is the upper-bound cost of GC. # # Output: a table with gc-median, bump-median, overhead %, and max RSS # for both modes. Designed to be captured verbatim into a JOURNAL entry. # # Requirements: bash, /usr/bin/time -v (GNU coreutils), bc, sort, awk, # a release-mode `ail` binary. # # Usage: bench/run.sh [-n RUNS] # -n RUNS number of timed runs per binary (default 5; min 3 so we # can drop the slowest and still take a median over 4). set -euo pipefail RUNS=5 while getopts "n:" opt; do case $opt in n) RUNS="$OPTARG" ;; *) echo "usage: $0 [-n RUNS]" >&2; exit 2 ;; esac done if (( RUNS < 3 )); then echo "RUNS must be >= 3" >&2 exit 2 fi # Anchor at the workspace root regardless of CWD. SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" cd "$ROOT" # We measure wall-clock + max RSS via a small Python helper that wraps # the binary, calls `time.monotonic()` around `os.waitpid`, and reads # `getrusage(RUSAGE_CHILDREN).ru_maxrss` (KB on Linux). This avoids a # /usr/bin/time dependency (Arch / minimal containers often don't have # the GNU coreutils `time` binary installed) without sacrificing # either signal: monotonic clocks for wall time, kernel-reported peak # resident set for RSS. PY="$(command -v python3 || true)" if [[ -z "$PY" ]]; then echo "error: python3 is required for the timing helper" >&2 exit 2 fi # Build the release `ail` binary if needed. echo ">>> ensuring release ail binary" cargo build --release -p ail >/dev/null AIL="$ROOT/target/release/ail" [[ -x "$AIL" ]] || { echo "ail binary missing: $AIL" >&2; exit 1; } OUTDIR="$ROOT/target/bench" mkdir -p "$OUTDIR" # Compile both modes for both fixtures up front so the bench loop only # measures runtime, not build time. fixtures=(bench_list_sum bench_tree_walk bench_closure_chain bench_hof_pipeline bench_compute_collatz bench_list_sum_explicit) modes=(gc bump rc) echo ">>> compiling fixtures (-O2)" for f in "${fixtures[@]}"; do src="$ROOT/examples/$f.ail" [[ -f "$src" ]] || { echo "missing fixture: $src" >&2; exit 1; } for m in "${modes[@]}"; do bin="$OUTDIR/${f}_${m}" echo " $f --alloc=$m -> $bin" "$AIL" build --opt=-O2 --alloc="$m" "$src" -o "$bin" >/dev/null done done # Time one binary one time. Wraps the binary in a Python helper that # measures wall-clock via time.monotonic() and max RSS (KB) via # getrusage(RUSAGE_CHILDREN).ru_maxrss after the child exits. Stdout # of the binary is discarded; we already verified correctness via a # smoke run earlier. Output: "wall_seconds rss_kb" on a single line. time_one() { local bin="$1" "$PY" -c ' import os, resource, subprocess, sys, time bin_path = sys.argv[1] t0 = time.monotonic() p = subprocess.Popen([bin_path], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) p.wait() t1 = time.monotonic() ru = resource.getrusage(resource.RUSAGE_CHILDREN) # ru_maxrss is in KB on Linux. We want the peak across only this child; # RUSAGE_CHILDREN is cumulative across all children of the helper, but # the helper only spawns this one child per invocation, so the value is # this run. print(f"{t1 - t0:.6f} {ru.ru_maxrss}") sys.exit(0 if p.returncode == 0 else 1) ' "$bin" } # Run a binary RUNS times, drop the slowest run by wall time, return # median of the rest as `wall rss` (rss = max across the kept runs). median_run() { local bin="$1" local times=() local rsses=() for ((i = 0; i < RUNS; i++)); do read -r w r < <(time_one "$bin") times+=("$w") rsses+=("$r") done # Compute index of slowest (largest wall) and drop it. local slowest_idx=0 for ((i = 1; i < ${#times[@]}; i++)); do if [[ $(awk -v a="${times[$i]}" -v b="${times[$slowest_idx]}" 'BEGIN { print (a > b) ? 1 : 0 }') == 1 ]]; then slowest_idx=$i fi done local kept_t=() local kept_r=() for ((i = 0; i < ${#times[@]}; i++)); do if [[ $i -ne $slowest_idx ]]; then kept_t+=("${times[$i]}") kept_r+=("${rsses[$i]}") fi done # Median wall over kept runs. local sorted_t sorted_t=$(printf "%s\n" "${kept_t[@]}" | sort -g) local n=${#kept_t[@]} local mid=$((n / 2)) local median_t if (( n % 2 == 1 )); then median_t=$(echo "$sorted_t" | sed -n "$((mid + 1))p") else local a b a=$(echo "$sorted_t" | sed -n "${mid}p") b=$(echo "$sorted_t" | sed -n "$((mid + 1))p") median_t=$(awk -v a="$a" -v b="$b" 'BEGIN { printf "%.6f", (a + b) / 2 }') fi # Max RSS across kept runs (peak memory is the natural per-run agg). local max_r=0 for r in "${kept_r[@]}"; do if (( r > max_r )); then max_r=$r; fi done printf "%s %s\n" "$median_t" "$max_r" } echo echo ">>> timing (RUNS=$RUNS, drop slowest, median of $((RUNS - 1)))" echo # Header. Iter 18f added the rc column + an "rc/bump" ratio, the # decisive number for Decision 10's Boehm-retirement target (1.3x). printf "%-22s | %10s | %10s | %10s | %10s | %10s | %12s | %12s | %12s\n" \ "workload" "gc(s)" "bump(s)" "rc(s)" "gc/bump" "rc/bump" "gc RSS(KB)" "bump RSS(KB)" "rc RSS(KB)" printf -- "-----------------------+------------+------------+------------+------------+------------+--------------+--------------+--------------\n" for f in "${fixtures[@]}"; do read -r gc_t gc_r < <(median_run "$OUTDIR/${f}_gc") read -r bp_t bp_r < <(median_run "$OUTDIR/${f}_bump") read -r rc_t rc_r < <(median_run "$OUTDIR/${f}_rc") # Guard against bump_t == 0 (LLVM-folded sub-microsecond fixtures). gc_ratio=$(awk -v g="$gc_t" -v b="$bp_t" 'BEGIN { if (b+0 == 0) printf "n/a"; else printf "%.2fx", g / b }') rc_ratio=$(awk -v r="$rc_t" -v b="$bp_t" 'BEGIN { if (b+0 == 0) printf "n/a"; else printf "%.2fx", r / b }') printf "%-22s | %10s | %10s | %10s | %10s | %10s | %12s | %12s | %12s\n" \ "$f" "$gc_t" "$bp_t" "$rc_t" "$gc_ratio" "$rc_ratio" "$gc_r" "$bp_r" "$rc_r" done # Iter 18g tidy: latency bench. The throughput table above is wall- # time-and-RSS — the wrong metric for Decision 10's real-time claim. # `bench/latency_harness.py` measures per-operation tail latency # (median + p99 + p99.9 + max) on PTY-line-buffered stdout for the # `bench_latency_*` fixtures. We invoke it for the three canonical # arms (Boehm-fair Implicit @ gc, RC-fair explicit @ rc, control # Implicit @ rc) and emit a second table. # # Skipped if the harness / fixtures aren't present (the latency bench # was added in 18f.2 and may not exist on older branches that share # this script). LAT_HARNESS="$ROOT/bench/latency_harness.py" LAT_IMPL_SRC="$ROOT/examples/bench_latency_implicit.ail" LAT_EXPL_SRC="$ROOT/examples/bench_latency_explicit.ail" if [[ -x "$LAT_HARNESS" && -f "$LAT_IMPL_SRC" && -f "$LAT_EXPL_SRC" ]]; then echo echo ">>> latency bench (PTY inter-arrival, 1000 samples per arm)" echo # Build the three arms. -O2 to match the throughput table. "$AIL" build --opt=-O2 --alloc=gc "$LAT_IMPL_SRC" -o "$OUTDIR/bench_latency_implicit_gc" >/dev/null "$AIL" build --opt=-O2 --alloc=rc "$LAT_EXPL_SRC" -o "$OUTDIR/bench_latency_explicit_rc" >/dev/null "$AIL" build --opt=-O2 --alloc=rc "$LAT_IMPL_SRC" -o "$OUTDIR/bench_latency_implicit_rc" >/dev/null # The harness prints a multi-line block per arm; we let it speak # for itself. The orchestrator captures the verbatim output into a # JOURNAL entry like the throughput table above. `--runs 5` runs # each arm five times; the harness drops the slowest run and # reports median + range per cell, matching the throughput # table's drop-slowest convention. "$PY" "$LAT_HARNESS" "$OUTDIR/bench_latency_implicit_gc" --runs 5 --label "implicit @ gc (Boehm-fair)" echo "$PY" "$LAT_HARNESS" "$OUTDIR/bench_latency_explicit_rc" --runs 5 --label "explicit @ rc (RC-fair)" echo "$PY" "$LAT_HARNESS" "$OUTDIR/bench_latency_implicit_rc" --runs 5 --label "implicit @ rc (control: leaks, no STW)" fi echo echo ">>> done"