{ "iter_id": "bench-harness-recalibration.1", "date": "2026-05-20", "mode": "standard", "outcome": "DONE", "tasks_total": 4, "tasks_completed": 4, "reloops_per_task": { "1": 0, "2": 0, "3": 0, "4": 0 }, "review_loops_spec": 0, "review_loops_quality": 0, "blocked_reason": null, "bench_results": { "check_py_exit_task3_replay": 0, "check_py_summary_task3": "57 metrics; 0 regressed, 0 improved beyond tolerance, 57 stable", "check_py_exit_task4_injection": 1, "check_py_injection_row": "throughput.bench_list_sum.bump_s 0.026 0.053 +99.57% 10.0% REGRESSION", "check_py_exit_task4_restored": 0, "check_py_summary_task4_restored": "57 metrics; 0 regressed, 2 improved beyond tolerance, 55 stable" }, "notes": [ "Task 3 same-HEAD replay passed on first try (no re-run needed for single-run noise mitigation).", "Task 4 restored replay shows 2 improvements on latency.explicit_at_rc.p99_us / p99_over_median (-33%) — improvements never gate per bench/check.py header; this is exactly the tail-distribution jitter that motivated dropping max_us / p99_9_us.", "All four tasks land as a single working-tree diff on bench/baseline.json (single-artefact iter)." ] }