"""flaky-test-rerun-math.py — how many reruns it takes to catch a flaky test. A test that passes with probability p on each independent run is flaky whenever 0 < p < 1. Run it n times and you only *learn* it is flaky if you see both a pass and a failure; if all n runs agree, the test looks deterministic and you move on. P(all n pass) = p**n P(all n fail) = (1-p)**n P(detect flake) = 1 - p**n - (1-p)**n That closed form is the whole argument, and the conditional case is where it bites. A test has just failed and you rerun it once. The rerun passes with probability p, and for a mildly flaky test p is close to 1 - so a green rerun is what you would see whether the first failure was noise or the first sighting of a real intermittent bug. The rerun does not distinguish between them. That is the problem with rerun-on-red as a policy, and it is arithmetic, not opinion. This script computes both views and writes the dataset the article uses. It also runs a Monte-Carlo simulation at each rate and checks the simulated detection rate against the closed form, so the table is confirmed two independent ways. No third-party packages. Standard library only. python flaky-test-rerun-math.py # print the tables python flaky-test-rerun-math.py --csv out.csv # write the dataset Output (unless --csv says otherwise): ../datasets/flaky-test-rerun-math.csv """ from __future__ import annotations import argparse import csv import os import random import sys # Pass probabilities to study. 0.99 is the test that fails one run in a hundred: # the kind everybody has and nobody catches. RATES = [0.99, 0.95, 0.90, 0.80, 0.70, 0.50] RUNS = [1, 2, 3, 5, 10, 20, 50, 100] TRIALS = 200_000 SEED = 20260921 def p_detect(p: float, n: int) -> float: """Probability that n independent runs show BOTH a pass and a failure.""" if n < 2: return 0.0 return 1.0 - p ** n - (1.0 - p) ** n def p_rerun_goes_green(p: float) -> float: """A test has just failed; you rerun it once. It passes with probability p. The higher p is, the less a green rerun tells you: at p = 0.99 almost every rerun is green, whatever caused the original failure.""" return p def runs_needed(p: float, target: float = 0.95) -> int: """Smallest n with P(detect) >= target. Capped so a coin-flip test does not run forever; returns -1 if the cap is reached.""" for n in range(2, 5001): if p_detect(p, n) >= target: return n return -1 def simulate(p: float, n: int, trials: int, rng: random.Random) -> float: """Monte-Carlo check of p_detect: fraction of trials where n runs disagree.""" hits = 0 for _ in range(trials): first = rng.random() < p for _ in range(n - 1): if (rng.random() < p) != first: hits += 1 break return hits / trials def build_rows(trials: int = TRIALS) -> list[dict]: rng = random.Random(SEED) rows = [] for p in RATES: for n in RUNS: closed = p_detect(p, n) sim = simulate(p, n, trials, rng) if n > 1 else 0.0 rows.append(dict( pass_rate=round(p, 4), failure_rate=round(1 - p, 4), runs=n, p_detect_closed_form=round(closed, 6), p_detect_simulated=round(sim, 6), abs_error=round(abs(closed - sim), 6), )) return rows def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--csv", default=None, help="where to write the dataset") ap.add_argument("--trials", type=int, default=TRIALS) args = ap.parse_args() rows = build_rows(args.trials) print(f"Monte-Carlo trials per cell: {args.trials:,} seed: {SEED}\n") print("P(n runs reveal the flake) = 1 - p^n - (1-p)^n\n") head = " fail rate |" + "".join(f"{n:>8}" for n in RUNS) print(head) print(" " + "-" * (len(head) - 2)) for p in RATES: cells = "".join(f"{p_detect(p, n) * 100:>7.1f}%" for n in RUNS) print(f" {(1 - p) * 100:>8.0f}% |{cells}") worst = max(r["abs_error"] for r in rows) print(f"\n largest gap between simulation and closed form: {worst:.4f}") print("\n runs needed for a 95% chance of ever seeing the flake:") for p in RATES: n = runs_needed(p, 0.95) print(f" failure rate {(1 - p) * 100:>5.1f}% -> " + (f"{n} runs" if n > 0 else "more than 5,000 runs")) print("\n a test has just failed; one rerun is allowed:") for p in RATES: print(f" failure rate {(1 - p) * 100:>5.1f}% -> " f"{p_rerun_goes_green(p) * 100:.1f}% chance the rerun is green") here = os.path.dirname(os.path.abspath(__file__)) out = args.csv or os.path.join(here, os.pardir, "datasets", "flaky-test-rerun-math.csv") out = os.path.normpath(out) with open(out, "w", newline="", encoding="utf-8") as f: w = csv.DictWriter(f, fieldnames=list(rows[0].keys())) w.writeheader() w.writerows(rows) print(f"\n dataset -> {out} ({len(rows)} rows)") return 0 if __name__ == "__main__": sys.exit(main())