"""swe-bench-verified-task-by-task.py - read SWE-bench Verified one task at a time, from the leaderboard's own records. What it does 1. Lists the Verified entries in the SWE-bench/experiments repository at one pinned commit, and reads each entry's metadata.yaml and its per-task record (per_instance_details.json, or results/results.json for older entries). 2. Reads the 500 SWE-bench Verified tasks, with their repository and the annotators' time estimate ("difficulty"), from the Hugging Face datasets-server rows API. 3. Takes the entries run with one harness version (mini-swe-agent 2.0.0 by default) whose per-task record agrees with the entry's own metadata, and prints: each run's resolve rate by difficulty and its recorded cost; how many of those runs resolved each task; what a cap on model calls would have kept and saved; and, across every entry with a per-task record, the tasks no entry has resolved. 4. Optionally writes one CSV row per task (--csv). Inputs: network access to api.github.com (one request, unauthenticated), raw.githubusercontent.com and datasets-server.huggingface.co, or a --cache folder holding responses saved by an earlier run. Standard library only (Python 3.9 or later). Run: python swe-bench-verified-task-by-task.py # live python swe-bench-verified-task-by-task.py --cache saved # read saved responses, save any it has to fetch python swe-bench-verified-task-by-task.py --cache saved --csv swe-bench-verified-task-by-task.csv Words: a "run" is one entry's pass over the 500 tasks; an "attempt" is one task in one run (one attempt per task). A cap of k calls is applied after the fact: an attempt that resolved its task in k calls or fewer is kept, any other attempt counts as unresolved, and every attempt is charged at most k calls. That is exact for mini-swe-agent, which checks its limits before each model call and ends with an empty submission; it does not say what a capped attempt costs in dollars, because the records hold one total cost per attempt, not one per call. """ from __future__ import annotations import argparse import csv import json import os import re import statistics import sys import urllib.error import urllib.request REPO = "SWE-bench/experiments" PINNED = "40f164d5b8f1d249bf95a6df8b74b577fd8e519d" # main on 2026-09-03, the latest commit on 2026-10-08 ROWS = ("https://datasets-server.huggingface.co/rows?dataset=SWE-bench%2FSWE-bench_Verified" "&config=default&split=test&offset={off}&length=100") UA = "swe-bench-verified-task-by-task/1.0 (research script)" DIFFICULTY = ["<15 min fix", "15 min - 1 hour", "1-4 hours", ">4 hours"] SHORT = {"<15 min fix": "<15 min", "15 min - 1 hour": "15m-1h", "1-4 hours": "1-4 h", ">4 hours": ">4 h"} class Source: """Fetch a URL, or read it from the cache folder; save what was fetched when a cache folder is given.""" def __init__(self, cache: str | None): self.cache = cache def get(self, url: str, rel: str, optional: bool = False, accept: str | None = None) -> bytes | None: path = os.path.join(self.cache, rel.replace("/", os.sep)) if self.cache else None if path and os.path.exists(path): with open(path, "rb") as f: return f.read() if path and os.path.exists(path + ".404"): return None headers = {"User-Agent": UA} if accept: headers["Accept"] = accept try: with urllib.request.urlopen(urllib.request.Request(url, headers=headers), timeout=90) as r: body = r.read() except urllib.error.HTTPError as e: if e.code == 404 and optional: if path: os.makedirs(os.path.dirname(path), exist_ok=True) open(path + ".404", "w").close() return None raise if path: os.makedirs(os.path.dirname(path), exist_ok=True) with open(path, "wb") as f: f.write(body) return body def info_block(yaml_text: str) -> dict: """The scalar keys under the top-level `info:` key of an entry's metadata.yaml (enough for this script).""" out, top = {}, None for line in yaml_text.splitlines(): if line and not line.startswith(" ") and line.rstrip().endswith(":"): top = line.strip()[:-1] continue m = re.match(r"^ ([A-Za-z0-9_-]+):\s*(.*)$", line) if top == "info" and m and m.group(2) and not m.group(2).startswith("-"): out[m.group(1)] = m.group(2).strip().strip("'\"") return out def load(src: Source, sha: str): raw = f"https://raw.githubusercontent.com/{REPO}/{sha}/" listing = json.loads(src.get(f"https://api.github.com/repos/{REPO}/contents/evaluation/verified?ref={sha}", "contents_evaluation_verified.json", accept="application/vnd.github+json")) entries = {} for e in sorted(listing, key=lambda e: e["name"]): if e["type"] != "dir": continue d, base = e["name"], f"evaluation/verified/{e['name']}/" meta = src.get(raw + base + "metadata.yaml", f"verified/{d}/metadata.yaml", optional=True) rec, kind = src.get(raw + base + "per_instance_details.json", f"verified/{d}/per_instance_details.json", optional=True), "per_instance_details" if rec is None: rec, kind = src.get(raw + base + "results/results.json", f"verified/{d}/results/results.json", optional=True), "results" runs, resolved = None, None if rec is not None: j = json.loads(rec) if kind == "per_instance_details": runs = j resolved = {i for i, v in j.items() if v.get("resolved")} else: resolved = set(j.get("resolved", [])) entries[d] = {"info": info_block(meta.decode("utf-8")) if meta else {}, "runs": runs, "resolved": resolved} tasks = {} for off in range(0, 500, 100): page = json.loads(src.get(ROWS.format(off=off), f"hf_rows_{off:03d}.json")) for r in page["rows"]: row = r["row"] tasks[row["instance_id"]] = {"repo": row["repo"], "difficulty": row["difficulty"]} if len(tasks) != 500: sys.exit(f"expected 500 Verified tasks, read {len(tasks)}") return entries, tasks def pct(n: float, d: float) -> str: return f"{100 * n / d:.1f}%" def kept_and_saved(runs: dict, cap: float) -> tuple[int, int, int, int]: """Resolves kept and calls saved if every attempt had been stopped after `cap` model calls.""" resolved = [v["api_calls"] for v in runs.values() if v["resolved"]] kept = sum(1 for c in resolved if c <= cap) saved = sum(max(0, v["api_calls"] - cap) for v in runs.values()) return kept, len(resolved), saved, sum(v["api_calls"] for v in runs.values()) def main(argv=None) -> int: ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) ap.add_argument("--cache", help="folder of saved responses (read first; anything fetched is saved there)") ap.add_argument("--sha", default=PINNED, help="SWE-bench/experiments commit to read (default: %(default).12s)") ap.add_argument("--harness", default="2.0.0", help="mini-swe-agent version to hold fixed (default: %(default)s)") ap.add_argument("--caps", default="25,50,75,100,150", help="call caps to test (default: %(default)s)") ap.add_argument("--csv", help="write one row per task to this file") a = ap.parse_args(argv) if hasattr(sys.stdout, "reconfigure"): sys.stdout.reconfigure(encoding="utf-8") entries, tasks = load(Source(a.cache), a.sha) ids = sorted(tasks) with_rec = {d: e for d, e in entries.items() if e["resolved"] is not None} labels = {b: [i for i in ids if tasks[i]["difficulty"] == b] for b in DIFFICULTY} print(f"{REPO} at {a.sha[:12]}: {len(entries)} Verified entries, {len(with_rec)} with a per-task record") print("SWE-bench Verified: 500 tasks; difficulty labels: " + ", ".join(f"{b} {len(labels[b])}" for b in DIFFICULTY)) # ---- the runs: one harness version, records that agree with their own metadata ---- same = [d for d, e in entries.items() if e["info"].get("mini-swe-agent_version") == a.harness] no_rec = [d for d in same if entries[d]["runs"] is None] unusable, used = [], {} for d in same: e = entries[d] if e["runs"] is None: continue n = len(e["resolved"] & set(ids)) stated = e["info"].get("resolved") if set(e["runs"]) != set(ids): unusable.append(f"{d} (the record covers {len(set(e['runs']) & set(ids))} of the 500 tasks)") elif stated is None: unusable.append(f"{d} (its metadata states no resolve rate to check the record against)") elif abs(100 * n / 500 - float(stated)) > 0.05: unusable.append(f"{d} ({n} resolved in the record, {stated}% in its metadata)") elif not all("api_calls" in v and "cost" in v for v in e["runs"].values()): unusable.append(f"{d} (the record lacks a call count or a cost for some tasks)") elif n in (0, 500): unusable.append(f"{d} (resolves {n} of 500, so resolved and unresolved runs cannot be compared)") else: used[e["info"].get("name", d)] = e["runs"] print(f"\nmini-swe-agent {a.harness} entries: {len(same)}") print(f" without a per-task record: {len(no_rec)}" + (f" ({', '.join(no_rec)})" if no_rec else "")) print(f" record incomplete or at odds with its own metadata: {len(unusable)}" + (f" ({'; '.join(unusable)})" if unusable else "")) print(f" used: {len(used)} runs x 500 tasks = {500 * len(used):,} attempts") if not used: print(" nothing to compare") return 1 names = sorted(used, key=lambda m: (-sum(v["resolved"] for v in used[m].values()), m)) w = max(len(m) for m in names) print(f"\n{'run':{w}} resolved " + " ".join(f"{SHORT[b]:>7}" for b in DIFFICULTY) + " cost $ $/resolve median calls, resolved / not") for m in names: r = used[m] res = sum(v["resolved"] for v in r.values()) cost = sum(v.get("cost", 0.0) for v in r.values()) by = [pct(sum(r[i]["resolved"] for i in labels[b]), len(labels[b])) for b in DIFFICULTY] cr = statistics.median(v["api_calls"] for v in r.values() if v["resolved"]) cf = statistics.median(v["api_calls"] for v in r.values() if not v["resolved"]) print(f"{m:{w}} {res:4} {pct(res, 500):>6} " + " ".join(f"{x:>7}" for x in by) + f" {cost:7.2f} {cost / res:9.3f} {cr:6g} / {cf:g}") allr = [v for r in used.values() for v in r.values()] by = [pct(sum(used[m][i]["resolved"] for m in used for i in labels[b]), len(labels[b]) * len(used)) for b in DIFFICULTY] tot_res = sum(v["resolved"] for v in allr) tot_cost = sum(v.get("cost", 0.0) for v in allr) print(f"{f'all {len(used)} runs':{w}} {tot_res:4} {pct(tot_res, len(allr)):>6} " + " ".join(f"{x:>7}" for x in by) + f" {tot_cost:7.2f} {tot_cost / tot_res:9.3f}") # ---- agreement between the runs, task by task ---- k = {i: sum(used[m][i]["resolved"] for m in used) for i in ids} n = len(used) print(f"\nTasks by how many of the {n} runs resolved them") print(" resolved by: " + " ".join(f"{j:>3}" for j in range(n + 1))) print(" tasks: " + " ".join(f"{sum(1 for i in ids if k[i] == j):>3}" for j in range(n + 1))) for b in DIFFICULTY: t = labels[b] print(f" {b:16} {len(t):3} tasks: all {n} runs {sum(1 for i in t if k[i] == n):3}, none " f"{sum(1 for i in t if k[i] == 0):3}, some {sum(1 for i in t if 0 < k[i] < n):3}") print(f" resolved by at least one run: {sum(1 for i in ids if k[i])}; by the best run alone: " f"{max(sum(v['resolved'] for v in used[m].values()) for m in used)}") # ---- what a cap on model calls would have kept and saved ---- print("\nA cap on model calls, applied to the recorded attempts") print(" cap resolves kept calls saved kept, lowest and highest run") def cap_row(label, per): kk, kr = sum(p[0] for p in per), sum(p[1] for p in per) ss, st = sum(p[2] for p in per), sum(p[3] for p in per) lo, hi = min(p[0] / p[1] for p in per), max(p[0] / p[1] for p in per) print(f" {label:30} {kk:5,} {pct(kk, kr):>7} {pct(ss, st):>9} {100 * lo:.1f}% to {100 * hi:.1f}%") for cap in [int(c) for c in a.caps.split(",")]: cap_row(f"{cap} calls", [kept_and_saved(used[m], cap) for m in used]) for mult in (1.5, 2, 3): per = [] for m in used: med = statistics.median(v["api_calls"] for v in used[m].values() if v["resolved"]) per.append(kept_and_saved(used[m], mult * med)) cap_row(f"{mult:g} x the run's median resolve", per) short = fails = 0 for m in used: med = statistics.median(v["api_calls"] for v in used[m].values() if v["resolved"]) f = [v["api_calls"] for v in used[m].values() if not v["resolved"]] fails += len(f) short += sum(1 for c in f if c <= med) ratios = [] for m in used: cr = statistics.median(v["api_calls"] for v in used[m].values() if v["resolved"]) cf = statistics.median(v["api_calls"] for v in used[m].values() if not v["resolved"]) ratios.append(cf / cr) stopped = sum(1 for v in allr if v["api_calls"] >= 250 or v.get("cost", 0.0) >= 3.0) print(f" median calls, unresolved attempt / resolved attempt, by run: {min(ratios):.2f}x to {max(ratios):.2f}x") print(f" unresolved attempts no longer than their run's median resolved attempt: {short:,} of {fails:,} {pct(short, fails)}") print(f" attempts that reached 250 calls or $3.00 (the limits in mini-swe-agent 2.0.0's SWE-bench config): {stopped}" f", resolved {sum(1 for v in allr if (v['api_calls'] >= 250 or v.get('cost', 0.0) >= 3.0) and v['resolved'])}") shares = [sum(v["cost"] for v in used[m].values() if not v["resolved"]) / sum(v["cost"] for v in used[m].values()) for m in used] print(f" share of recorded cost spent on unresolved tasks: {100 * min(shares):.1f}% to {100 * max(shares):.1f}% by run, " f"{pct(sum(v['cost'] for v in allr if not v['resolved']), tot_cost)} over all attempts") # ---- every entry with a per-task record ---- ever = {i: sum(1 for e in with_rec.values() if i in e["resolved"]) for i in ids} never = [i for i in ids if ever[i] == 0] top = max(len(e["resolved"] & set(ids)) for e in with_rec.values()) best = [d for d, e in with_rec.items() if len(e["resolved"] & set(ids)) == top] print(f"\nAll {len(with_rec)} entries with a per-task record") print(f" most resolved by one entry: {top} ({', '.join(best)})") print(f" tasks resolved by at least one entry: {500 - len(never)}; by exactly one: " f"{sum(1 for i in ids if ever[i] == 1)}; by none: {len(never)}") print(" never resolved, by difficulty: " + ", ".join( f"{b} {sum(1 for i in never if tasks[i]['difficulty'] == b)}" for b in DIFFICULTY)) repos = {} for i in never: repos[tasks[i]["repo"]] = repos.get(tasks[i]["repo"], 0) + 1 print(" never resolved, by repository: " + ", ".join(f"{r} {c}" for r, c in sorted(repos.items(), key=lambda x: (-x[1], x[0])))) print(" never resolved, <15 min fix: " + ", ".join(i for i in never if tasks[i]["difficulty"] == "<15 min fix")) if a.csv: with open(a.csv, "w", encoding="utf-8", newline="") as f: wr = csv.writer(f, lineterminator="\n") wr.writerow(["instance_id", "repo", "difficulty", "resolved_by_runs", "runs", "resolved_by_entries", "entries", "median_calls", "median_cost_usd"]) for i in ids: calls = [used[m][i]["api_calls"] for m in used] costs = [used[m][i]["cost"] for m in used] wr.writerow([i, tasks[i]["repo"], tasks[i]["difficulty"], k[i], n, ever[i], len(with_rec), f"{statistics.median(calls):g}", f"{statistics.median(costs):.4f}"]) print(f"\nwrote {a.csv} ({len(ids)} rows)") return 0 if __name__ == "__main__": sys.exit(main())