"""Empirical filter (step F, second layer): run accepted eval tasks with real models. A task where a strong model fails although the reference passes can have an unclear spec (flag it). A task that every model passes with ease does not separate models (candidate for removal). python3 -m harness.empirical --model deepseek-v4.1-flash:cloud --run-base 10000 [G0100 ...] Results: runs/emp/results.jsonl and /empirical.json (one entry per model). Resumable: a task that already has a result for the model is skipped. Tasks with review decision "reject" are skipped. """ import argparse import glob import json import os from .adt_client import load_env from .agents import LlmAgent from .ledger import BudgetExceeded from .runner import Runner ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) POOL = os.path.join(ROOT, "tasks_gen", "eval") def _json(path, default): try: return json.load(open(path)) except (OSError, ValueError): return default def candidates(): out = [] for d in sorted(glob.glob(os.path.join(POOL, "G*"))): if not _json(os.path.join(d, "generation.json"), {}).get("accepted"): continue if _json(os.path.join(d, "review.json"), {}).get("decision") == "reject": continue out.append(os.path.basename(d)) return out def decision(score, calls): """The filter decisions of step F: an easy candidate (DeepSeek >= 95 and <= 20 tool calls, tasks_gen/eval/easy_candidates.json), a flag (a strong model fails, score under 50, although the reference passes: the spec may be unclear), else normal.""" if score is None: return "no result" if score >= 95 and (calls or 99) <= 20: return "easy candidate" return "flag: strong model fails" if score < 50 else "normal" def rerun(a): """Rerun some tasks (after the proxy fix of 2026-10-06) WITHOUT touching empirical.json or the eval set: results go to runs/emp_rerun/, then the old and the new filter decision are printed. A changed decision is reported to Kral and Opus before anything in the eval set changes.""" out_dir = os.path.join(ROOT, "runs", "emp_rerun") os.makedirs(out_dir, exist_ok=True) runner = Runner(POOL, out_dir) changed = [] for i, tid in enumerate(a.tasks): old = _json(os.path.join(POOL, tid, "empirical.json"), {}).get(a.model, {}) try: rep, run_dir = runner.run(tid, LlmAgent(a.model, a.base_url), a.run_base + i) except BudgetExceeded as e: print("BUDGET", e, flush=True) break h = rep.get("hidden_tests") or {} new = {"score": (rep.get("score") or {}).get("total"), "hidden": f"{h.get('passed')}/{h.get('total')}", "tool_calls": rep.get("tool_calls"), "end_reason": rep.get("end_reason"), "run_dir": os.path.relpath(run_dir, ROOT)} d_old, d_new = decision(old.get("score"), old.get("tool_calls")), decision(new["score"], new["tool_calls"]) line = {"task": tid, "old": {k: old.get(k) for k in ("score", "hidden", "tool_calls")}, "new": new, "decision_old": d_old, "decision_new": d_new, "decision_changed": d_old != d_new} open(os.path.join(out_dir, "results.jsonl"), "a").write(json.dumps(line) + "\n") print(json.dumps(line), flush=True) if line["decision_changed"]: changed.append(tid) print("DECISION CHANGED for:", changed or "none", "(nothing in tasks_gen/eval was changed)", flush=True) def main(): load_env(os.path.join(ROOT, ".env")) ap = argparse.ArgumentParser() ap.add_argument("tasks", nargs="*") ap.add_argument("--model", required=True) ap.add_argument("--base-url") ap.add_argument("--run-base", type=int, required=True) ap.add_argument("--rerun", action="store_true", help="rerun the named tasks into runs/emp_rerun/ and compare the filter decision; the eval set is not changed") a = ap.parse_args() if a.rerun: if not a.tasks: raise SystemExit("--rerun needs task ids") return rerun(a) runs_root = os.path.join(ROOT, "runs", "emp") os.makedirs(runs_root, exist_ok=True) runner = Runner(POOL, runs_root) tasks = a.tasks or candidates() for i, tid in enumerate(tasks): path = os.path.join(POOL, tid, "empirical.json") res = _json(path, {}) if a.model in res: continue try: rep, run_dir = runner.run(tid, LlmAgent(a.model, a.base_url), a.run_base + i) except BudgetExceeded as e: print("BUDGET", e, flush=True) break except Exception as e: # noqa: BLE001 one broken run must not stop the series print(json.dumps({"task": tid, "error": str(e)[:300]}), flush=True) continue h = rep.get("hidden_tests") or {} entry = {"score": (rep.get("score") or {}).get("total"), "parts": rep.get("score"), "hidden": f"{h.get('passed')}/{h.get('total')}", "tool_calls": rep.get("tool_calls"), "seconds": rep.get("seconds"), "run_dir": os.path.relpath(run_dir, ROOT)} res[a.model] = entry json.dump(res, open(path, "w"), indent=1) line = dict(task=tid, model=a.model, **{k: entry[k] for k in ("score", "hidden", "tool_calls")}) open(os.path.join(runs_root, "results.jsonl"), "a").write(json.dumps(line) + "\n") print(json.dumps(line), flush=True) if __name__ == "__main__": main()