Restart plan step 0b: rerun of the four empirical filter runs (G0183 G0180 G0143 G0185) without touching the eval set; empirical --rerun
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -40,6 +40,43 @@ def candidates():
|
||||
return out
|
||||
|
||||
|
||||
def decision(score, calls):
|
||||
"""The filter decisions of step F: an easy candidate (DeepSeek >= 95 and <= 20 tool calls, tasks_gen/eval/easy_candidates.json),
|
||||
a flag (a strong model fails, score under 50, although the reference passes: the spec may be unclear), else normal."""
|
||||
if score is None:
|
||||
return "no result"
|
||||
if score >= 95 and (calls or 99) <= 20:
|
||||
return "easy candidate"
|
||||
return "flag: strong model fails" if score < 50 else "normal"
|
||||
|
||||
|
||||
def rerun(a):
|
||||
"""Rerun some tasks (after the proxy fix of 2026-10-06) WITHOUT touching empirical.json or the eval set: results go to runs/emp_rerun/,
|
||||
then the old and the new filter decision are printed. A changed decision is reported to Kral and Opus before anything in the eval set changes."""
|
||||
out_dir = os.path.join(ROOT, "runs", "emp_rerun")
|
||||
os.makedirs(out_dir, exist_ok=True)
|
||||
runner = Runner(POOL, out_dir)
|
||||
changed = []
|
||||
for i, tid in enumerate(a.tasks):
|
||||
old = _json(os.path.join(POOL, tid, "empirical.json"), {}).get(a.model, {})
|
||||
try:
|
||||
rep, run_dir = runner.run(tid, LlmAgent(a.model, a.base_url), a.run_base + i)
|
||||
except BudgetExceeded as e:
|
||||
print("BUDGET", e, flush=True)
|
||||
break
|
||||
h = rep.get("hidden_tests") or {}
|
||||
new = {"score": (rep.get("score") or {}).get("total"), "hidden": f"{h.get('passed')}/{h.get('total')}", "tool_calls": rep.get("tool_calls"),
|
||||
"end_reason": rep.get("end_reason"), "run_dir": os.path.relpath(run_dir, ROOT)}
|
||||
d_old, d_new = decision(old.get("score"), old.get("tool_calls")), decision(new["score"], new["tool_calls"])
|
||||
line = {"task": tid, "old": {k: old.get(k) for k in ("score", "hidden", "tool_calls")}, "new": new, "decision_old": d_old, "decision_new": d_new,
|
||||
"decision_changed": d_old != d_new}
|
||||
open(os.path.join(out_dir, "results.jsonl"), "a").write(json.dumps(line) + "\n")
|
||||
print(json.dumps(line), flush=True)
|
||||
if line["decision_changed"]:
|
||||
changed.append(tid)
|
||||
print("DECISION CHANGED for:", changed or "none", "(nothing in tasks_gen/eval was changed)", flush=True)
|
||||
|
||||
|
||||
def main():
|
||||
load_env(os.path.join(ROOT, ".env"))
|
||||
ap = argparse.ArgumentParser()
|
||||
@@ -47,7 +84,12 @@ def main():
|
||||
ap.add_argument("--model", required=True)
|
||||
ap.add_argument("--base-url")
|
||||
ap.add_argument("--run-base", type=int, required=True)
|
||||
ap.add_argument("--rerun", action="store_true", help="rerun the named tasks into runs/emp_rerun/ and compare the filter decision; the eval set is not changed")
|
||||
a = ap.parse_args()
|
||||
if a.rerun:
|
||||
if not a.tasks:
|
||||
raise SystemExit("--rerun needs task ids")
|
||||
return rerun(a)
|
||||
runs_root = os.path.join(ROOT, "runs", "emp")
|
||||
os.makedirs(runs_root, exist_ok=True)
|
||||
runner = Runner(POOL, runs_root)
|
||||
|
||||
Reference in New Issue
Block a user