Restart plan step 0b: rerun of the four empirical filter runs (G0183 G0180 G0143 G0185) without touching the eval set; empirical --rerun

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-06 09:07:10 +02:00
parent 8ef85c2713
commit 4261054eac
2 changed files with 53 additions and 0 deletions

View File

@@ -23,6 +23,17 @@ Ollama reset on 12 October. Until then only no-cloud work. This is the plan for
or exception task, so the trained model could not be measured on them. Each slot is checked against the training pool. Review: `docs/eval-spotcheck-new-kinds.md` or exception task, so the trained model could not be measured on them. Each slot is checked against the training pool. Review: `docs/eval-spotcheck-new-kinds.md`
(Kral checks about 10 tasks). Run it before the training tasks so that the training tasks can be checked against the finished eval tasks. (Kral checks about 10 tasks). Run it before the training tasks so that the training tasks can be checked against the finished eval tasks.
## 1c. Step 0b: rerun of four empirical filter runs (Kral + Opus 2026-10-06)
Four eval candidates saw or read leftover objects of other runs in the first empirical filter (`docs/foreign-objects-report.md`): **G0183, G0180, G0143, G0185** (DeepSeek, scores 80, 85, 85, 85).
After step 0 (eval generation for the new kinds), with the fixed proxy:
`python3 -m harness.empirical --model deepseek-v4.1-flash:cloud --run-base 450000 --rerun G0183 G0180 G0143 G0185`
This writes only to `runs/emp_rerun/` (4 runs, about 1 ledger) and prints old and new filter decision per task. `empirical.json` and the eval set are **not** touched. Decision rules (step F):
easy candidate = score 95 or more and at most 20 tool calls; flag = score under 50 (a strong model fails although the reference passes); else normal. **If a decision changes for one of the four, report it to Kral and
Opus first; the eval set is changed only after their answer.** If nothing changes, say so in `docs/eval-inceleme.md` and leave the old results.
## 2. Order of work (what the controller does by itself) ## 2. Order of work (what the controller does by itself)
Only kinds below their target share are generated and run (`harness/mix.py`, `below_target`); the kind with the biggest Only kinds below their target share are generated and run (`harness/mix.py`, `below_target`); the kind with the biggest

View File

@@ -40,6 +40,43 @@ def candidates():
return out return out
def decision(score, calls):
"""The filter decisions of step F: an easy candidate (DeepSeek >= 95 and <= 20 tool calls, tasks_gen/eval/easy_candidates.json),
a flag (a strong model fails, score under 50, although the reference passes: the spec may be unclear), else normal."""
if score is None:
return "no result"
if score >= 95 and (calls or 99) <= 20:
return "easy candidate"
return "flag: strong model fails" if score < 50 else "normal"
def rerun(a):
"""Rerun some tasks (after the proxy fix of 2026-10-06) WITHOUT touching empirical.json or the eval set: results go to runs/emp_rerun/,
then the old and the new filter decision are printed. A changed decision is reported to Kral and Opus before anything in the eval set changes."""
out_dir = os.path.join(ROOT, "runs", "emp_rerun")
os.makedirs(out_dir, exist_ok=True)
runner = Runner(POOL, out_dir)
changed = []
for i, tid in enumerate(a.tasks):
old = _json(os.path.join(POOL, tid, "empirical.json"), {}).get(a.model, {})
try:
rep, run_dir = runner.run(tid, LlmAgent(a.model, a.base_url), a.run_base + i)
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
break
h = rep.get("hidden_tests") or {}
new = {"score": (rep.get("score") or {}).get("total"), "hidden": f"{h.get('passed')}/{h.get('total')}", "tool_calls": rep.get("tool_calls"),
"end_reason": rep.get("end_reason"), "run_dir": os.path.relpath(run_dir, ROOT)}
d_old, d_new = decision(old.get("score"), old.get("tool_calls")), decision(new["score"], new["tool_calls"])
line = {"task": tid, "old": {k: old.get(k) for k in ("score", "hidden", "tool_calls")}, "new": new, "decision_old": d_old, "decision_new": d_new,
"decision_changed": d_old != d_new}
open(os.path.join(out_dir, "results.jsonl"), "a").write(json.dumps(line) + "\n")
print(json.dumps(line), flush=True)
if line["decision_changed"]:
changed.append(tid)
print("DECISION CHANGED for:", changed or "none", "(nothing in tasks_gen/eval was changed)", flush=True)
def main(): def main():
load_env(os.path.join(ROOT, ".env")) load_env(os.path.join(ROOT, ".env"))
ap = argparse.ArgumentParser() ap = argparse.ArgumentParser()
@@ -47,7 +84,12 @@ def main():
ap.add_argument("--model", required=True) ap.add_argument("--model", required=True)
ap.add_argument("--base-url") ap.add_argument("--base-url")
ap.add_argument("--run-base", type=int, required=True) ap.add_argument("--run-base", type=int, required=True)
ap.add_argument("--rerun", action="store_true", help="rerun the named tasks into runs/emp_rerun/ and compare the filter decision; the eval set is not changed")
a = ap.parse_args() a = ap.parse_args()
if a.rerun:
if not a.tasks:
raise SystemExit("--rerun needs task ids")
return rerun(a)
runs_root = os.path.join(ROOT, "runs", "emp") runs_root = os.path.join(ROOT, "runs", "emp")
os.makedirs(runs_root, exist_ok=True) os.makedirs(runs_root, exist_ok=True)
runner = Runner(POOL, runs_root) runner = Runner(POOL, runs_root)