diff --git a/docs/restart-12-oktober.md b/docs/restart-12-oktober.md index e137d6c..88f165e 100644 --- a/docs/restart-12-oktober.md +++ b/docs/restart-12-oktober.md @@ -23,6 +23,17 @@ Ollama reset on 12 October. Until then only no-cloud work. This is the plan for or exception task, so the trained model could not be measured on them. Each slot is checked against the training pool. Review: `docs/eval-spotcheck-new-kinds.md` (Kral checks about 10 tasks). Run it before the training tasks so that the training tasks can be checked against the finished eval tasks. +## 1c. Step 0b: rerun of four empirical filter runs (Kral + Opus 2026-10-06) + +Four eval candidates saw or read leftover objects of other runs in the first empirical filter (`docs/foreign-objects-report.md`): **G0183, G0180, G0143, G0185** (DeepSeek, scores 80, 85, 85, 85). +After step 0 (eval generation for the new kinds), with the fixed proxy: + +`python3 -m harness.empirical --model deepseek-v4.1-flash:cloud --run-base 450000 --rerun G0183 G0180 G0143 G0185` + +This writes only to `runs/emp_rerun/` (4 runs, about 1 ledger) and prints old and new filter decision per task. `empirical.json` and the eval set are **not** touched. Decision rules (step F): +easy candidate = score 95 or more and at most 20 tool calls; flag = score under 50 (a strong model fails although the reference passes); else normal. **If a decision changes for one of the four, report it to Kral and +Opus first; the eval set is changed only after their answer.** If nothing changes, say so in `docs/eval-inceleme.md` and leave the old results. + ## 2. Order of work (what the controller does by itself) Only kinds below their target share are generated and run (`harness/mix.py`, `below_target`); the kind with the biggest diff --git a/harness/empirical.py b/harness/empirical.py index 0b84cca..3618cdd 100644 --- a/harness/empirical.py +++ b/harness/empirical.py @@ -40,6 +40,43 @@ def candidates(): return out +def decision(score, calls): + """The filter decisions of step F: an easy candidate (DeepSeek >= 95 and <= 20 tool calls, tasks_gen/eval/easy_candidates.json), + a flag (a strong model fails, score under 50, although the reference passes: the spec may be unclear), else normal.""" + if score is None: + return "no result" + if score >= 95 and (calls or 99) <= 20: + return "easy candidate" + return "flag: strong model fails" if score < 50 else "normal" + + +def rerun(a): + """Rerun some tasks (after the proxy fix of 2026-10-06) WITHOUT touching empirical.json or the eval set: results go to runs/emp_rerun/, + then the old and the new filter decision are printed. A changed decision is reported to Kral and Opus before anything in the eval set changes.""" + out_dir = os.path.join(ROOT, "runs", "emp_rerun") + os.makedirs(out_dir, exist_ok=True) + runner = Runner(POOL, out_dir) + changed = [] + for i, tid in enumerate(a.tasks): + old = _json(os.path.join(POOL, tid, "empirical.json"), {}).get(a.model, {}) + try: + rep, run_dir = runner.run(tid, LlmAgent(a.model, a.base_url), a.run_base + i) + except BudgetExceeded as e: + print("BUDGET", e, flush=True) + break + h = rep.get("hidden_tests") or {} + new = {"score": (rep.get("score") or {}).get("total"), "hidden": f"{h.get('passed')}/{h.get('total')}", "tool_calls": rep.get("tool_calls"), + "end_reason": rep.get("end_reason"), "run_dir": os.path.relpath(run_dir, ROOT)} + d_old, d_new = decision(old.get("score"), old.get("tool_calls")), decision(new["score"], new["tool_calls"]) + line = {"task": tid, "old": {k: old.get(k) for k in ("score", "hidden", "tool_calls")}, "new": new, "decision_old": d_old, "decision_new": d_new, + "decision_changed": d_old != d_new} + open(os.path.join(out_dir, "results.jsonl"), "a").write(json.dumps(line) + "\n") + print(json.dumps(line), flush=True) + if line["decision_changed"]: + changed.append(tid) + print("DECISION CHANGED for:", changed or "none", "(nothing in tasks_gen/eval was changed)", flush=True) + + def main(): load_env(os.path.join(ROOT, ".env")) ap = argparse.ArgumentParser() @@ -47,7 +84,12 @@ def main(): ap.add_argument("--model", required=True) ap.add_argument("--base-url") ap.add_argument("--run-base", type=int, required=True) + ap.add_argument("--rerun", action="store_true", help="rerun the named tasks into runs/emp_rerun/ and compare the filter decision; the eval set is not changed") a = ap.parse_args() + if a.rerun: + if not a.tasks: + raise SystemExit("--rerun needs task ids") + return rerun(a) runs_root = os.path.join(ROOT, "runs", "emp") os.makedirs(runs_root, exist_ok=True) runner = Runner(POOL, runs_root)