Empirical filter runner (DeepSeek); real cost ratio in docs
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat
This commit is contained in:
80
harness/empirical.py
Normal file
80
harness/empirical.py
Normal file
@@ -0,0 +1,80 @@
|
||||
"""Empirical filter (step F, second layer): run accepted eval tasks with real models.
|
||||
|
||||
A task where a strong model fails although the reference passes can have an unclear spec (flag it).
|
||||
A task that every model passes with ease does not separate models (candidate for removal).
|
||||
|
||||
python3 -m harness.empirical --model deepseek-v4.1-flash:cloud --run-base 10000 [G0100 ...]
|
||||
|
||||
Results: runs/emp/results.jsonl and <task>/empirical.json (one entry per model). Resumable: a task that
|
||||
already has a result for the model is skipped. Tasks with review decision "reject" are skipped.
|
||||
"""
|
||||
import argparse
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
|
||||
from .adt_client import load_env
|
||||
from .agents import LlmAgent
|
||||
from .ledger import BudgetExceeded
|
||||
from .runner import Runner
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
POOL = os.path.join(ROOT, "tasks_gen", "eval")
|
||||
|
||||
|
||||
def _json(path, default):
|
||||
try:
|
||||
return json.load(open(path))
|
||||
except (OSError, ValueError):
|
||||
return default
|
||||
|
||||
|
||||
def candidates():
|
||||
out = []
|
||||
for d in sorted(glob.glob(os.path.join(POOL, "G*"))):
|
||||
if not _json(os.path.join(d, "generation.json"), {}).get("accepted"):
|
||||
continue
|
||||
if _json(os.path.join(d, "review.json"), {}).get("decision") == "reject":
|
||||
continue
|
||||
out.append(os.path.basename(d))
|
||||
return out
|
||||
|
||||
|
||||
def main():
|
||||
load_env(os.path.join(ROOT, ".env"))
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("tasks", nargs="*")
|
||||
ap.add_argument("--model", required=True)
|
||||
ap.add_argument("--base-url")
|
||||
ap.add_argument("--run-base", type=int, required=True)
|
||||
a = ap.parse_args()
|
||||
runs_root = os.path.join(ROOT, "runs", "emp")
|
||||
os.makedirs(runs_root, exist_ok=True)
|
||||
runner = Runner(POOL, runs_root)
|
||||
tasks = a.tasks or candidates()
|
||||
for i, tid in enumerate(tasks):
|
||||
path = os.path.join(POOL, tid, "empirical.json")
|
||||
res = _json(path, {})
|
||||
if a.model in res:
|
||||
continue
|
||||
try:
|
||||
rep, run_dir = runner.run(tid, LlmAgent(a.model, a.base_url), a.run_base + i)
|
||||
except BudgetExceeded as e:
|
||||
print("BUDGET", e, flush=True)
|
||||
break
|
||||
except Exception as e: # noqa: BLE001 one broken run must not stop the series
|
||||
print(json.dumps({"task": tid, "error": str(e)[:300]}), flush=True)
|
||||
continue
|
||||
h = rep.get("hidden_tests") or {}
|
||||
entry = {"score": (rep.get("score") or {}).get("total"), "parts": rep.get("score"),
|
||||
"hidden": f"{h.get('passed')}/{h.get('total')}", "tool_calls": rep.get("tool_calls"),
|
||||
"seconds": rep.get("seconds"), "run_dir": os.path.relpath(run_dir, ROOT)}
|
||||
res[a.model] = entry
|
||||
json.dump(res, open(path, "w"), indent=1)
|
||||
line = dict(task=tid, model=a.model, **{k: entry[k] for k in ("score", "hidden", "tool_calls")})
|
||||
open(os.path.join(runs_root, "results.jsonl"), "a").write(json.dumps(line) + "\n")
|
||||
print(json.dumps(line), flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user