From c062df2ec5145235d665fadae8b0b733dad79fad Mon Sep 17 00:00:00 2001 From: Kral Date: Mon, 5 Oct 2026 14:08:33 +0200 Subject: [PATCH] Trajectory runs: tool-call budget 100 for tasks with a CDS contract object (eval keeps 60) Co-Authored-By: Claude Sonnet 5.5 --- harness/record.py | 2 +- harness/runner.py | 8 +++++++- harness/trajectories.py | 3 ++- train/STATE.md | 2 ++ 4 files changed, 12 insertions(+), 3 deletions(-) diff --git a/harness/record.py b/harness/record.py index 7369dcc..1de3117 100644 --- a/harness/record.py +++ b/harness/record.py @@ -56,7 +56,7 @@ def write_record(run_dir, task, agent, rep, pool): "cost_ledger_usd": cost, "cost_usage_usd": round(cost / LEDGER_TO_USAGE, 5), "end_reason": rep.get("end_reason"), "tool_calls": rep.get("tool_calls"), "activations": rep.get("activations"), "activation_failures": rep.get("activation_failures"), - "max_fail_streak": rep.get("max_fail_streak"), "syntax_hints": rep.get("syntax_hints"), + "max_fail_streak": rep.get("max_fail_streak"), "tool_budget": rep.get("tool_budget"), "syntax_hints": rep.get("syntax_hints"), "adt_fallbacks": bool(rep.get("adt_fallbacks")), "agent_seconds": rep.get("agent_seconds"), "turn_usage": getattr(agent, "turn_usage", []), "setup_failed": rep.get("setup_failed", False), diff --git a/harness/runner.py b/harness/runner.py index b08f251..31d53e3 100644 --- a/harness/runner.py +++ b/harness/runner.py @@ -199,9 +199,14 @@ class Runner: "line": i.get("start", {}).get("row"), "msg": i.get("description")} for i in issues] # ---------- main ---------- - def run(self, task_id, agent, run_no, teardown=True, rescore_dir=None): + def run(self, task_id, agent, run_no, teardown=True, rescore_dir=None, cds_calls=None): + """cds_calls: raise max_tool_calls to this value for tasks with a CDS (DDLS) contract object + (trajectory runs, 2026-10-05: the own CDS test class needs more than 60 calls; eval stays at the task value).""" prefix = prefix_for(run_no, task_id) task = Task(os.path.join(self.tasks_root, task_id), prefix) + if cds_calls and any(c.get("type") == "DDLS" for c in task.meta.get("contract", [])): + b = task.meta.setdefault("budget", {}) + b["max_tool_calls"] = max(b.get("max_tool_calls", 60), cds_calls) run_dir = rescore_dir or os.path.join(self.runs_root, f"{int(run_no):03d}_{task_id}_{agent.name.replace(':', '_').replace('/', '_')}") os.makedirs(run_dir, exist_ok=True) rep = {"task": task_id, "agent": agent.name, "run": run_no, "prefix": prefix, @@ -241,6 +246,7 @@ class Runner: rep["activation_failures"] = proxy.activation_failures rep["activation_error_messages"] = proxy.activation_errors rep["syntax_hints"] = proxy.syntax_hints + rep["tool_budget"] = proxy.max_calls rep["end_reason"] = getattr(agent, "end_reason", None) if proxy.fallbacks: rep["adt_fallbacks"] = proxy.fallbacks diff --git a/harness/trajectories.py b/harness/trajectories.py index fc4a310..b2ac32f 100644 --- a/harness/trajectories.py +++ b/harness/trajectories.py @@ -26,6 +26,7 @@ POOL = os.path.join(ROOT, "tasks_gen", "train") OUT = os.path.join(ROOT, "runs", "traj") RUN_BASE = 200000 # 200000 + (task number - 1000) * 3 + attempt (a digit must lead the 4-char base36 run: < 466560) MODEL = "deepseek-v4.1-flash:cloud" +CDS_CALLS = 100 # tool-call budget for tasks with a CDS contract object (eval keeps 60; Kral 2026-10-05) LOCK = threading.Lock() @@ -76,7 +77,7 @@ def one(task_id, attempt, stop): t0 = time.time() row = {"task": task_id, "attempt": attempt, "run": run_no, "model": MODEL} try: - rep, run_dir = runner.run(task_id, agent, run_no, teardown=True) + rep, run_dir = runner.run(task_id, agent, run_no, teardown=True, cds_calls=CDS_CALLS) score = (rep.get("score") or {}).get("total") rec = os.path.exists(os.path.join(run_dir, "record.json")) row.update(score=score, setup_failed=bool(rep.get("setup_failed")), end_reason=rep.get("end_reason"), diff --git a/train/STATE.md b/train/STATE.md index cfe31ac..499a3d5 100644 --- a/train/STATE.md +++ b/train/STATE.md @@ -140,3 +140,5 @@ Training runs on HF Jobs with Unsloth, not on the Mac. No `mlx_lm` training. round-trip check of every tool call; assistant spans for loss masking). B2d filter: `train/accept.py`. - Test: scripted fake model on T01 (A4H, no cloud): score 100, 1 syntax hint, record, filter and converter OK. - B3 trajectory runner: `harness/trajectories.py` (6 workers, DeepSeek V4.1 Flash, output `runs/traj/`). Starts after generation. + +- 2026-10-05 Kral: tool-call budget 100 (eval stays 60) for trajectory runs on tasks with a CDS contract object (`Runner.run(cds_calls=100)`, `trajectories.CDS_CALLS`). Reason: first 3 CDS runs were correct (hidden tests all passed) but ended at 60 calls or by an empty response while writing the own CDS test class (score 80, not accepted). `metadata.tool_budget` is in every record. Failed CDS tasks get the second attempt with the new budget.