Trajectory runs: tool-call budget 100 for tasks with a CDS contract object (eval keeps 60)
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -56,7 +56,7 @@ def write_record(run_dir, task, agent, rep, pool):
|
|||||||
"cost_ledger_usd": cost, "cost_usage_usd": round(cost / LEDGER_TO_USAGE, 5),
|
"cost_ledger_usd": cost, "cost_usage_usd": round(cost / LEDGER_TO_USAGE, 5),
|
||||||
"end_reason": rep.get("end_reason"), "tool_calls": rep.get("tool_calls"),
|
"end_reason": rep.get("end_reason"), "tool_calls": rep.get("tool_calls"),
|
||||||
"activations": rep.get("activations"), "activation_failures": rep.get("activation_failures"),
|
"activations": rep.get("activations"), "activation_failures": rep.get("activation_failures"),
|
||||||
"max_fail_streak": rep.get("max_fail_streak"), "syntax_hints": rep.get("syntax_hints"),
|
"max_fail_streak": rep.get("max_fail_streak"), "tool_budget": rep.get("tool_budget"), "syntax_hints": rep.get("syntax_hints"),
|
||||||
"adt_fallbacks": bool(rep.get("adt_fallbacks")), "agent_seconds": rep.get("agent_seconds"),
|
"adt_fallbacks": bool(rep.get("adt_fallbacks")), "agent_seconds": rep.get("agent_seconds"),
|
||||||
"turn_usage": getattr(agent, "turn_usage", []),
|
"turn_usage": getattr(agent, "turn_usage", []),
|
||||||
"setup_failed": rep.get("setup_failed", False),
|
"setup_failed": rep.get("setup_failed", False),
|
||||||
|
|||||||
@@ -199,9 +199,14 @@ class Runner:
|
|||||||
"line": i.get("start", {}).get("row"), "msg": i.get("description")} for i in issues]
|
"line": i.get("start", {}).get("row"), "msg": i.get("description")} for i in issues]
|
||||||
|
|
||||||
# ---------- main ----------
|
# ---------- main ----------
|
||||||
def run(self, task_id, agent, run_no, teardown=True, rescore_dir=None):
|
def run(self, task_id, agent, run_no, teardown=True, rescore_dir=None, cds_calls=None):
|
||||||
|
"""cds_calls: raise max_tool_calls to this value for tasks with a CDS (DDLS) contract object
|
||||||
|
(trajectory runs, 2026-10-05: the own CDS test class needs more than 60 calls; eval stays at the task value)."""
|
||||||
prefix = prefix_for(run_no, task_id)
|
prefix = prefix_for(run_no, task_id)
|
||||||
task = Task(os.path.join(self.tasks_root, task_id), prefix)
|
task = Task(os.path.join(self.tasks_root, task_id), prefix)
|
||||||
|
if cds_calls and any(c.get("type") == "DDLS" for c in task.meta.get("contract", [])):
|
||||||
|
b = task.meta.setdefault("budget", {})
|
||||||
|
b["max_tool_calls"] = max(b.get("max_tool_calls", 60), cds_calls)
|
||||||
run_dir = rescore_dir or os.path.join(self.runs_root, f"{int(run_no):03d}_{task_id}_{agent.name.replace(':', '_').replace('/', '_')}")
|
run_dir = rescore_dir or os.path.join(self.runs_root, f"{int(run_no):03d}_{task_id}_{agent.name.replace(':', '_').replace('/', '_')}")
|
||||||
os.makedirs(run_dir, exist_ok=True)
|
os.makedirs(run_dir, exist_ok=True)
|
||||||
rep = {"task": task_id, "agent": agent.name, "run": run_no, "prefix": prefix,
|
rep = {"task": task_id, "agent": agent.name, "run": run_no, "prefix": prefix,
|
||||||
@@ -241,6 +246,7 @@ class Runner:
|
|||||||
rep["activation_failures"] = proxy.activation_failures
|
rep["activation_failures"] = proxy.activation_failures
|
||||||
rep["activation_error_messages"] = proxy.activation_errors
|
rep["activation_error_messages"] = proxy.activation_errors
|
||||||
rep["syntax_hints"] = proxy.syntax_hints
|
rep["syntax_hints"] = proxy.syntax_hints
|
||||||
|
rep["tool_budget"] = proxy.max_calls
|
||||||
rep["end_reason"] = getattr(agent, "end_reason", None)
|
rep["end_reason"] = getattr(agent, "end_reason", None)
|
||||||
if proxy.fallbacks:
|
if proxy.fallbacks:
|
||||||
rep["adt_fallbacks"] = proxy.fallbacks
|
rep["adt_fallbacks"] = proxy.fallbacks
|
||||||
|
|||||||
@@ -26,6 +26,7 @@ POOL = os.path.join(ROOT, "tasks_gen", "train")
|
|||||||
OUT = os.path.join(ROOT, "runs", "traj")
|
OUT = os.path.join(ROOT, "runs", "traj")
|
||||||
RUN_BASE = 200000 # 200000 + (task number - 1000) * 3 + attempt (a digit must lead the 4-char base36 run: < 466560)
|
RUN_BASE = 200000 # 200000 + (task number - 1000) * 3 + attempt (a digit must lead the 4-char base36 run: < 466560)
|
||||||
MODEL = "deepseek-v4.1-flash:cloud"
|
MODEL = "deepseek-v4.1-flash:cloud"
|
||||||
|
CDS_CALLS = 100 # tool-call budget for tasks with a CDS contract object (eval keeps 60; Kral 2026-10-05)
|
||||||
LOCK = threading.Lock()
|
LOCK = threading.Lock()
|
||||||
|
|
||||||
|
|
||||||
@@ -76,7 +77,7 @@ def one(task_id, attempt, stop):
|
|||||||
t0 = time.time()
|
t0 = time.time()
|
||||||
row = {"task": task_id, "attempt": attempt, "run": run_no, "model": MODEL}
|
row = {"task": task_id, "attempt": attempt, "run": run_no, "model": MODEL}
|
||||||
try:
|
try:
|
||||||
rep, run_dir = runner.run(task_id, agent, run_no, teardown=True)
|
rep, run_dir = runner.run(task_id, agent, run_no, teardown=True, cds_calls=CDS_CALLS)
|
||||||
score = (rep.get("score") or {}).get("total")
|
score = (rep.get("score") or {}).get("total")
|
||||||
rec = os.path.exists(os.path.join(run_dir, "record.json"))
|
rec = os.path.exists(os.path.join(run_dir, "record.json"))
|
||||||
row.update(score=score, setup_failed=bool(rep.get("setup_failed")), end_reason=rep.get("end_reason"),
|
row.update(score=score, setup_failed=bool(rep.get("setup_failed")), end_reason=rep.get("end_reason"),
|
||||||
|
|||||||
@@ -140,3 +140,5 @@ Training runs on HF Jobs with Unsloth, not on the Mac. No `mlx_lm` training.
|
|||||||
round-trip check of every tool call; assistant spans for loss masking). B2d filter: `train/accept.py`.
|
round-trip check of every tool call; assistant spans for loss masking). B2d filter: `train/accept.py`.
|
||||||
- Test: scripted fake model on T01 (A4H, no cloud): score 100, 1 syntax hint, record, filter and converter OK.
|
- Test: scripted fake model on T01 (A4H, no cloud): score 100, 1 syntax hint, record, filter and converter OK.
|
||||||
- B3 trajectory runner: `harness/trajectories.py` (6 workers, DeepSeek V4.1 Flash, output `runs/traj/`). Starts after generation.
|
- B3 trajectory runner: `harness/trajectories.py` (6 workers, DeepSeek V4.1 Flash, output `runs/traj/`). Starts after generation.
|
||||||
|
|
||||||
|
- 2026-10-05 Kral: tool-call budget 100 (eval stays 60) for trajectory runs on tasks with a CDS contract object (`Runner.run(cds_calls=100)`, `trajectories.CDS_CALLS`). Reason: first 3 CDS runs were correct (hidden tests all passed) but ended at 60 calls or by an empty response while writing the own CDS test class (score 80, not accepted). `metadata.tool_budget` is in every record. Failed CDS tasks get the second attempt with the new budget.
|
||||||
|
|||||||
Reference in New Issue
Block a user