Infrastructure outage handling (MCP/A4H down: wait, clean up, rerun), dashboard fix, budget limit 129
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -23,8 +23,9 @@ import urllib.request
|
||||
|
||||
from .adt_client import load_env
|
||||
from .agents import LlmAgent
|
||||
from . import mix
|
||||
from . import infra, mix
|
||||
from .runner import Runner
|
||||
from .task import prefix_for
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
sys.path.insert(0, os.path.join(ROOT, "train"))
|
||||
@@ -169,6 +170,21 @@ class Series:
|
||||
try:
|
||||
rep, run_dir = runner.run(item["task"], agent, run_no, teardown=True, tool_budget=SETTINGS["tool_call_budget"])
|
||||
except Exception as e: # noqa: BLE001 a harness exception is not a model result
|
||||
if infra.is_outage(e): # MCP or A4H is down: wait, clean up, run the same task again (not a failure)
|
||||
self.event("infra_outage", item, repr(e)[:150])
|
||||
self.state["status"] = "paused"
|
||||
self.save()
|
||||
infra.wait_until_up(max_seconds=max(60, self.window_end() - now()))
|
||||
try:
|
||||
infra.cleanup_prefix(prefix_for(run_no, item["task"]))
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
for d in __import__("glob").glob(os.path.join(OUT, "runs", f"{run_no}_{item['task']}_*")):
|
||||
self.park(d, "_paused")
|
||||
if now() >= self.window_end():
|
||||
return {"window_over": True}
|
||||
restarts = min(restarts + 1, 9)
|
||||
continue
|
||||
self.event("harness_exception", item, repr(e)[:200])
|
||||
return {"task": item["task"], "kind": item["kind"], "category": item["category"], "harness_error": repr(e)[:200]}
|
||||
er = rep.get("end_reason")
|
||||
|
||||
Reference in New Issue
Block a user