Infrastructure outage handling (MCP/A4H down: wait, clean up, rerun), dashboard fix, budget limit 129

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-05 22:27:48 +02:00
parent 5c07026bbb
commit 9c753a37ff
5 changed files with 105 additions and 3 deletions

View File

@@ -23,8 +23,9 @@ import urllib.request
from .adt_client import load_env
from .agents import LlmAgent
from . import mix
from . import infra, mix
from .runner import Runner
from .task import prefix_for
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, os.path.join(ROOT, "train"))
@@ -169,6 +170,21 @@ class Series:
try:
rep, run_dir = runner.run(item["task"], agent, run_no, teardown=True, tool_budget=SETTINGS["tool_call_budget"])
except Exception as e: # noqa: BLE001 a harness exception is not a model result
if infra.is_outage(e): # MCP or A4H is down: wait, clean up, run the same task again (not a failure)
self.event("infra_outage", item, repr(e)[:150])
self.state["status"] = "paused"
self.save()
infra.wait_until_up(max_seconds=max(60, self.window_end() - now()))
try:
infra.cleanup_prefix(prefix_for(run_no, item["task"]))
except Exception: # noqa: BLE001
pass
for d in __import__("glob").glob(os.path.join(OUT, "runs", f"{run_no}_{item['task']}_*")):
self.park(d, "_paused")
if now() >= self.window_end():
return {"window_over": True}
restarts = min(restarts + 1, 9)
continue
self.event("harness_exception", item, repr(e)[:200])
return {"task": item["task"], "kind": item["kind"], "category": item["category"], "harness_error": repr(e)[:200]}
er = rep.get("end_reason")