Infrastructure outage handling (MCP/A4H down: wait, clean up, rerun), dashboard fix, budget limit 129

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-05 22:27:48 +02:00
parent 5c07026bbb
commit 9c753a37ff
5 changed files with 105 additions and 3 deletions

View File

@@ -17,7 +17,9 @@ from concurrent.futures import ThreadPoolExecutor
from .adt_client import load_env
from .agents import LlmAgent
from . import infra
from .ledger import BudgetExceeded, check_budget, spent
from .task import prefix_for
from .record import LEDGER_TO_USAGE
from .runner import Runner
@@ -77,7 +79,25 @@ def one(task_id, attempt, stop):
t0 = time.time()
row = {"task": task_id, "attempt": attempt, "run": run_no, "model": MODEL}
try:
rep, run_dir = runner.run(task_id, agent, run_no, teardown=True, cds_calls=CDS_CALLS)
for _try in range(4): # an outage of MCP or A4H is waited out and the same run starts again
try:
rep, run_dir = runner.run(task_id, agent, run_no, teardown=True, cds_calls=CDS_CALLS)
break
except Exception as e: # noqa: BLE001
if not infra.is_outage(e) or _try == 3:
raise
print(time.strftime("%F %T"), "infra outage at", task_id, "->", repr(e)[:120], "; waiting", flush=True)
append("outages.jsonl", {"t": time.time(), "task": task_id, "attempt": attempt, "error": repr(e)[:200]})
if not infra.wait_until_up():
raise
try:
infra.cleanup_prefix(prefix_for(run_no, task_id))
except Exception: # noqa: BLE001
pass
for d in glob.glob(os.path.join(OUT, f"{run_no}_{task_id}_*")):
os.makedirs(os.path.join(OUT, "_aborted"), exist_ok=True)
os.rename(d, os.path.join(OUT, "_aborted", os.path.basename(d) + "_" + str(int(time.time()))))
agent = LlmAgent(MODEL, loop_guard=3)
score = (rep.get("score") or {}).get("total")
rec = os.path.exists(os.path.join(run_dir, "record.json"))
row.update(score=score, setup_failed=bool(rep.get("setup_failed")), end_reason=rep.get("end_reason"),