69 lines
2.8 KiB
Python
69 lines
2.8 KiB
Python
"""Trajectory record of one run (stage 2 training data source).
|
|
|
|
record.json : task metadata, the exact messages the teacher saw and wrote, the tool schemas, the raw tool
|
|
results (as the proxy logged them, before the 12000-character cut of the message), run metadata.
|
|
reasoning.json: the teacher reasoning per assistant message. Never training data.
|
|
"""
|
|
import json
|
|
import os
|
|
|
|
from .ledger import LEDGER
|
|
|
|
LEDGER_TO_USAGE = 2.7 # usage = ledger / 2.7 (measured 2026-10-03)
|
|
|
|
|
|
def run_cost(prefix):
|
|
"""Ledger USD of the 'run' entries of this run (by prefix)."""
|
|
if not os.path.exists(LEDGER):
|
|
return 0.0
|
|
tot = 0.0
|
|
for line in open(LEDGER):
|
|
e = json.loads(line)
|
|
if e.get("kind") == "run" and e.get("ref") == prefix:
|
|
tot += e["usd"]
|
|
return round(tot, 5)
|
|
|
|
|
|
def raw_tool_results(traj_path):
|
|
out = []
|
|
for line in open(traj_path):
|
|
e = json.loads(line)
|
|
if "tool" in e:
|
|
out.append({k: e[k] for k in ("tool", "args", "is_error", "result", "raw_result", "variant_tool") if k in e})
|
|
return out
|
|
|
|
|
|
def write_record(run_dir, task, agent, rep, pool):
|
|
msgs = getattr(agent, "messages", None)
|
|
if not msgs:
|
|
return None
|
|
meta = task.meta
|
|
cost = run_cost(rep["prefix"])
|
|
score = rep.get("score") or {}
|
|
record = {
|
|
"version": 1,
|
|
"run": rep["run"], "prefix": rep["prefix"],
|
|
"task": {"id": task.id, "pool": pool, "category": meta.get("category"),
|
|
"object_type": meta.get("object_type"), "release_target": meta.get("release_target"),
|
|
"difficulty": meta.get("difficulty"), "error_kind": meta.get("error_kind"),
|
|
"tool_schema": meta.get("tool_schema")},
|
|
"teacher": agent.model,
|
|
"messages": msgs,
|
|
"tools": agent.tools,
|
|
"tool_results_raw": raw_tool_results(os.path.join(run_dir, "trajectory.jsonl")),
|
|
"metadata": {
|
|
"score": score.get("total"), "score_parts": score, "gates": rep.get("gates"),
|
|
"cost_ledger_usd": cost, "cost_usage_usd": round(cost / LEDGER_TO_USAGE, 5),
|
|
"end_reason": rep.get("end_reason"), "tool_calls": rep.get("tool_calls"),
|
|
"activations": rep.get("activations"), "activation_failures": rep.get("activation_failures"),
|
|
"max_fail_streak": rep.get("max_fail_streak"), "syntax_hints": rep.get("syntax_hints"),
|
|
"adt_fallbacks": bool(rep.get("adt_fallbacks")), "agent_seconds": rep.get("agent_seconds"),
|
|
"turn_usage": getattr(agent, "turn_usage", []),
|
|
"setup_failed": rep.get("setup_failed", False),
|
|
"harness_error": rep.get("harness_error"),
|
|
},
|
|
}
|
|
json.dump(record, open(os.path.join(run_dir, "record.json"), "w"), indent=1)
|
|
json.dump(getattr(agent, "reasoning", []), open(os.path.join(run_dir, "reasoning.json"), "w"), indent=1)
|
|
return record
|