Files
abap-llm/harness/record.py

69 lines
2.9 KiB
Python

"""Trajectory record of one run (stage 2 training data source).
record.json : task metadata, the exact messages the teacher saw and wrote, the tool schemas, the raw tool
results (as the proxy logged them, before the 12000-character cut of the message), run metadata.
reasoning.json: the teacher reasoning per assistant message. Never training data.
"""
import json
import os
from .ledger import LEDGER
LEDGER_TO_USAGE = 2.7 # usage = ledger / 2.7 (measured 2026-10-03)
def run_cost(prefix):
"""Ledger USD of the 'run' entries of this run (by prefix)."""
if not os.path.exists(LEDGER):
return 0.0
tot = 0.0
for line in open(LEDGER):
e = json.loads(line)
if e.get("kind") == "run" and e.get("ref") == prefix:
tot += e["usd"]
return round(tot, 5)
def raw_tool_results(traj_path):
out = []
for line in open(traj_path):
e = json.loads(line)
if "tool" in e:
out.append({k: e[k] for k in ("tool", "args", "is_error", "result", "raw_result", "variant_tool") if k in e})
return out
def write_record(run_dir, task, agent, rep, pool):
msgs = getattr(agent, "messages", None)
if not msgs:
return None
meta = task.meta
cost = run_cost(rep["prefix"])
score = rep.get("score") or {}
record = {
"version": 1,
"run": rep["run"], "prefix": rep["prefix"],
"task": {"id": task.id, "pool": pool, "category": meta.get("category"),
"object_type": meta.get("object_type"), "release_target": meta.get("release_target"),
"difficulty": meta.get("difficulty"), "error_kind": meta.get("error_kind"),
"tool_schema": meta.get("tool_schema")},
"teacher": agent.model,
"messages": msgs,
"tools": agent.tools,
"tool_results_raw": raw_tool_results(os.path.join(run_dir, "trajectory.jsonl")),
"metadata": {
"score": score.get("total"), "score_parts": score, "gates": rep.get("gates"),
"cost_ledger_usd": cost, "cost_usage_usd": round(cost / LEDGER_TO_USAGE, 5),
"end_reason": rep.get("end_reason"), "tool_calls": rep.get("tool_calls"),
"activations": rep.get("activations"), "activation_failures": rep.get("activation_failures"),
"max_fail_streak": rep.get("max_fail_streak"), "tool_budget": rep.get("tool_budget"), "syntax_hints": rep.get("syntax_hints"),
"adt_fallbacks": bool(rep.get("adt_fallbacks")), "agent_seconds": rep.get("agent_seconds"),
"turn_usage": getattr(agent, "turn_usage", []),
"setup_failed": rep.get("setup_failed", False),
"harness_error": rep.get("harness_error"),
},
}
json.dump(record, open(os.path.join(run_dir, "record.json"), "w"), indent=1)
json.dump(getattr(agent, "reasoning", []), open(os.path.join(run_dir, "reasoning.json"), "w"), indent=1)
return record