Trajectory record (messages, raw tool results, metadata; reasoning apart) and trajectory runner

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-05 12:03:54 +02:00
parent a4eb567e4c
commit d5e43e1a83
4 changed files with 194 additions and 1 deletions

View File

@@ -85,6 +85,7 @@ class LlmAgent:
self.empty_retries = 2
self.loop_guard = loop_guard # end the run after this many identical pushes in a row (None = off)
self.end_reason = None
self.messages, self.tools, self.reasoning, self.turn_usage = [], [], [], [] # for the trajectory record
self.chat_template_kwargs = chat_template_kwargs # local server only, e.g. {"enable_thinking": False}
def _chat(self, messages, tools):
@@ -118,6 +119,7 @@ class LlmAgent:
for t in proxy.schemas()]
messages = [{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": task.spec}]
self.messages, self.tools, self.reasoning, self.turn_usage = messages, tools, [], []
if ":cloud" in self.model:
check_budget()
final = ""
@@ -132,6 +134,7 @@ class LlmAgent:
try:
msg, usage = self._chat(messages, tools)
add_usage(self.model, usage, kind="run", ref=proxy.prefix)
self.turn_usage.append(usage)
except RuntimeError as e:
final = f"Stopped: {e}"
self.end_reason = "model_error"
@@ -139,6 +142,9 @@ class LlmAgent:
proxy.note("assistant", {"content": msg.get("content"),
"tool_calls": msg.get("tool_calls"), "usage": usage})
messages.append({k: v for k, v in msg.items() if k in ("role", "content", "tool_calls")})
think = msg.get("reasoning") or msg.get("reasoning_content") or msg.get("thinking")
if think: # kept apart from the messages: not training data
self.reasoning.append({"message_index": len(messages) - 1, "reasoning": think})
calls = msg.get("tool_calls") or []
if not calls and not (msg.get("content") or "").strip() and empty < self.empty_retries:
empty += 1 # empty turn (output limit reached in reasoning): ask the same question again