Harness: runner, generator, tasks T01 T13 T14 T15, CLAUDE.md

This commit is contained in:
Kral
2026-10-02 21:42:51 +02:00
commit 1c6c85492e
54 changed files with 3732 additions and 0 deletions

139
harness/agents.py Normal file
View File

@@ -0,0 +1,139 @@
"""Agents: oracle (reference solution), null (does nothing), llm (OpenAI-compatible tool loop)."""
import json
import os
import time
import urllib.request
from .ledger import add_usage, check_budget
from .proxy import BudgetExceeded
SYSTEM_PROMPT = """You are an ABAP developer. You implement a plan on an SAP system with the tools.
Rules:
- Read the plan. Implement the contract exactly. Do not change a public signature in the contract.
- You make the craft decisions: table types, keys, access paths, SQL strategy, exception design,
and syntax that fits the release target.
- If a business rule is missing, or two rules contradict, stop. Do not create objects.
Write the gap in your report.
- Do not change an object in the out-of-scope list.
- Create a new object in package $TMP with sap_create_object. Then write the source with
sap_push_source.
- Read an object that you do not know before you use it.
- After each write, check the result. If the same check fails two times, stop and report.
- Do not add behavior that the plan does not ask for.
At the end, write a short report:
1. Decisions: one line for each craft decision, with the reason.
2. Objects: object, type, action, reason.
3. Verification: what you checked and the result.
4. Deviations and open points.
"""
class NullAgent:
name = "null"
def run(self, task, proxy):
proxy.note("final", "No action.")
return "No action."
class OracleAgent:
"""Writes the reference solution. Validates harness and scoring (expected score ~ max)."""
name = "oracle"
def run(self, task, proxy):
for o in task.objects("reference"):
ident = {"objectType": o["type"], "objectName": o["name"]}
if o.get("functionGroup"):
ident["functionGroup"] = o["functionGroup"]
proxy.call("sap_create_object", dict(ident, packageName="$TMP",
description=o.get("description", o["name"])[:60]))
if o.get("source"):
proxy.call("sap_push_source", dict(ident, source=o["source"]))
if o.get("testclasses_source"):
proxy.call("sap_push_source", dict(ident, includeType="testclasses",
source=o["testclasses_source"]))
proxy.note("final", "Reference solution written.")
return "Reference solution written."
class LlmAgent:
"""OpenAI-compatible chat completions with tool calls (Ollama, MLX server, vLLM, ...).
Local models are slow: no time limit by default (tool-call budget limits the run).
One request at a time; tool calls run in sequence.
"""
def __init__(self, model, base_url=None, api_key=None, max_turns=80, temperature=0.2,
max_seconds=None):
self.model = model
self.name = f"llm:{model}"
self.base_url = (base_url or os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")).rstrip("/")
self.api_key = api_key or os.environ.get("LLM_API_KEY", "none")
self.max_turns = max_turns
self.temperature = temperature
self.max_seconds = max_seconds
self.request_timeout = 3600 # local models are slow; a hung request still ends
def _chat(self, messages, tools):
body = {"model": self.model, "messages": messages, "tools": tools,
"temperature": self.temperature, "parallel_tool_calls": False}
req = urllib.request.Request(f"{self.base_url}/chat/completions", json.dumps(body).encode(),
{"Content-Type": "application/json",
"Authorization": f"Bearer {self.api_key}"})
last = None
for attempt in range(4): # model server errors (HTTP 5xx, timeouts): retry with backoff
try:
with urllib.request.urlopen(req, timeout=self.request_timeout) as r:
data = json.loads(r.read().decode())
return data["choices"][0]["message"], data.get("usage", {})
except Exception as e: # noqa: BLE001
last = e
code = getattr(e, "code", None)
if code is not None and code < 500 and code != 429:
raise
time.sleep(10 * (attempt + 1))
raise RuntimeError(f"model request failed after retries: {last}")
def run(self, task, proxy):
tools = [{"type": "function", "function": {"name": t["name"],
"description": t.get("description", ""),
"parameters": t.get("inputSchema", {})}}
for t in proxy.schemas()]
messages = [{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": task.spec}]
if ":cloud" in self.model:
check_budget()
final = ""
start = time.time()
for _ in range(self.max_turns):
if self.max_seconds and time.time() - start > self.max_seconds:
final = f"Stopped: time budget exceeded ({self.max_seconds} s)."
break
try:
msg, usage = self._chat(messages, tools)
add_usage(self.model, usage, kind="run", ref=proxy.prefix)
except RuntimeError as e:
final = f"Stopped: {e}"
break
proxy.note("assistant", {"content": msg.get("content"),
"tool_calls": msg.get("tool_calls"), "usage": usage})
messages.append({k: v for k, v in msg.items() if k in ("role", "content", "tool_calls")})
calls = msg.get("tool_calls") or []
if not calls:
final = msg.get("content") or ""
break
try:
for c in calls:
fn = c["function"]
args = fn.get("arguments") or "{}"
args = json.loads(args) if isinstance(args, str) else args
err, text = proxy.call(fn["name"], args)
messages.append({"role": "tool", "tool_call_id": c.get("id", ""),
"content": ("ERROR: " if err else "") + text[:12000]})
except BudgetExceeded as e:
final = f"Stopped: budget exceeded ({e})."
break
proxy.note("final", final)
return final