Files
abap-llm/harness/runner.py

444 lines
24 KiB
Python

"""One task run: setup -> agent -> collect -> gates -> tests -> checks -> score -> teardown."""
import json
import os
import re
import subprocess
import time
from .adt_client import AdtClient, load_env
from .judge import gap_named
from .mcp_client import McpClient
from .proxy import ToolProxy
from .task import Task, prefix_for
ABAPLINT = os.path.join(os.path.dirname(__file__), "..", "node_modules", ".bin", "abaplint")
CLEAN_RULES = {
"unused_variables": True, "unused_types": True, "prefer_xsdbool": True, "use_new": True,
"prefer_returning_to_exporting": True, "functional_writing": True,
"preferred_compare_operator": True, "use_line_exists": True, "line_length": {"length": 120},
"max_one_statement": True, "empty_statement": True, "commented_code": True,
"omit_parameter_name": True, "prefer_is_not": True, "exporting": True,
}
DELETE_ORDER = ["CLAS", "INTF", "PROG", "FUNC", "FUGR", "SRVD", "DDLX", "DCLS", "DDLS",
"TTYP", "TABL", "STRU", "DTEL", "DOMA", "MSAG"]
SOURCE_TYPES = ("CLAS", "INTF", "PROG", "FUNC", "DDLS", "DCLS", "DDLX", "TABL")
def _cds_elements(src):
"""Element names of a CDS view select list (alias, else last path component). Used when a
SELECT on the view is not possible, for example a view with parameters."""
body = re.sub(r"//[^\n]*|/\*.*?\*/", "", src or "", flags=re.S)
m = re.search(r"\{(.*)\}", body, re.S)
out = set()
for el in re.split(r",(?![^()]*\))", m.group(1) if m else ""):
el = re.sub(r"@\S+(\s*:\s*('[^']*'|\S+))?", "", el).strip()
if not el or el.lower().startswith("_") and " as " not in el.lower():
continue
alias = re.search(r"\bas\s+(\w+)\s*$", el, re.I)
name = alias.group(1) if alias else re.split(r"[.\s]", el.split()[-1])[-1]
out.add(name.upper())
return out
def _obj_args(otype, name, fg=None):
a = {"objectType": otype, "objectName": name}
if fg:
a["functionGroup"] = fg
return a
def _json(text):
try:
return json.loads(text)
except ValueError:
return None
def _norm(src):
return "\n".join(l.rstrip().lower() for l in (src or "").splitlines() if l.strip())
def trajectory_stats(path):
"""Rebuild agent statistics from a trajectory (same rules as ToolProxy)."""
from .proxy import WRITE_TOOLS
calls = activations = streak = max_streak = 0
final, t_first, t_last = "", None, None
for line in open(path):
e = json.loads(line)
t_first = t_first or e["t"]
t_last = e["t"]
if e.get("kind") == "final":
final = e.get("content") or ""
if "tool" not in e or e["result"].startswith("Tool ") and e["is_error"]:
continue
calls += 1
tool, args = e["tool"], e.get("args", {})
if tool in WRITE_TOOLS:
if tool == "sap_activate" or args.get("activate", True):
activations += 1
failed = e["is_error"] or '"success":false' in e["result"].replace(" ", "")
streak = streak + 1 if failed else 0
max_streak = max(max_streak, streak)
return {"tool_calls": calls, "activations": activations, "max_fail_streak": max_streak,
"final_report": final, "agent_seconds": round((t_last or 0) - (t_first or 0), 1)}
class Runner:
def __init__(self, tasks_root, runs_root):
self.tasks_root = tasks_root
self.runs_root = runs_root
# ---------- helpers ----------
def _install(self, mcp, objs):
out = []
for o in objs:
fg = o.get("functionGroup")
cargs = dict(_obj_args(o["type"], o["name"], fg), packageName="$TMP",
description=o.get("description", o["name"])[:60])
e1, t1 = mcp.call("sap_create_object", cargs)
ok, t2 = not e1, ""
if o.get("source"):
e2, t2 = mcp.call("sap_push_source", dict(_obj_args(o["type"], o["name"], fg),
source=o["source"]))
ok = not e2 and (_json(t2) or {}).get("success", False)
if o.get("testclasses_source"):
e3, t3 = mcp.call("sap_push_source", dict(_obj_args(o["type"], o["name"], fg),
includeType="testclasses",
source=o["testclasses_source"]))
ok = ok and not e3 and (_json(t3) or {}).get("success", False)
if o.get("run"):
e4, t4 = mcp.call("sap_run_class", {"className": o["name"]})
ok = ok and not e4 and (_json(t4) or {}).get("success", False)
out.append({"name": o["name"], "ok": ok, "create": t1[:300], "push": t2[:500]})
return out
def _objects_with_prefix(self, mcp, prefix):
_, text = mcp.call("sap_search_object", {"query": prefix + "*", "maxResults": 200})
return [d for d in (_json(text) or []) if d.get("name", "").upper().startswith(prefix)
and d.get("objectType")] # skips STOB entries of CDS entities
def _source(self, mcp, otype, name, fg=None):
err, text = mcp.call("sap_pull_source", _obj_args(otype, name, fg))
return None if err else text
def _abaplint(self, run_dir, task, sources):
d = os.path.join(run_dir, "abaplint")
os.makedirs(os.path.join(d, "src"), exist_ok=True)
ext = {"CLAS": "clas", "INTF": "intf", "PROG": "prog"}
for (otype, name), src in sources.items():
if otype not in ext:
continue
base = os.path.join(d, "src", f"{name.lower()}.{ext[otype]}")
open(base + ".abap", "w").write(src)
if otype == "PROG":
open(base + ".xml", "w").write(
f'<?xml version="1.0" encoding="utf-8"?><abapGit version="v1.0.0" serializer="LCL_OBJECT_PROG" '
f'serializer_version="v1.0.0"><asx:abap xmlns:asx="http://www.sap.com/abapxml" version="1.0">'
f'<asx:values><PROGDIR><NAME>{name}</NAME><SUBC>1</SUBC><RLOAD>E</RLOAD><FIXPT>X</FIXPT>'
f'<UCCHECK>X</UCCHECK></PROGDIR></asx:values></asx:abap></abapGit>')
continue
tag = "VSEOCLASS" if otype == "CLAS" else "VSEOINTERF"
ser = "LCL_OBJECT_CLAS" if otype == "CLAS" else "LCL_OBJECT_INTF"
open(base + ".xml", "w").write(
f'<?xml version="1.0" encoding="utf-8"?><abapGit version="v1.0.0" serializer="{ser}" '
f'serializer_version="v1.0.0"><asx:abap xmlns:asx="http://www.sap.com/abapxml" version="1.0">'
f'<asx:values><{tag}><CLSNAME>{name}</CLSNAME><LANGU>E</LANGU><DESCRIPT>x</DESCRIPT>'
f'<STATE>1</STATE><UNICODE>X</UNICODE></{tag}></asx:values></asx:abap></abapGit>')
rules = {"check_syntax": True, "unknown_types": True}
rules.update(CLEAN_RULES)
for r in task.meta.get("craft_checks", []):
rules[r] = {"statements": 40} if r == "method_length" else True
cfg = {"global": {"files": "/src/**/*.*"}, "dependencies": [],
"syntax": {"version": task.meta.get("release_target", "v758"),
# objects of this run that abaplint cannot read (TABL, DDLS, FUNC) are not errors;
# A4H syntax check (G1) covers them
"errorNamespace": f"^(?!{task.prefix})(Z|Y|LCL_|TY_|LIF_|LTC_)"},
"rules": rules}
json.dump(cfg, open(os.path.join(d, "abaplint.json"), "w"), indent=1)
p = subprocess.run([ABAPLINT, "abaplint.json", "-f", "json"], cwd=d,
capture_output=True, text=True, timeout=300)
try:
issues = json.loads(p.stdout or "[]")
except ValueError:
issues = [{"key": "abaplint_failed", "description": (p.stdout + p.stderr)[:500],
"file": {"filename": ""}}]
def _fname(i):
f = i.get("file", "")
return f.get("filename", "") if isinstance(f, dict) else str(f)
return [{"rule": i.get("key"), "file": os.path.basename(_fname(i)),
"line": i.get("start", {}).get("row"), "msg": i.get("description")} for i in issues]
# ---------- main ----------
def run(self, task_id, agent, run_no, teardown=True, rescore_dir=None):
prefix = prefix_for(run_no, task_id)
task = Task(os.path.join(self.tasks_root, task_id), prefix)
run_dir = rescore_dir or os.path.join(self.runs_root, f"{int(run_no):03d}_{task_id}_{agent.name.replace(':', '_').replace('/', '_')}")
os.makedirs(run_dir, exist_ok=True)
rep = {"task": task_id, "agent": agent.name, "run": run_no, "prefix": prefix,
"started": time.strftime("%Y-%m-%dT%H:%M:%S")}
t0 = time.time()
with McpClient() as mcp:
# 1 setup
seed = task.objects("seed")
if rescore_dir:
seed_src = {o["name"].upper(): o["source"] for o in seed}
else:
rep["setup"] = self._install(mcp, seed)
if not all(x["ok"] for x in rep["setup"]):
rep["setup_failed"] = True
rep["score"] = {"total": None, "note": "setup failed; run not scored"}
all_objs = self._objects_with_prefix(mcp, prefix)
rep["teardown"] = self._teardown(all_objs, run_dir) if teardown else "skipped"
json.dump(rep, open(os.path.join(run_dir, "report.json"), "w"), indent=1)
return rep, run_dir
seed_src = {o["name"].upper(): self._source(mcp, o["type"], o["name"], o.get("functionGroup"))
for o in seed}
# 2 run
if rescore_dir:
rep.update(trajectory_stats(os.path.join(run_dir, "trajectory.jsonl")))
rep["rescored"] = True
else:
proxy = ToolProxy(mcp, prefix, task.meta.get("budget", {}),
os.path.join(run_dir, "trajectory.jsonl"), task.meta.get("tool_schema"))
proxy.note("spec", task.spec)
t1 = time.time()
rep["final_report"] = agent.run(task, proxy)
rep["agent_seconds"] = round(time.time() - t1, 1)
rep["tool_calls"], rep["activations"] = proxy.calls, proxy.activations
rep["max_fail_streak"] = proxy.max_fail_streak
if proxy.fallbacks:
rep["adt_fallbacks"] = proxy.fallbacks
# 3 collect
hidden_names = {o["name"].upper() for o in task.objects("hidden_tests")}
seed_names = set(seed_src)
objs = self._objects_with_prefix(mcp, prefix)
model_objs = [o for o in objs if o["name"].upper() not in seed_names | hidden_names]
sources = {}
for o in objs:
if o["objectType"] in SOURCE_TYPES:
src = self._source(mcp, o["objectType"], o["name"], o.get("functionGroup"))
if src is not None:
sources[(o["objectType"], o["name"].upper())] = src
os.makedirs(os.path.join(run_dir, "sources"), exist_ok=True)
for (t, n), s in sources.items():
open(os.path.join(run_dir, "sources", f"{n.lower()}.{t.lower()}.abap"), "w").write(s)
rep["model_objects"] = [o["name"] for o in model_objs]
# 4 gates
g = {}
_, inact = mcp.call("sap_inactive_objects", {})
inactive = {d.get("name", "").upper() for d in (_json(inact) or []) if isinstance(d, dict)}
contract = task.objects("contract")
g1 = []
for c in contract:
exists = any(o["name"].upper() == c["name"].upper() for o in objs)
ok = exists and c["name"].upper() not in inactive
if ok:
_, st = mcp.call("sap_syntax_check", _obj_args(c["type"], c["name"], c.get("functionGroup")))
ok = (_json(st) or {}).get("errorCount", 1) == 0
g1.append({"name": c["name"], "ok": ok})
g["G1_active"] = all(x["ok"] for x in g1) and bool(g1)
g2 = True
g2_detail = []
for c in contract:
src = sources.get((c["type"], c["name"].upper()), "")
if c.get("implements") and not re.search(rf"INTERFACES\s+{re.escape(c['implements'])}\b", src, re.I):
g2 = False
g2_detail.append(f"{c['name']}: does not implement {c['implements']}")
if c["type"] == "FUNC": # signature: every parameter with its type in the FUNCTION header
# the FUNCTION statement, not the first statement: local classes can come before it (G0107)
fm = re.search(rf"^\s*FUNCTION\s+{re.escape(c['name'])}\b[^.]*\.", src, re.I | re.M)
header = fm.group(0) if fm else src.split(".", 1)[0]
for prm in c.get("params", []):
if not re.search(rf"\b{re.escape(prm['name'])}\b\)?\s+TYPE\s+{re.escape(prm['type'])}\b",
header, re.I):
g2 = False
g2_detail.append(f"{c['name']}: no parameter {prm['name']} TYPE {prm['type']} "
"in the FUNCTION header")
if c["type"] == "PROG": # selection screen parameters
for prm in c.get("parameters", []):
if not re.search(rf"(PARAMETERS|SELECT-OPTIONS)\s*:?[^.]*\b{re.escape(prm)}\b", src, re.I):
g2 = False
g2_detail.append(f"{c['name']}: no selection screen parameter {prm}")
if c["type"] == "DDLS" and c.get("fields") and g["G1_active"]:
_, q = mcp.call("sap_sql_query", {"query": f"SELECT * FROM {c['name']}", "maxRows": 1})
cols = {col.get("name", "").upper() for col in (_json(q) or {}).get("columns", [])}
if not cols: # view with parameters: SELECT without parameters fails
cols = _cds_elements(src)
missing = {f.upper() for f in c["fields"]} - cols
if missing:
g2 = False
g2_detail.append(f"{c['name']}: CDS elements missing: {sorted(missing)}")
# Categories E and I: the contract object is a seed object (refactor or fix in place). It must
# change (else the null agent passes on the legacy code), and it is not out of scope.
contract_upper = {c["name"].upper() for c in contract}
for c in contract:
if c["name"].upper() in seed_src and c["type"] != "FUGR":
now = self._source(mcp, c["type"], c["name"], c.get("functionGroup"))
if _norm(now) == _norm(seed_src[c["name"].upper()]):
g2 = False
g2_detail.append(f"{c['name']}: the seed object is not changed")
g["G2_contract"] = g2 and g["G1_active"]
rep["contract_check"] = {"g1": g1, "g2": g2_detail}
g["G4_out_of_scope"] = all(
_norm(self._source(mcp, o["type"], o["name"], o.get("functionGroup")))
== _norm(seed_src[o["name"].upper()])
for o in seed if o["name"].upper() not in contract_upper)
atc = []
for c in contract:
if g["G1_active"]:
_, t = mcp.call("sap_atc_run", _obj_args(c["type"], c["name"], c.get("functionGroup")))
for f in (_json(t) or []):
if isinstance(f, dict):
atc.append({"obj": c["name"], "prio": str(f.get("priority", "")),
"check": f.get("checkTitle") or f.get("checkId"),
"msg": (f.get("messageTitle") or f.get("text") or "")[:200]})
rep["atc"] = atc
g["G5_no_p1"] = not any(a["prio"] == "1" for a in atc)
# 5 hidden tests
hidden = {"total": 0, "passed": 0, "detail": []}
if g["G2_contract"]:
inst = self._install(mcp, task.objects("hidden_tests"))
hidden["install"] = inst
for o in task.objects("hidden_tests"):
_, t = mcp.call("sap_run_unit_test", {"objectType": "CLAS", "objectName": o["name"]})
for prog in (_json(t) or {}).get("programs", []):
for tc in prog.get("testClasses", []):
for m in tc.get("testMethods", []):
ok = not m.get("alerts")
hidden["total"] += 1
hidden["passed"] += ok
hidden["detail"].append({"method": m["name"], "ok": ok,
"alerts": json.dumps(m.get("alerts"))[:300]})
rep["hidden_tests"] = hidden
g["G3_hidden_runs"] = hidden["passed"] > 0
# 6 own tests (local test include or own global test class)
own = {"tests": 0, "failures": 0, "coverage": None}
for c in contract:
if c["type"] == "CLAS" and g["G1_active"]:
_, t = mcp.call("sap_check_object", {"objectType": "CLAS", "objectName": c["name"],
"runAtc": False, "runUnitTest": True, "coverage": True})
for st in (_json(t) or {}).get("steps", []):
if st.get("step") == "unittest":
own["tests"] += st.get("tests", 0)
own["failures"] += st.get("failures", 0)
# several contract classes (exception classes + main class): only a class with
# tests has a coverage figure; a later class without tests must not reset it to 0
cov = st.get("coverage", {}).get("class", {}).get("statement", {}).get("percent")
if st.get("tests", 0) and cov is not None:
own["coverage"] = max(own["coverage"] or 0, cov)
if c["type"] == "PROG" and g["G1_active"]: # local test classes inside the report
_, t = mcp.call("sap_run_unit_test", {"objectType": "PROG", "objectName": c["name"]})
for prog in (_json(t) or {}).get("programs", []):
for tc in prog.get("testClasses", []):
for m in tc.get("testMethods", []):
own["tests"] += 1
own["failures"] += bool(m.get("alerts"))
# own global test classes (FM and CDS tasks; also classes)
contract_names = {c["name"].upper() for c in contract}
for (otype, name), src in sources.items():
if (otype == "CLAS" and name not in contract_names | seed_names | hidden_names
and re.search(r"FOR\s+TESTING", src, re.I)):
_, t = mcp.call("sap_run_unit_test", {"objectType": "CLAS", "objectName": name})
for prog in (_json(t) or {}).get("programs", []):
for tc in prog.get("testClasses", []):
for m in tc.get("testMethods", []):
own["tests"] += 1
own["failures"] += bool(m.get("alerts"))
own.setdefault("global_test_classes", []).append(name)
rep["own_tests"] = own
# 7 abaplint
lint_sources = {k: v for k, v in sources.items() if k[1] not in hidden_names}
issues = self._abaplint(run_dir, task, lint_sources)
model_files = {f"{n.lower()}.{t.lower()}.abap" for (t, n) in lint_sources if n not in seed_names}
issues = [i for i in issues if i["file"] in model_files or i["rule"] == "abaplint_failed"]
rep["abaplint"] = issues
# Release gate only on contract objects. Helper objects (smoke runners, test classes)
# can use standard APIs that abaplint does not know; A4H syntax check covers them.
contract_files = {f"{c['name'].lower()}.{c['type'].lower()}.abap" for c in contract}
# abaplint does not know standard superclasses (CX_STATIC_CHECK ...); an exception hierarchy
# then gives "Super class ... not found or contains errors". A4H activation (G1) covers this.
g["G6_release"] = not any(i["rule"] == "check_syntax" and i["file"] in contract_files
and not re.match(r"Super class .* not found or contains errors", i["msg"] or "")
for i in issues)
# 8 score
rep["gates"] = g
rep["score"] = self._score(task, rep)
rep["seconds"] = round(time.time() - t0, 1)
# 9 teardown
all_objs = self._objects_with_prefix(mcp, prefix)
rep["teardown"] = self._teardown(all_objs, run_dir) if teardown else "skipped"
json.dump(rep, open(os.path.join(run_dir, "report.json"), "w"), indent=1)
return rep, run_dir
def _score(self, task, rep):
g = rep["gates"]
if task.meta.get("expected_outcome") == "stop": # category H, docs/faz1-tasarim.md 6.3
if rep.get("model_objects"):
return {"total": 0, "stop": "implemented", "objects": rep["model_objects"]}
named, detail = gap_named(rep.get("final_report"), task.meta.get("expected_gap", {}))
return {"total": 100 if named else 30, "stop": "stopped", "gap": detail}
if not all(g.values()):
return {"total": 0, "failed_gates": [k for k, v in g.items() if not v]}
h = rep["hidden_tests"]
correctness = 40 * h["passed"] / max(h["total"], 1)
craft_rules = set(task.meta.get("craft_checks", []))
contract_files = {f"{c['name'].lower()}.{c['type'].lower()}.abap" for c in task.objects("contract")}
craft_n = sum(1 for i in rep["abaplint"] if i["rule"] in craft_rules and i["file"] in contract_files)
craft_n += sum(1 for a in rep["atc"] if a["prio"] == "2")
craft = 20 * max(0.0, 1 - 0.25 * craft_n)
clean_n = sum(1 for i in rep["abaplint"] if i["rule"] in CLEAN_RULES)
clean = 15 * max(0.0, 1 - 0.1 * clean_n)
own = rep["own_tests"]
own_score = 0.0
if own["tests"] > 0 and own["failures"] == 0:
own_score += 7.5
if own["coverage"] is not None:
own_score += 7.5 * min(1.0, own["coverage"] / 70.0)
elif own_score: # no coverage figure for this object type (FUNC, PROG, DDLS): tests count fully
own_score = 15.0
b = task.meta.get("budget", {})
disc = 10.0
if rep["tool_calls"] > b.get("max_tool_calls", 60):
disc -= 5
if rep["activations"] > b.get("max_activations", 15):
disc -= 5
if rep.get("max_fail_streak", 0) > 2: # rule: stop after two failures of the same check
disc -= 5
final = (rep.get("final_report") or "").strip()
if not final or final.startswith("Stopped:"): # no report from the model
disc -= 5
disc = max(disc, 0.0)
parts = {"correctness": round(correctness, 1), "craft": round(craft, 1), "clean": round(clean, 1),
"own_tests": round(own_score, 1), "discipline": disc}
parts["total"] = round(sum(parts.values()), 1)
parts["note"] = "own_tests: faulty-reference part not implemented in skeleton"
return parts
def _teardown(self, objs, run_dir):
objs = sorted(objs, key=lambda o: DELETE_ORDER.index(o["objectType"])
if o["objectType"] in DELETE_ORDER else 99)
uris = [o["uri"] for o in objs]
json.dump(uris, open(os.path.join(run_dir, "delete_uris.json"), "w"), indent=1)
load_env()
if not os.environ.get("A4H_PASSWORD"):
return {"status": "pending", "count": len(uris),
"hint": "Set A4H_URL/A4H_USER/A4H_PASSWORD in .env, then: python3 -m harness.cli teardown <run_dir>"}
return delete_uris(uris)
def delete_uris(uris):
"""Delete one by one; this keeps the dependency order explicit."""
adt = AdtClient()
res = {}
for u in uris:
res.update(adt.delete([u]))
return {u: {"deleted": ok, "msg": msg} for u, (ok, msg) in res.items()}