diff --git a/docs/epod-syntax-hint.md b/docs/epod-syntax-hint.md new file mode 100644 index 0000000..8c0019f --- /dev/null +++ b/docs/epod-syntax-hint.md @@ -0,0 +1,79 @@ +# EPOD: syntax messages for a failed write + +Status: proposal (2026-10-05). Kral decides later if it goes into the EPOD server. +The harness proxy does this today (`harness/proxy.py`, `harness/localcheck.py`). The proxy is not changed by this note. + +## Problem + +`sap_push_source` fails with only this result when SAP does not accept the source: + +```json +{"success":false,"error":"[WRITE] An error occured during the save operation. The changes were not stored."} +``` + +Example cause: `METHODS run IMPORTING iv TYPE c LENGTH 4.` (a generic type with LENGTH in a method signature). +The model cannot see the cause, so it cannot repair. In the pilot this was 7 failures in 3 tasks. + +`sap_syntax_check` does not help: it reads the stored version (inactive or active), not the rejected source. +Tested 2026-10-05: after the failed write it returned `{"version":"inactive","errorCount":0,"findings":[]}` +(it checked the empty stub that `sap_create_object` made). + +## What the proxy does now + +1. A result of `sap_push_source` is a bare save failure when the JSON has `success:false`, the error text contains + "during the save operation", and `activation.messages` is empty (`localcheck.is_bare_save_failure`). +2. The proxy takes the source from the call arguments and runs the abaplint parser on it + (`localcheck.parser_messages`): one temporary file named by object type (`.clas.abap`, + `.clas.testclasses.abap` for `includeType` testclasses, `.intf.abap`, `.prog.abap`, + `.fugr..abap`, `.ddls.asddls`), rule `parser_error` only, syntax version from the task release target + (default v758). TABL, FUGR and other types without a parser: no hint. +3. It adds a `syntaxCheck` field to the same JSON object (`localcheck.augment`). The original result stays in + the trajectory log as `raw_result`: + +```json +{"success":false, + "error":"[WRITE] An error occured during the save operation. The changes were not stored.", + "syntaxCheck":{"source":"local abaplint (the rejected source, not stored in SAP)", + "messages":[{"line":3,"severity":"E", + "message":"Statement does not exist in the configured ABAP version(or a parser error), \"METHODS\""}]}} +``` + +Limits: abaplint reports the line and the statement that it cannot parse, not the SAP reason ("TYPE c LENGTH in a +signature is not allowed"). It does not know SAP rules that are not syntax (reserved words in DDL, name length, CDS +annotations). It finds only parser errors, no type or semantic errors. + +## Same logic in the EPOD server + +The server has the source and the lock, so it can do better than abaplint: check the rejected source with the +SAP syntax check, without saving it. + +1. In `sap_push_source` (and `sap_push_element`), when the save step fails, run the ADT syntax check on the + source in the request. Eclipse checks unsaved editor content through the ADT check run (not tested by the harness yet): POST + `/sap/bc/adt/checkruns?reporters=abapCheckRun` with a `chkrun:checkObject` that has the object URI and an + `artifacts` entry with the new source (base64) as `chkrun:content` . + The answer has `chkrun:checkMessage` entries with `chkrun:uri` (line and column), `chkrun:type` (E, W), and + `chkrun:shortText`. +2. Add the messages to the failed result in the same field shape as the proxy, so a client does not care who made + them: + +```json +{"success":false, + "error":"[WRITE] An error occured during the save operation. The changes were not stored.", + "syntaxCheck":{"source":"sap_checkrun (the rejected source, not stored)", + "messages":[{"line":3,"column":41,"severity":"E","message":"..."}]}} +``` + +3. Keep the order: save, then (only when the save fails) check. A good write costs no extra call. +4. Extend `sap_syntax_check` with an optional `source` argument (check this source, do not store it). The model + can then check before it writes. This is useful by itself, also when no write failed. +5. If the check run itself fails (lock, timeout), return the original result without `syntaxCheck`. The field is optional. + +Open points for EPOD: whether the checkrun accepts an unsaved source for every object type (classes with include +types, function modules in a group, CDS); the harness would test this on A4H first (null/oracle runs only). + +## What stays in the harness either way + +- A server-side change replaces `localcheck.py` for the types that the server supports. The proxy code + checks `syntaxCheck` first: if the field is there, it adds nothing. +- Training data from before the change has the abaplint messages, data after it has the SAP messages. Keep the + `syntaxCheck.source` text so the two kinds can be told apart (the field is in every record). diff --git a/harness/generator.py b/harness/generator.py index fb7e8de..e8b8ce8 100644 --- a/harness/generator.py +++ b/harness/generator.py @@ -490,11 +490,12 @@ def contract_terms(task): return out -def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2): +def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2, + pool="eval"): """Category K: free-text or incomplete input + other tool schema. Same reference and hidden tests as the source task; only spec.md, category and tool_schema change.""" check_budget() - pool_root = os.path.join(ROOT, "tasks_gen", "eval") + pool_root = os.path.join(ROOT, "tasks_gen", pool) src_dir, task_dir = os.path.join(pool_root, src_id), os.path.join(pool_root, new_id) b = bundle_of(src_dir) b["files"] = {k: v for k, v in b["files"].items() @@ -507,7 +508,7 @@ def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema "ALV). Keep these names exactly as written (the tests use them): " f"{', '.join(terms)}. Keep the placeholder {{{{P}}}} in names. Use Simplified Technical English. " f"Return only the new spec text.\n\nSpec:\n{spec}"}] - log = {"id": new_id, "pool": "eval", "category": "K", "base_task": src_id, "style": style, + log = {"id": new_id, "pool": pool, "category": "K", "base_task": src_id, "style": style, "tool_schema": tool_schema, "attempts": []} for attempt in range(max_repairs + 1): text = chat(model, messages, base_url).strip() @@ -520,7 +521,9 @@ def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema + ". Return the full spec again with all names."}) continue b["files"]["spec.md"] = text + "\n" - b["task"].update(category="K", base_task=src_id, input_style=style, tool_schema=tool_schema) + b["task"].update(category="K", base_task=src_id, input_style=style) + if tool_schema: # None: EPOD tool names (training: free-text input only) + b["task"]["tool_schema"] = tool_schema write_bundle(b, task_dir, new_id) rep_o, rep_n = validate(pool_root, new_id, run_base + RUNS_PER_ATTEMPT * attempt) so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total") diff --git a/harness/trainset.py b/harness/trainset.py index 28f1aa9..db6d95b 100644 --- a/harness/trainset.py +++ b/harness/trainset.py @@ -16,7 +16,7 @@ import time from .adt_client import load_env from .evalset import SLOTS, RELEASES, accepted_goals -from .generator import ROOT, generate +from .generator import ROOT, generate, make_k_variant from .ledger import BudgetExceeded, spent from . import overlap @@ -140,10 +140,58 @@ def run(part, parts, target, stop_ledger): print(json.dumps(log), flush=True) +K_FIRST_ID = 1300 +K_RUN_BASE = 41800 # 20 per variant; above the trajectory run numbers (41000-41700) +K_COUNT = 18 # K share of the eval plan: 10 of 110 (9 %); counted inside the 200 accepted tasks +K_STYLES_CYCLE = ["free_text", "incomplete"] + + +def run_k(count): + """K tasks for training: free-text or incomplete spec of an accepted training task, EPOD tool names + (no generic_v0). Same reference and hidden tests as the source task.""" + base_url = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1") + for n in range(count): + new_id = f"G{K_FIRST_ID + n}" + if os.path.exists(os.path.join(POOL, "_logs", new_id + ".json")): + continue + used = {json.load(open(f)).get("base_task") for f in glob.glob(os.path.join(POOL, "_logs", "G13*.json"))} + cands = [] # accepted, not K, not H (a stop task has no free-text form), not used yet + for f in sorted(glob.glob(os.path.join(POOL, "_logs", "G1[0-2]*.json"))): + lg = json.load(open(f)) + if lg.get("accepted") and lg.get("category") not in ("H", "K") and lg["id"] not in used: + cands.append(lg) + if not cands: + print("no source task left", flush=True) + return + kinds = {} + for lg in cands: # spread over object types: take the type with the fewest K variants so far + kinds.setdefault(lg["object_type"], []).append(lg) + done_types = [json.load(open(f)).get("object_type") for f in glob.glob(os.path.join(POOL, "_logs", "G13*.json"))] + otype = min(kinds, key=lambda t: done_types.count(t)) + src = kinds[otype][0] + style = K_STYLES_CYCLE[n % 2] + if spent() >= json.load(open(PLAN))["ledger_at_start"] + json.load(open(PLAN))["stop_ledger"]: + print("PHASE LIMIT", flush=True) + return + try: + log = make_k_variant(src["id"], new_id, style, "deepseek-v4.1-flash:cloud", base_url, + K_RUN_BASE + 20 * n, tool_schema=None, pool="train") + except BudgetExceeded as e: + print("BUDGET", e, flush=True) + return + log.update(object_type=otype, spent_total=spent()) + stray = os.path.join(POOL, "generation.json") + if os.path.exists(stray): + os.remove(stray) + os.makedirs(os.path.join(POOL, "_logs"), exist_ok=True) + json.dump(log, open(os.path.join(POOL, "_logs", new_id + ".json"), "w"), indent=1) + print(json.dumps(log), flush=True) + + def main(): load_env(os.path.join(ROOT, ".env")) ap = argparse.ArgumentParser() - ap.add_argument("cmd", choices=["plan", "run"]) + ap.add_argument("cmd", choices=["plan", "run", "k"]) ap.add_argument("--part", type=int, default=0) ap.add_argument("--parts", type=int, default=1) ap.add_argument("--target", type=int, default=200) @@ -155,6 +203,9 @@ def main(): print(len(p), "slots", collections.Counter(x["category"] for x in p)) print(collections.Counter(x["object_type"] for x in p), collections.Counter(x.get("error_kind") for x in p)) return + if a.cmd == "k": + run_k(K_COUNT) + return run(a.part, a.parts, a.target, a.stop_ledger) diff --git a/harness/trajectories.py b/harness/trajectories.py index 762fff4..ba0aeb8 100644 --- a/harness/trajectories.py +++ b/harness/trajectories.py @@ -47,6 +47,15 @@ def done_keys(): return {(r["task"], r["attempt"]) for r in map(json.loads, open(path))} +def failed_first_attempts(min_score): + """Tasks whose attempt 0 did not give an acceptable run (score, end reason, harness error).""" + path = os.path.join(OUT, "summary.jsonl") + rows = [json.loads(l) for l in open(path)] if os.path.exists(path) else [] + return sorted(r["task"] for r in rows if r["attempt"] == 0 and ( + r.get("harness_error") or r.get("setup_failed") or r.get("end_reason") != "report" + or (r.get("score") or 0) < min_score)) + + def append(name, row): with LOCK, open(os.path.join(OUT, name), "a") as f: f.write(json.dumps(row) + "\n") @@ -91,7 +100,10 @@ def main(): ap = argparse.ArgumentParser() ap.add_argument("cmd", choices=["run", "list"]) ap.add_argument("--workers", type=int, default=6) - ap.add_argument("--attempts", type=int, default=1) + ap.add_argument("--attempts", type=int, default=1, + help="1 = one attempt per task; with --retry-failed: the second attempt only for failed tasks") + ap.add_argument("--retry-failed", action="store_true") + ap.add_argument("--min-score", type=float, default=80) ap.add_argument("--limit", type=int, help="maximum number of runs in this call") ap.add_argument("--tasks", nargs="*") a = ap.parse_args() @@ -99,6 +111,8 @@ def main(): tasks = a.tasks or accepted_tasks() done = done_keys() todo = [(t, k) for k in range(a.attempts) for t in tasks if (t, k) not in done] + if a.retry_failed: # attempt 1 only for tasks whose attempt 0 failed + todo = [(t, 1) for t in failed_first_attempts(a.min_score) if t in tasks and (t, 1) not in done] if a.limit: todo = todo[:a.limit] print(len(tasks), "accepted tasks,", len(todo), "runs to do, spent", spent(), flush=True)