"""Task generator: a cloud model writes a task bundle; the harness validates it (oracle = 100, null = 0). python3 -m harness.generator --id G0001 --pool eval --object-type CLAS --category C --difficulty 2 """ import argparse import json import os import re import shutil import subprocess import tempfile import time import urllib.request from .adt_client import load_env from .agents import NullAgent, OracleAgent from .ledger import add_usage, check_budget from .runner import Runner ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) ABAPLINT = os.path.join(ROOT, "node_modules", ".bin", "abaplint") MAX_OUT_TOKENS = 100000 ABAPLINT_VERSIONS = {"v702", "v740sp05", "v750", "v758"} EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15"} CATEGORIES = { "A": "pure logic in a new class (language, OO design, Clean ABAP)", "B": "database access (ABAP SQL, CDS) with test doubles", "C": "internal table craft (table kind, keys, access path)", "D": "exception design (class-based exceptions, messages)", "E": "refactoring legacy code (seed contains legacy code; behavior must stay the same)", "F": "use an unknown object: the spec names a seed object but not its signature; the model must read it", "G": "old release target (v702 or v740sp05): no newer syntax", "I": "bug fix: the seed contains a faulty object; the spec describes the symptom", } SYSTEM = """You write evaluation tasks for an ABAP developer model. Each task is a bundle of files. The model gets only spec.md and works on an SAP ABAP Platform 2025 system (SAP_BASIS 816, client 001) through ADT tools. The harness installs the seed objects, runs the model, then checks the result with hidden ABAP Unit tests, ATC, and abaplint. Rules for the bundle: - Use the placeholder {{P}} (upper case) and {{p}} (lower case) at the start of EVERY object name. The harness replaces it with a run prefix of 9 characters (for example Z005P001_). Name length after replacement: classes, programs, function modules, CDS entities max 30; database tables max 16; function groups max 26. - Every other ABAP name (methods, test methods, local classes, data, types, constants) has max 30 characters. Keep test method names short, for example "rejects_zero_qty". - Method parameters need a complete type: "TYPE c LENGTH 4" is not allowed in a signature. Declare a type first (TYPES ty_zone TYPE c LENGTH 4) and use it. - Field names in tables and CDS views are not SQL reserved words (no HOURS, MODE, ORDER, DATE, COUNT ...). - CDS: use "define view entity". Parameters have no default value. In a UNION, all branches have the same key elements and element names. Hidden tests for a view with parameters pass all parameters. - Legacy seed code (categories E, I) must activate on SAP_BASIS 816: do not mix old and new syntax in one SQL statement (with new syntax, every host variable needs "@"). - The contract lists only the objects that the hidden tests call. Test classes are never in the contract. Every reference class (except exception classes) has a "testclasses_file". - A database table seed has type "TABL" and DDL source "define table ...". Type "DDLS" is only for CDS views. - All objects are in package $TMP. Do not use transports. - Do not use SAP application module data (no FI, SD, MM tables). Use generic business domains and only objects that exist in every ABAP Platform system (language, ABAP SQL, CDS, CL_ABAP_*, CL_SALV_TABLE, CL_OSQL_TEST_ENVIRONMENT, CL_CDS_TEST_ENVIRONMENT). Seed your own tables and data if needed. - No dynpro (CALL SCREEN), no SmartForms, no BAdI, no RAP behavior definitions. - spec.md uses Simplified Technical English and these sections in this order: 1. Goal, 2. Open questions (write "None."), 3. Context, 4. Contract, 5. Business rules, 6. Constraints (release target, coding standards, out of scope), 7. Acceptance. The Contract fixes every public name the hidden tests use: object names, method signatures, function module parameters, report parameters, ALV column names, CDS element names. The Business rules are complete and unambiguous. Every rule is checked by at least one hidden test. Do not tell the model HOW to implement (no table kinds, no SQL). Craft decisions belong to the model. - Seed objects: TABL as DDL source ("define table ..."), FUGR without source, FUNC with "functionGroup" and full source including the signature in the FUNCTION statement, classes with full source. Seed table data: a seed class that implements IF_OO_ADT_CLASSRUN, with "run": true. - Hidden tests: one global class, "FOR TESTING DURATION SHORT RISK LEVEL HARMLESS", 5 to 12 test methods. Use only the public contract. Function modules: CALL FUNCTION with EXCEPTIONS. Reports: SUBMIT ... AND RETURN with cl_salv_bs_runtime_info=>set( display = abap_false metadata = abap_false data = abap_true ) and get_data_ref. CDS: cl_cds_test_environment. - Reference solution: correct, Clean ABAP, methods below 40 statements, passes all hidden tests, no ATC priority 1 or 2 findings (for example: pass large parameters by reference). Include the model's expected own tests: for classes a "testclasses_file" (local test classes); for reports local test classes inside the program; for function modules and CDS a global test class. - task.json keys: id, category, object_type, difficulty, release_target, expected_outcome ("implement"), budget {max_tool_calls, max_activations}, seed[], contract[], out_of_scope[], hidden_tests[], reference[], craft_checks[]. Contract entries: CLAS {"implements"} optional; FUNC {"functionGroup", "params":[{"name","type"}]}; PROG {"parameters":[...]}; DDLS {"fields":[...]}. Return ONLY one JSON object, no markdown fence: {"task": , "files": {"": "", ...}} """ def bundle_of(task_dir): files = {} for base, _, names in os.walk(task_dir): for n in names: if n.startswith(".") or n == "generation.json": continue path = os.path.join(base, n) rel = os.path.relpath(path, task_dir) if rel != "task.json": files[rel] = open(path).read() return {"task": json.load(open(os.path.join(task_dir, "task.json"))), "files": files} def chat(model, messages, base_url): # max_tokens: one pilot call ran to 393k output tokens (reasoning) and returned no content body = {"model": model, "messages": messages, "temperature": 0.7, "max_tokens": MAX_OUT_TOKENS} req = urllib.request.Request(f"{base_url}/chat/completions", json.dumps(body).encode(), {"Content-Type": "application/json", "Authorization": "Bearer none"}) for attempt in range(4): try: with urllib.request.urlopen(req, timeout=1800) as r: data = json.loads(r.read().decode()) add_usage(model, data.get("usage", {}), kind="generate") content = data["choices"][0]["message"].get("content") or "" if content.strip(): return content # empty content (output limit reached in reasoning): retry once more with the next attempt if attempt == 3: return "" continue except Exception as e: # noqa: BLE001 if getattr(e, "code", 500) < 500 and getattr(e, "code", 500) != 429: raise time.sleep(15 * (attempt + 1)) raise RuntimeError("generation request failed") def parse_bundle(text): text = text.strip() text = re.sub(r"^```(json)?\s*|\s*```$", "", text) # raw_decode stops after the first complete object; text after it (a note, a second object) is ignored return json.JSONDecoder().raw_decode(text[text.find("{"):])[0] NAME_RE = re.compile(r"^\s*(?:CLASS-)?(?:METHODS|METHOD|DATA|TYPES|CONSTANTS|FORM|CLASS|INTERFACE)\b:?\s+(\w+)", re.I | re.M) # reserved words seen in activation errors, plus common SQL keywords RESERVED = {"HOURS", "MODE", "ORDER", "GROUP", "DATE", "TIME", "VALUE", "USER", "KEY", "COUNT", "SUM", "MIN", "MAX", "AVG", "SELECT", "FROM", "WHERE", "TABLE", "VIEW", "UNION", "JOIN", "LEVEL", "SIZE", "TYPE", "INDEX", "CASE", "WHEN", "THEN", "ELSE", "END", "AS", "ON", "BY", "DESC", "ASC", "DAYS", "MINUTES", "SECONDS", "YEAR", "MONTH", "DAY", "LIMIT", "OFFSET", "PARAMETERS"} GLOBAL_TEST_CLASS = re.compile(r"\A\s*(?:\*[^\n]*\n|\"[^\n]*\n|\s)*CLASS\s+\S+\s+DEFINITION[^.]*FOR\s+TESTING", re.I) DDL_KIND = re.compile(r"^\s*define\s+(?:root\s+)?(table|view|structure)\b", re.I | re.M) def lint_files(b): """Static checks that need no SAP system: identifier length, seed type against DDL source.""" errs = [] for rel, src in b.get("files", {}).items(): if not rel.endswith(".abap"): continue for name in set(NAME_RE.findall(src.replace("{{P}}", "Z0000000_").replace("{{p}}", "z0000000_"))): if len(name) > 30: errs.append(f"{rel}: name {name} has {len(name)} characters (ABAP max 30)") t = b.get("task", {}) oos = {n.upper() for n in t.get("out_of_scope", [])} test_names = {o.get("name", "").upper() for o in t.get("hidden_tests", [])} for o in t.get("reference", []): name = o.get("name", "") if name.upper() in oos: errs.append(f"reference object {name} is also in out_of_scope; the solution must change it") if o.get("type") == "CLAS" and GLOBAL_TEST_CLASS.search(b.get("files", {}).get(o.get("file", ""), "")): test_names.add(name.upper()) for c in t.get("contract", []): if c.get("name", "").upper() in test_names: errs.append(f"contract contains the test class {c.get('name')}; only objects that the hidden " "tests call belong in the contract") for o in t.get("reference", []): if (o.get("type") == "CLAS" and o.get("name", "").upper() not in test_names and not o.get("testclasses_file") and not re.search(r"INHERITING\s+FROM\s+\S*CX_", b.get("files", {}).get(o.get("file", ""), ""), re.I)): errs.append(f"reference class {o.get('name')} has no testclasses_file (local unit tests)") for rel, src in b.get("files", {}).items(): if DDL_KIND.search(src): for f in re.findall(r"^\s*(?:key\s+)?(\w+)\s*:", src, re.I | re.M): if f.upper() in RESERVED: errs.append(f"{rel}: field name {f} is a reserved word; choose another name") for k in ("seed", "reference"): for o in t.get(k, []): m = DDL_KIND.search(b.get("files", {}).get(o.get("file", ""), "")) if not m: continue want = {"table": "TABL", "structure": "TABL", "view": "DDLS"}[m.group(1).lower()] if o.get("type") != want: errs.append(f"{k}: {o.get('name')} has type {o.get('type')}, but its source is " f"'define {m.group(1)}'; use type {want}") return errs def abaplint_files(b): """Parser errors of ABAP sources (classes, interfaces, programs) found by local abaplint. SAP often reports only "save operation failed" for these; abaplint gives the line. Only parser_error: check_syntax gives false errors for standard objects abaplint does not know.""" t, files = b.get("task", {}), b.get("files", {}) ext = {"CLAS": "clas", "INTF": "intf", "PROG": "prog"} version = t.get("release_target") if t.get("release_target") in ABAPLINT_VERSIONS else "v758" d = tempfile.mkdtemp(prefix="genlint_") origin = {} try: os.makedirs(os.path.join(d, "src")) for k in ("seed", "reference", "hidden_tests"): for o in t.get(k, []): if o.get("type") not in ext: continue name = re.sub(r"\W", "_", o.get("name", "").replace("{{P}}", "Z0000000_")).lower() for fk, suffix in (("file", ""), ("testclasses_file", ".testclasses")): if o.get(fk) in files: fn = f"{name}.{ext[o['type']]}{suffix}.abap" origin[fn] = o[fk] src = files[o[fk]].replace("{{P}}", "Z0000000_").replace("{{p}}", "z0000000_") open(os.path.join(d, "src", fn), "w").write(src) if not origin: return [] cfg = {"global": {"files": "/src/**/*.*"}, "dependencies": [], "syntax": {"version": version, "errorNamespace": "^(Z|Y)"}, "rules": {"parser_error": True}} json.dump(cfg, open(os.path.join(d, "abaplint.json"), "w")) p = subprocess.run([ABAPLINT, "abaplint.json", "-f", "json"], cwd=d, capture_output=True, text=True, timeout=300) issues = json.loads(p.stdout or "[]") except Exception: # noqa: BLE001 the pre-check is optional; SAP validation follows return [] finally: shutil.rmtree(d, ignore_errors=True) out = [] for i in issues: f = i.get("file", "") fn = os.path.basename(f.get("filename", "") if isinstance(f, dict) else str(f)) if i.get("key") == "parser_error" and fn in origin: out.append(f"{origin[fn]} line {i.get('start', {}).get('row')}: syntax error for release {version} " f"(abaplint): {i.get('description')}") return out[:20] def check_bundle(b): errs = [] t, files = b.get("task", {}), b.get("files", {}) for k in ("seed", "contract", "hidden_tests", "reference"): if k not in t: errs.append(f"task.json misses '{k}'") if "spec.md" not in files: errs.append("spec.md missing") for k in ("seed", "hidden_tests", "reference"): for o in t.get(k, []): for fk in ("file", "testclasses_file"): if fk in o and o[fk] not in files: errs.append(f"{k}: file {o[fk]} missing") if not o.get("name", "").startswith("{{P}}"): errs.append(f"{k}: name {o.get('name')} does not start with {{{{P}}}}") limit = 16 if o.get("type") == "TABL" else 26 if o.get("type") == "FUGR" else 30 if len(o.get("name", "").replace("{{P}}", "Z0000000_")) > limit: errs.append(f"{k}: name {o.get('name')} too long (max {limit})") if not t.get("hidden_tests"): errs.append("no hidden test class") return errs def write_bundle(b, task_dir, task_id): os.makedirs(task_dir, exist_ok=True) b["task"]["id"] = task_id json.dump(b["task"], open(os.path.join(task_dir, "task.json"), "w"), indent=2) for rel, content in b["files"].items(): path = os.path.join(task_dir, rel) os.makedirs(os.path.dirname(path), exist_ok=True) open(path, "w").write(content) def validate(pool_root, task_id, run_base): r = Runner(pool_root, os.path.join(ROOT, "runs", "gen")) os.makedirs(os.path.join(ROOT, "runs", "gen"), exist_ok=True) rep_o, dir_o = r.run(task_id, OracleAgent(), run_base) rep_o["_dir"] = dir_o rep_n, _ = r.run(task_id, NullAgent(), run_base + 1) return rep_o, rep_n def write_errors(run_dir): """Failed writes and activation messages of the oracle run.""" out = [] path = os.path.join(run_dir or "", "trajectory.jsonl") if not os.path.exists(path): return out for line in open(path): e = json.loads(line) if e.get("tool") in ("sap_create_object", "sap_push_source") and ( e["is_error"] or '"success":false' in e["result"].replace(" ", "")): out.append({"tool": e["tool"], "object": e["args"].get("objectName"), "include": e["args"].get("includeType", "main"), "result": e["result"][:1500]}) return out def failure_summary(rep): out = {"reference_write_errors": write_errors(rep.get("_dir")), "gates": rep.get("gates"), "contract_check": rep.get("contract_check"), "score": rep.get("score"), "setup": rep.get("setup"), "atc": rep.get("atc"), "abaplint": rep.get("abaplint"), "own_tests": rep.get("own_tests"), "hidden_failed": [d for d in rep.get("hidden_tests", {}).get("detail", []) if not d["ok"]], "hidden_install": rep.get("hidden_tests", {}).get("install")} return json.dumps(out)[:6000] def generate(task_id, pool, object_type, category, difficulty, model, base_url, run_base, topic=None, max_repairs=3): check_budget() pool_root = os.path.join(ROOT, "tasks_gen", pool) task_dir = os.path.join(pool_root, task_id) example = bundle_of(os.path.join(ROOT, "tasks", EXAMPLE_FOR[object_type])) ask = (f"Write one new task.\nObject type of the main contract object: {object_type}.\n" f"Skill category {category}: {CATEGORIES[category]}.\nDifficulty {difficulty} of 3.\n" + (f"Topic idea: {topic}\n" if topic else "Choose a new, realistic business topic.\n") + "Here is an example bundle of a different task (same format):\n" + json.dumps(example)) messages = [{"role": "system", "content": SYSTEM}, {"role": "user", "content": ask}] log = {"id": task_id, "pool": pool, "object_type": object_type, "category": category, "attempts": []} for attempt in range(max_repairs + 1): text = chat(model, messages, base_url) messages.append({"role": "assistant", "content": text}) try: b = parse_bundle(text) found = {"structure": check_bundle(b), "static": lint_files(b), "abaplint": abaplint_files(b)} except Exception as e: # noqa: BLE001 b, found = None, {"json": [f"invalid JSON: {e}"]} errs = [e for v in found.values() for e in v] if errs: # one entry per source, so a batch shows what the checks found before SAP log["attempts"].append({"stage": "bundle", "errors": errs, "by_check": {k: len(v) for k, v in found.items() if v}}) messages.append({"role": "user", "content": "Fix these problems and return the full bundle again:\n" + "\n".join(errs)}) continue b["task"].setdefault("object_type", object_type) b["task"].setdefault("category", category) write_bundle(b, task_dir, task_id) rep_o, rep_n = validate(pool_root, task_id, run_base + 2 * attempt) so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total") log["attempts"].append({"stage": "validate", "oracle": so, "null": sn}) if so == 100 and sn == 0: log["accepted"] = True break messages.append({"role": "user", "content": "The harness ran your reference solution (oracle) and an empty solution (null). " f"Required: oracle 100, null 0. Result: oracle {so}, null {sn}.\n" f"Oracle report: {failure_summary(rep_o)}\n" "Fix the bundle (reference, hidden tests, seed, or contract) and return the full " "bundle again."}) else: log["accepted"] = False json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"), indent=1) return log def main(): load_env(os.path.join(ROOT, ".env")) ap = argparse.ArgumentParser() ap.add_argument("--id", required=True) ap.add_argument("--pool", default="eval", choices=["eval", "train"]) ap.add_argument("--object-type", required=True, choices=sorted(EXAMPLE_FOR)) ap.add_argument("--category", required=True, choices=sorted(CATEGORIES)) ap.add_argument("--difficulty", type=int, default=2) ap.add_argument("--topic") ap.add_argument("--model", default="deepseek-v4.1-flash:cloud") ap.add_argument("--base-url", default=os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")) ap.add_argument("--run-base", type=int, required=True) a = ap.parse_args() os.makedirs(os.path.join(ROOT, "tasks_gen", a.pool), exist_ok=True) log = generate(a.id, a.pool, a.object_type, a.category, a.difficulty, a.model, a.base_url, a.run_base, a.topic) print(json.dumps(log)) if __name__ == "__main__": main()