Files
abap-llm/harness/generator.py

563 lines
32 KiB
Python

"""Task generator: a cloud model writes a task bundle; the harness validates it (oracle = 100, null = 0).
python3 -m harness.generator --id G0001 --pool eval --object-type CLAS --category C --difficulty 2
"""
import argparse
import json
import os
import re
import shutil
import subprocess
import tempfile
import time
import urllib.request
from .adt_client import load_env
from .agents import NullAgent, OracleAgent
from .ledger import add_usage, check_budget
from .mutation import check_task
from .runner import Runner
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
ABAPLINT = os.path.join(ROOT, "node_modules", ".bin", "abaplint")
# pilot: accepted final calls used 4k-76k output tokens (p90 57k); one runaway call used 393k
MAX_OUT_TOKENS = 80000
RUNS_PER_ATTEMPT = 10
ABAPLINT_VERSIONS = {"v702", "v740sp05", "v750", "v758"}
EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15"}
CATEGORIES = {
"A": "pure logic in a new class (language, OO design, Clean ABAP)",
"B": "database access (ABAP SQL, CDS) with test doubles",
"C": "internal table craft (table kind, keys, access path)",
"D": "exception design (class-based exceptions, messages)",
"E": "refactoring legacy code: the seed contains legacy code, and the contract object IS this seed object "
"(refactor it in place; same name and same public interface; it is not in out_of_scope). The spec "
"gives the quality goals and says 'the behavior stays the same'. It does not list the business "
"rules: the model must read the legacy code. Hidden tests check the unchanged behavior",
"F": "use an unknown object: the spec names a seed object but not its signature; the model must read it",
"G": "old release target (v702 or v740sp05): no newer syntax",
"I": "bug fix: the seed contains a faulty object, and the contract object IS this seed object (fix it in "
"place; it is not in out_of_scope). The spec describes the symptoms and the expected behavior for "
"them, plus 'all other behavior stays the same'. The spec does not list all rules: the model must "
"read the existing code to keep the other behavior. Hidden tests check the fixed cases and the "
"unchanged behavior",
"H": "stop task: the spec looks complete, but one business rule is missing or two rules contradict, and "
"a correct solution is not possible without an answer. The model must stop, create no object and "
"name the gap. task.json: expected_outcome \"stop\", expected_gap {\"description\": \"<the gap in one "
"sentence>\", \"keywords\": [[\"word\", \"word\"], ...]} (2 to 4 groups; a group matches when all its "
"words occur in a report that names the gap). Keep the contract, seed and a reference of the "
"intended objects (the reference shows the task is otherwise clear). No hidden tests: "
"hidden_tests is []. Do not write the gap in Open questions (write \"None.\")",
}
SYSTEM = """You write evaluation tasks for an ABAP developer model. Each task is a bundle of files.
The model gets only spec.md and works on an SAP ABAP Platform 2025 system (SAP_BASIS 816, client 001)
through ADT tools. The harness installs the seed objects, runs the model, then checks the result with
hidden ABAP Unit tests, ATC, and abaplint.
Rules for the bundle:
- Use the placeholder {{P}} (upper case) and {{p}} (lower case) at the start of EVERY object name.
The harness replaces it with a run prefix of 9 characters (for example Z005P001_).
Name length after replacement: classes, programs, function modules, CDS entities max 30;
database tables max 16; function groups max 26.
- Every other ABAP name (methods, test methods, local classes, data, types, constants) has max 30
characters. Keep test method names short, for example "rejects_zero_qty".
- Method parameters need a complete type: "TYPE c LENGTH 4" is not allowed in a signature.
Declare a type first (TYPES ty_zone TYPE c LENGTH 4) and use it.
- Field names in tables and CDS views are not SQL reserved words (no HOURS, MODE, ORDER, DATE, COUNT ...).
- CDS: use "define view entity". Parameters have no default value. Use a parameter as
$parameters.p_name (not :p_name). In a UNION, all branches have the same
key elements and element names, and the view
has the annotation @Metadata.ignorePropagatedAnnotations: true. Hidden tests for a view with parameters pass all parameters.
- Legacy seed code (categories E, I) must activate on SAP_BASIS 816: do not mix old and new syntax in one
SQL statement (with new syntax, every host variable needs "@").
- The contract lists only the objects that the hidden tests call. Test classes are never in the contract.
Every reference class (except exception classes) has a "testclasses_file".
- A database table seed has type "TABL" and DDL source "define table ...". Type "DDLS" is only for CDS views.
- All objects are in package $TMP. Do not use transports.
- Do not use SAP application module data (no FI, SD, MM tables). Use generic business domains and
only objects that exist in every ABAP Platform system (language, ABAP SQL, CDS, CL_ABAP_*, CL_SALV_TABLE,
CL_OSQL_TEST_ENVIRONMENT, CL_CDS_TEST_ENVIRONMENT). Seed your own tables and data if needed.
- No dynpro (CALL SCREEN), no SmartForms, no BAdI, no RAP behavior definitions.
- spec.md uses Simplified Technical English and these sections in this order:
1. Goal, 2. Open questions (write "None."), 3. Context, 4. Contract, 5. Business rules,
6. Constraints (release target, coding standards, out of scope), 7. Acceptance.
The Contract fixes every public name the hidden tests use: object names, method signatures,
function module parameters, report parameters, ALV column names, CDS element names.
The Business rules are complete and unambiguous. Every rule is checked by at least one hidden test.
Do not tell the model HOW to implement (no table kinds, no SQL). Craft decisions belong to the model.
- Seed objects: TABL as DDL source ("define table ..."), FUGR without source, FUNC with "functionGroup"
and full source including the signature in the FUNCTION statement, classes with full source.
Seed table data: a seed class that implements IF_OO_ADT_CLASSRUN, with "run": true.
- Hidden tests: one global class, "FOR TESTING DURATION SHORT RISK LEVEL HARMLESS", 5 to 12 test
methods. Use only the public contract. Function modules: CALL FUNCTION with EXCEPTIONS.
Reports: SUBMIT ... AND RETURN with cl_salv_bs_runtime_info=>set( display = abap_false
metadata = abap_false data = abap_true ) and get_data_ref. CDS: cl_cds_test_environment.
- Reference solution: correct, Clean ABAP, methods below 40 statements, passes all hidden tests,
no ATC priority 1 or 2 findings (for example: pass large parameters by reference).
Include the model's expected own tests: for classes a "testclasses_file" (local test classes);
for reports local test classes inside the program; for function modules and CDS a global test class.
- task.json keys: id, category, object_type, difficulty, release_target, expected_outcome ("implement"),
budget {max_tool_calls, max_activations}, seed[], contract[], out_of_scope[], hidden_tests[],
reference[], craft_checks[]. Contract entries: CLAS {"implements"} optional; FUNC {"functionGroup",
"params":[{"name","type"}]}; PROG {"parameters":[...]}; DDLS {"fields":[...]}.
Return ONLY one JSON object, no markdown fence:
{"task": <task.json object>, "files": {"<relative path>": "<file content>", ...}}
"""
def bundle_of(task_dir):
files = {}
for base, _, names in os.walk(task_dir):
for n in names:
if n.startswith(".") or n == "generation.json":
continue
path = os.path.join(base, n)
rel = os.path.relpath(path, task_dir)
if rel != "task.json":
files[rel] = open(path).read()
return {"task": json.load(open(os.path.join(task_dir, "task.json"))), "files": files}
def chat(model, messages, base_url):
# max_tokens: one pilot call ran to 393k output tokens (reasoning) and returned no content
body = {"model": model, "messages": messages, "temperature": 0.7, "max_tokens": MAX_OUT_TOKENS}
req = urllib.request.Request(f"{base_url}/chat/completions", json.dumps(body).encode(),
{"Content-Type": "application/json", "Authorization": "Bearer none"})
for attempt in range(4):
try:
with urllib.request.urlopen(req, timeout=1800) as r:
data = json.loads(r.read().decode())
add_usage(model, data.get("usage", {}), kind="generate")
content = data["choices"][0]["message"].get("content") or ""
if content.strip():
return content
# empty content (output limit reached in reasoning): retry once more with the next attempt
if attempt == 3:
return ""
continue
except Exception as e: # noqa: BLE001
if getattr(e, "code", 500) < 500 and getattr(e, "code", 500) != 429:
raise
time.sleep(15 * (attempt + 1))
raise RuntimeError("generation request failed")
def parse_bundle(text):
text = text.strip()
text = re.sub(r"^```(json)?\s*|\s*```$", "", text)
# raw_decode stops after the first complete object; text after it (a note, a second object) is ignored
return json.JSONDecoder().raw_decode(text[text.find("{"):])[0]
NAME_RE = re.compile(r"^\s*(?:CLASS-)?(?:METHODS|METHOD|DATA|TYPES|CONSTANTS|FORM|CLASS|INTERFACE)\b:?\s+(\w+)",
re.I | re.M)
# reserved words seen in activation errors, plus common SQL keywords
RESERVED = {"HOURS", "MODE", "ORDER", "GROUP", "DATE", "TIME", "VALUE", "USER", "KEY", "COUNT", "SUM", "MIN",
"MAX", "AVG", "SELECT", "FROM", "WHERE", "TABLE", "VIEW", "UNION", "JOIN", "LEVEL", "SIZE", "TYPE",
"INDEX", "CASE", "WHEN", "THEN", "ELSE", "END", "AS", "ON", "BY", "DESC", "ASC", "DAYS", "MINUTES",
"SECONDS", "YEAR", "MONTH", "DAY", "LIMIT", "OFFSET", "PARAMETERS"}
GLOBAL_TEST_CLASS = re.compile(r"\A\s*(?:\*[^\n]*\n|\"[^\n]*\n|\s)*CLASS\s+\S+\s+DEFINITION[^.]*FOR\s+TESTING",
re.I)
DDL_KIND = re.compile(r"^\s*define\s+(?:root\s+)?(table|view|structure)\b", re.I | re.M)
def lint_files(b):
"""Static checks that need no SAP system: identifier length, seed type against DDL source."""
errs = []
for rel, src in b.get("files", {}).items():
if not rel.endswith(".abap"):
continue
for name in set(NAME_RE.findall(src.replace("{{P}}", "Z0000000_").replace("{{p}}", "z0000000_"))):
if len(name) > 30:
errs.append(f"{rel}: name {name} has {len(name)} characters (ABAP max 30)")
t = b.get("task", {})
oos = {n.upper() for n in t.get("out_of_scope", [])}
if t.get("category") in ("E", "I"):
seed_names = {o.get("name", "").upper() for o in t.get("seed", [])}
if not any(c.get("name", "").upper() in seed_names for c in t.get("contract", [])):
errs.append(f"category {t['category']}: the contract object must be the seed object that the model "
"changes in place (same name in seed, contract and reference)")
test_names = {o.get("name", "").upper() for o in t.get("hidden_tests", [])}
for o in t.get("reference", []):
name = o.get("name", "")
if name.upper() in oos:
errs.append(f"reference object {name} is also in out_of_scope; the solution must change it")
if o.get("type") == "CLAS" and GLOBAL_TEST_CLASS.search(b.get("files", {}).get(o.get("file", ""), "")):
test_names.add(name.upper())
for c in t.get("contract", []):
if c.get("name", "").upper() in test_names:
errs.append(f"contract contains the test class {c.get('name')}; only objects that the hidden "
"tests call belong in the contract")
for o in t.get("reference", []):
if (o.get("type") == "CLAS" and o.get("name", "").upper() not in test_names
and not o.get("testclasses_file")
and not re.search(r"INHERITING\s+FROM\s+\S*CX_", b.get("files", {}).get(o.get("file", ""), ""),
re.I)):
errs.append(f"reference class {o.get('name')} has no testclasses_file (local unit tests)")
for rel, src in b.get("files", {}).items():
# G0121, G0122: SAP says only "Can't save due to errors in source" for these
if re.search(r"^\s*define\s+table\b", src, re.I | re.M):
for ann in ("tableCategory", "deliveryClass"):
if not re.search(rf"@AbapCatalog\.{ann}\s*:", src, re.I):
errs.append(f"{rel}: a table needs the annotation @AbapCatalog.{ann} "
"(for example #TRANSPARENT, #A)")
if DDL_KIND.search(src):
for f in re.findall(r"^\s*(?:key\s+)?(\w+)\s*:", src, re.I | re.M):
if f.upper() in RESERVED:
errs.append(f"{rel}: field name {f} is a reserved word; choose another name")
for rel, src in b.get("files", {}).items():
if rel.endswith(".asddls") and re.search(r"define\s+(root\s+)?view\s+entity", src, re.I):
if re.search(r"\bunion\b", src, re.I) and not re.search(r"@Metadata\.ignorePropagatedAnnotations\s*:\s*true",
src, re.I):
errs.append(f"{rel}: a view entity with UNION needs @Metadata.ignorePropagatedAnnotations: true")
for p in sorted(set(re.findall(r"[=<>(,]\s*:(\w+)|\bbetween\s+:(\w+)|\band\s+:(\w+)", src, re.I))):
name = next(x for x in p if x)
errs.append(f"{rel}: parameter :{name}; a view entity needs $parameters.{name}")
for k in ("seed", "reference"):
for o in t.get(k, []):
m = DDL_KIND.search(b.get("files", {}).get(o.get("file", ""), ""))
if not m:
continue
want = {"table": "TABL", "structure": "TABL", "view": "DDLS"}[m.group(1).lower()]
if o.get("type") != want:
errs.append(f"{k}: {o.get('name')} has type {o.get('type')}, but its source is "
f"'define {m.group(1)}'; use type {want}")
return errs
def abaplint_files(b):
"""Parser errors of ABAP sources (classes, interfaces, programs) found by local abaplint.
SAP often reports only "save operation failed" for these; abaplint gives the line.
Only parser_error: check_syntax gives false errors for standard objects abaplint does not know."""
t, files = b.get("task", {}), b.get("files", {})
ext = {"CLAS": "clas", "INTF": "intf", "PROG": "prog"}
version = t.get("release_target") if t.get("release_target") in ABAPLINT_VERSIONS else "v758"
d = tempfile.mkdtemp(prefix="genlint_")
origin = {}
try:
os.makedirs(os.path.join(d, "src"))
for k in ("seed", "reference", "hidden_tests"):
for o in t.get(k, []):
if o.get("type") not in ext:
continue
name = re.sub(r"\W", "_", o.get("name", "").replace("{{P}}", "Z0000000_")).lower()
for fk, suffix in (("file", ""), ("testclasses_file", ".testclasses")):
if o.get(fk) in files:
fn = f"{name}.{ext[o['type']]}{suffix}.abap"
origin[fn] = o[fk]
src = files[o[fk]].replace("{{P}}", "Z0000000_").replace("{{p}}", "z0000000_")
open(os.path.join(d, "src", fn), "w").write(src)
if not origin:
return []
cfg = {"global": {"files": "/src/**/*.*"}, "dependencies": [],
"syntax": {"version": version, "errorNamespace": "^(Z|Y)"}, "rules": {"parser_error": True}}
json.dump(cfg, open(os.path.join(d, "abaplint.json"), "w"))
p = subprocess.run([ABAPLINT, "abaplint.json", "-f", "json"], cwd=d, capture_output=True, text=True,
timeout=300)
issues = json.loads(p.stdout or "[]")
except Exception: # noqa: BLE001 the pre-check is optional; SAP validation follows
return []
finally:
shutil.rmtree(d, ignore_errors=True)
out = []
for i in issues:
f = i.get("file", "")
fn = os.path.basename(f.get("filename", "") if isinstance(f, dict) else str(f))
if i.get("key") == "parser_error" and fn in origin:
out.append(f"{origin[fn]} line {i.get('start', {}).get('row')}: syntax error for release {version} "
f"(abaplint): {i.get('description')}")
return out[:20]
def check_bundle(b):
errs = []
t, files = b.get("task", {}), b.get("files", {})
stop = t.get("expected_outcome") == "stop"
for k in ("seed", "contract", "hidden_tests", "reference"):
if k not in t:
errs.append(f"task.json misses '{k}'")
if stop:
gap = t.get("expected_gap") or {}
if not gap.get("description") or not gap.get("keywords"):
errs.append("stop task: expected_gap needs description and keywords")
if "spec.md" not in files:
errs.append("spec.md missing")
for rel in files: # G0106: a key "seed/" (a directory) crashed write_bundle
if not rel or rel.endswith("/") or rel.startswith("/") or ".." in rel.split("/"):
errs.append(f"files: '{rel}' is not a valid relative file path")
for k in ("seed", "hidden_tests", "reference"):
for o in t.get(k, []):
for fk in ("file", "testclasses_file"):
if fk in o and o[fk] not in files:
errs.append(f"{k}: file {o[fk]} missing")
if not o.get("name", "").startswith("{{P}}"):
errs.append(f"{k}: name {o.get('name')} does not start with {{{{P}}}}")
limit = 16 if o.get("type") == "TABL" else 26 if o.get("type") == "FUGR" else 30
if len(o.get("name", "").replace("{{P}}", "Z0000000_")) > limit:
errs.append(f"{k}: name {o.get('name')} too long (max {limit})")
if not t.get("hidden_tests") and not stop:
errs.append("no hidden test class")
return errs
def dependency_order(objs, files):
"""Stable order in which every object comes after the objects that its sources name
(for example exception classes before the class that raises them). A cycle keeps the given order."""
def text(o):
return " ".join(files.get(o.get(k), "") for k in ("file", "testclasses_file")).upper()
names = [o.get("name", "").upper() for o in objs]
needs = []
for o in objs:
src, own = text(o), o.get("name", "").upper()
dep = {n for n in names if n != own and re.search(rf"(?<![\w{{}}]){re.escape(n)}(?!\w)", src)}
if o.get("functionGroup"):
dep.add(o["functionGroup"].upper())
needs.append(dep & set(names))
out, done = [], set()
while len(out) < len(objs):
ready = [i for i, o in enumerate(objs) if i not in done and needs[i] <= {names[j] for j in done}]
i = ready[0] if ready else min(set(range(len(objs))) - done)
done.add(i)
out.append(objs[i])
return out
def floor_budget(task):
"""The model needs more calls than the oracle (reads, repairs). G0022: budget 12 activations,
reference alone 13. Minimum: 2 x oracle activations, 3 x oracle tool calls."""
ref = task.get("reference", [])
pushes = sum(bool(o.get("file")) + bool(o.get("testclasses_file")) for o in ref)
calls = len(ref) + pushes
bud = task.setdefault("budget", {})
# absolute floor (design default 60): with 40 calls DeepSeek used the budget to read the seed tables and
# stopped before its own tests (empirical runs G0017, G0024)
bud["max_activations"] = max(int(bud.get("max_activations", 0)), 2 * pushes, 15)
bud["max_tool_calls"] = max(int(bud.get("max_tool_calls", 0)), 3 * calls, 60)
KEEP_FILES = {"generation.json", "review.json"}
def write_bundle(b, task_dir, task_id):
# remove the files of an earlier attempt (G0003 kept an unused seed/course.ddls.abap)
if os.path.isdir(task_dir):
for name in os.listdir(task_dir):
if name not in KEEP_FILES:
path = os.path.join(task_dir, name)
shutil.rmtree(path) if os.path.isdir(path) else os.remove(path)
os.makedirs(task_dir, exist_ok=True)
b["task"]["id"] = task_id
for k in ("seed", "reference"):
b["task"][k] = dependency_order(b["task"].get(k, []), b["files"])
floor_budget(b["task"])
json.dump(b["task"], open(os.path.join(task_dir, "task.json"), "w"), indent=2)
for rel, content in b["files"].items():
path = os.path.join(task_dir, rel)
os.makedirs(os.path.dirname(path), exist_ok=True)
open(path, "w").write(content)
def validate(pool_root, task_id, run_base):
r = Runner(pool_root, os.path.join(ROOT, "runs", "gen"))
os.makedirs(os.path.join(ROOT, "runs", "gen"), exist_ok=True)
rep_o, dir_o = r.run(task_id, OracleAgent(), run_base)
rep_o["_dir"] = dir_o
rep_n, _ = r.run(task_id, NullAgent(), run_base + 1)
return rep_o, rep_n
def write_errors(run_dir):
"""Failed writes and activation messages of the oracle run."""
out = []
path = os.path.join(run_dir or "", "trajectory.jsonl")
if not os.path.exists(path):
return out
for line in open(path):
e = json.loads(line)
if e.get("tool") in ("sap_create_object", "sap_push_source") and (
e["is_error"] or '"success":false' in e["result"].replace(" ", "")):
out.append({"tool": e["tool"], "object": e["args"].get("objectName"),
"include": e["args"].get("includeType", "main"), "result": e["result"][:1500]})
return out
def failure_summary(rep):
out = {"reference_write_errors": write_errors(rep.get("_dir")), "gates": rep.get("gates"),
"contract_check": rep.get("contract_check"), "score": rep.get("score"), "setup": rep.get("setup"),
"atc": rep.get("atc"), "abaplint": rep.get("abaplint"), "own_tests": rep.get("own_tests"),
"hidden_failed": [d for d in rep.get("hidden_tests", {}).get("detail", []) if not d["ok"]],
"hidden_install": rep.get("hidden_tests", {}).get("install")}
return json.dumps(out)[:6000]
def generate(task_id, pool, object_type, category, difficulty, model, base_url, run_base, topic=None,
max_repairs=3, extra_check=None):
"""extra_check(bundle) -> list of problems; it runs with the static checks (training mode: overlap)."""
check_budget()
pool_root = os.path.join(ROOT, "tasks_gen", pool)
task_dir = os.path.join(pool_root, task_id)
example = bundle_of(os.path.join(ROOT, "tasks", EXAMPLE_FOR[object_type]))
ask = (f"Write one new task.\nObject type of the main contract object: {object_type}.\n"
f"Skill category {category}: {CATEGORIES[category]}.\nDifficulty {difficulty} of 3.\n"
+ (f"Topic idea: {topic}\n" if topic else "Choose a new, realistic business topic.\n")
+ "Here is an example bundle of a different task (same format):\n" + json.dumps(example))
messages = [{"role": "system", "content": SYSTEM}, {"role": "user", "content": ask}]
log = {"id": task_id, "pool": pool, "object_type": object_type, "category": category, "attempts": []}
for attempt in range(max_repairs + 1):
text = chat(model, messages, base_url)
messages.append({"role": "assistant", "content": text})
try:
b = parse_bundle(text)
b.get("task", {}).setdefault("category", category)
found = {"structure": check_bundle(b), "static": lint_files(b), "abaplint": abaplint_files(b)}
if extra_check and not found["structure"]:
found["extra"] = extra_check(b)
except Exception as e: # noqa: BLE001
b, found = None, {"json": [f"invalid JSON: {e}"]}
errs = [e for v in found.values() for e in v]
if errs:
# one entry per source, so a batch shows what the checks found before SAP
log["attempts"].append({"stage": "bundle", "errors": errs,
"by_check": {k: len(v) for k, v in found.items() if v}})
messages.append({"role": "user", "content": "Fix these problems and return the full bundle again:\n"
+ "\n".join(errs)})
continue
b["task"].setdefault("object_type", object_type)
b["task"].setdefault("category", category)
write_bundle(b, task_dir, task_id)
# run numbers per attempt: +0 oracle, +1 null, +2..+6 mutants (RUNS_PER_ATTEMPT)
base = run_base + RUNS_PER_ATTEMPT * attempt
rep_o, rep_n = validate(pool_root, task_id, base)
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
if category == "H": # oracle stops with the gap: 100; null stops without a report: 30
if so == 100 and sn == 30:
log["accepted"] = True
break
messages.append({"role": "user", "content":
f"Stop task check: required oracle 100 and null 30. Result: oracle {so}, null {sn}. "
f"Oracle report: {failure_summary(rep_o)}\nFix expected_gap (description, keywords) "
"and return the full bundle again."})
continue
if so == 100 and sn == 0:
mut = check_task(pool_root, task_id, base + 2, keep=True)
log["attempts"][-1]["mutation"] = {k: mut[k] for k in ("valid", "killed", "ok")}
if mut["ok"]:
log["accepted"] = True
break
survived = [f"{m['object']} {m['mutant']}" for m in mut["mutants"] if m["status"] == "survived"]
shutil.rmtree(os.path.join(task_dir, "faulty"), ignore_errors=True)
messages.append({"role": "user", "content":
"Oracle 100 and null 0: good. But the hidden tests are too weak. The harness changed "
"the reference (mutation check) and all hidden tests still passed for these changes:\n"
+ "\n".join(survived or ["(fewer than 2 changes possible: add more business logic "
"checks to the hidden tests)"])
+ "\nAdd or improve hidden tests so that each of these changes makes a test fail. "
"If a change does not change the behavior, ignore it. Do not change the behavior "
"of the reference. Return the full bundle again."})
continue
messages.append({"role": "user", "content":
"The harness ran your reference solution (oracle) and an empty solution (null). "
f"Required: oracle 100, null 0. Result: oracle {so}, null {sn}.\n"
f"Oracle report: {failure_summary(rep_o)}\n"
"Fix the bundle (reference, hidden tests, seed, or contract) and return the full "
"bundle again."})
else:
log["accepted"] = False
json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"),
indent=1)
return log
K_STYLES = {
"free_text": "Rewrite the spec as a short free-text request from a functional consultant: no section "
"headings, no numbered rules, plain sentences, as in an e-mail. Keep every business rule.",
"incomplete": "Rewrite the spec as a short free-text request. Leave out the craft hints and the context "
"that a good ABAP developer can find in the system (for example how the seed table looks). "
"Keep every business rule; the task must stay solvable without questions.",
}
def contract_terms(task):
"""Names that a K spec must keep: object names, FM parameters, report parameters, CDS elements."""
out = []
for c in task.get("contract", []):
out.append(c["name"])
out += [p["name"] if isinstance(p, dict) else p for p in c.get("params", []) + c.get("parameters", [])]
out += list(c.get("fields", []))
return out
def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2):
"""Category K: free-text or incomplete input + other tool schema. Same reference and hidden tests
as the source task; only spec.md, category and tool_schema change."""
check_budget()
pool_root = os.path.join(ROOT, "tasks_gen", "eval")
src_dir, task_dir = os.path.join(pool_root, src_id), os.path.join(pool_root, new_id)
b = bundle_of(src_dir)
b["files"] = {k: v for k, v in b["files"].items()
if not (k.startswith("faulty/") or k in ("mutation.json", "review.json", "empirical.json"))}
spec = b["files"]["spec.md"]
terms = contract_terms(b["task"])
messages = [{"role": "user", "content":
f"{K_STYLES[style]}\nKeep every contract item: the output form (for example an ALV list "
"with CL_SALV_TABLE), the selection screen, the signature, and the column names (G0181 lost the "
"ALV). Keep these names exactly as written (the tests use them): "
f"{', '.join(terms)}. Keep the placeholder {{{{P}}}} in names. Use Simplified Technical English. "
f"Return only the new spec text.\n\nSpec:\n{spec}"}]
log = {"id": new_id, "pool": "eval", "category": "K", "base_task": src_id, "style": style,
"tool_schema": tool_schema, "attempts": []}
for attempt in range(max_repairs + 1):
text = chat(model, messages, base_url).strip()
text = re.sub(r"^```\w*\s*|\s*```$", "", text)
messages.append({"role": "assistant", "content": text})
missing = [t for t in terms if t.upper() not in text.upper()]
if missing:
log["attempts"].append({"stage": "spec", "missing_names": missing})
messages.append({"role": "user", "content": "These names are missing: " + ", ".join(missing)
+ ". Return the full spec again with all names."})
continue
b["files"]["spec.md"] = text + "\n"
b["task"].update(category="K", base_task=src_id, input_style=style, tool_schema=tool_schema)
write_bundle(b, task_dir, new_id)
rep_o, rep_n = validate(pool_root, new_id, run_base + RUNS_PER_ATTEMPT * attempt)
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
log["accepted"] = so == 100 and sn == 0
break
else:
log["accepted"] = False
json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"),
indent=1)
return log
def main():
load_env(os.path.join(ROOT, ".env"))
ap = argparse.ArgumentParser()
ap.add_argument("--id", required=True)
ap.add_argument("--pool", default="eval", choices=["eval", "train"])
ap.add_argument("--object-type", choices=sorted(EXAMPLE_FOR))
ap.add_argument("--category", choices=sorted(CATEGORIES))
ap.add_argument("--difficulty", type=int, default=2)
ap.add_argument("--topic")
ap.add_argument("--model", default="deepseek-v4.1-flash:cloud")
ap.add_argument("--base-url", default=os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"))
ap.add_argument("--run-base", type=int, required=True)
ap.add_argument("--k-from", help="category K: make a variant of this accepted eval task")
ap.add_argument("--k-style", default="free_text", choices=sorted(K_STYLES))
a = ap.parse_args()
if a.k_from:
print(json.dumps(make_k_variant(a.k_from, a.id, a.k_style, a.model, a.base_url, a.run_base)))
return
os.makedirs(os.path.join(ROOT, "tasks_gen", a.pool), exist_ok=True)
log = generate(a.id, a.pool, a.object_type, a.category, a.difficulty, a.model, a.base_url,
a.run_base, a.topic)
print(json.dumps(log))
if __name__ == "__main__":
main()