635 lines
38 KiB
Python
635 lines
38 KiB
Python
"""Task generator: a cloud model writes a task bundle; the harness validates it (oracle = 100, null = 0).
|
|
|
|
python3 -m harness.generator --id G0001 --pool eval --object-type CLAS --category C --difficulty 2
|
|
"""
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import tempfile
|
|
import time
|
|
import urllib.request
|
|
|
|
from .adt_client import load_env
|
|
from .agents import NullAgent, OracleAgent
|
|
from .ledger import add_usage, check_budget
|
|
from .mutation import check_task
|
|
from .runner import Runner
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
ABAPLINT = os.path.join(ROOT, "node_modules", ".bin", "abaplint")
|
|
# pilot: accepted final calls used 4k-76k output tokens (p90 57k); one runaway call used 393k
|
|
MAX_OUT_TOKENS = 80000
|
|
RUNS_PER_ATTEMPT = 10
|
|
ABAPLINT_VERSIONS = {"v702", "v740sp05", "v750", "v758"}
|
|
EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15",
|
|
"STRU": "T15", "MSAG": "T13"}
|
|
CATEGORIES = {
|
|
"A": "pure logic in a new class (language, OO design, Clean ABAP)",
|
|
"B": "database access (ABAP SQL, CDS) with test doubles",
|
|
"C": "internal table craft (table kind, keys, access path)",
|
|
"D": "exception design (class-based exceptions, messages)",
|
|
"E": "refactoring legacy code: the seed contains legacy code, and the contract object IS this seed object "
|
|
"(refactor it in place; same name and same public interface; it is not in out_of_scope). The spec "
|
|
"gives the quality goals and says 'the behavior stays the same'. It does not list the business "
|
|
"rules: the model must read the legacy code. Hidden tests check the unchanged behavior",
|
|
"F": "use an unknown object: the spec names a seed object but not its signature; the model must read it",
|
|
"G": "old release target (v702 or v740sp05): no newer syntax",
|
|
"I": "bug fix: the seed contains a faulty object, and the contract object IS this seed object (fix it in "
|
|
"place; it is not in out_of_scope). The spec describes the symptoms and the expected behavior for "
|
|
"them, plus 'all other behavior stays the same'. The spec does not list all rules: the model must "
|
|
"read the existing code to keep the other behavior. Hidden tests check the fixed cases and the "
|
|
"unchanged behavior",
|
|
"H": "stop task: the spec looks complete, but one business rule is missing or two rules contradict, and "
|
|
"a correct solution is not possible without an answer. The model must stop, create no object and "
|
|
"name the gap. task.json: expected_outcome \"stop\", expected_gap {\"description\": \"<the gap in one "
|
|
"sentence>\", \"keywords\": [[\"word\", \"word\"], ...]} (2 to 4 groups; a group matches when all its "
|
|
"words occur in a report that names the gap). Keep the contract, seed and a reference of the "
|
|
"intended objects (the reference shows the task is otherwise clear). No hidden tests: "
|
|
"hidden_tests is []. Do not write the gap in Open questions (write \"None.\")",
|
|
}
|
|
|
|
# Extra format notes for object types that the main prompt does not describe (training mode, 2026-10-05).
|
|
# The example bundle is of another type; these notes say how this type differs.
|
|
TYPE_NOTES = {
|
|
"INTF": """The model must create an INTERFACE (type INTF) as the main contract object: the methods, types and constants
|
|
named in the spec. Contract entry: {"type": "INTF", "name": "{{P}}IF_X"}. Reference: the interface source
|
|
("reference/x.intf.abap") and ONE global own-test class (type CLAS, source with a global class DEFINITION FOR TESTING;
|
|
an interface has no test include). Hidden tests: one global test class with a local class that IMPLEMENTS the
|
|
interface (it only compiles when every signature matches), calls the methods through the interface, reads the
|
|
constants (values), and checks type lengths with RTTI (cl_abap_typedescr=>describe_by_name). The business rules give the
|
|
exact constant values, parameter names and types. No other object is needed.""",
|
|
"TABL": """The model must create a transparent DATABASE TABLE (type TABL) as the main contract object. The table name has at
|
|
most 16 characters including the prefix: use a short name, for example {{P}}ORD (7 characters after the prefix at most).
|
|
Source is DDL: annotations @EndUserText.label, @AbapCatalog.enhancement.category : #NOT_EXTENSIBLE,
|
|
@AbapCatalog.tableCategory : #TRANSPARENT, @AbapCatalog.deliveryClass : #A, @AbapCatalog.dataMaintenance : #RESTRICTED,
|
|
then "define table {{p}}ord { key client : abap.clnt not null; key item_id : abap.char(10) not null; ... }".
|
|
Contract entry: {"type": "TABL", "name": "{{P}}ORD", "fields": ["client", "item_id", ...]} (all fields, lower case).
|
|
The spec (Business rules) lists every field with its data type and length, the key fields, and not-null fields. Use
|
|
only built-in types (abap.char, abap.numc, abap.dec, abap.int4, abap.dats). Avoid abap.curr and abap.quan; if the spec
|
|
needs an amount or a quantity, the reference field needs a reference annotation in the form 'tablename.fieldname'
|
|
(for example @Semantics.amount.currencyCode : '{{p}}ord.currency_code' on the amount and a field currency_code :
|
|
abap.cuky in the same table; @Semantics.quantity.unitOfMeasure : '{{p}}ord.unit' with unit : abap.unit(3)); a
|
|
reference without the table name fails with "Annotation with reference to unit code ... is uncomplete". Reference: the DDL file and ONE global own-test class
|
|
(type CLAS). Hidden tests (one global class): RTTI on the table: cast cl_abap_typedescr=>describe_by_name( '{{P}}ORD' ) to
|
|
cl_abap_structdescr and check get_ddic_field_list( ) (field names, key flags, lengths, decimals, types), and a
|
|
cl_osql_test_environment test (create( i_dependency_list = VALUE #( ( '{{P}}ORD' ) ) ), insert, select back; a second
|
|
INSERT with the same key gives sy-subrc = 4). No seed is needed.""",
|
|
"STRU": """The model must create a DDIC STRUCTURE (type STRU) as the main contract object. DDL source: annotations
|
|
@EndUserText.label and @AbapCatalog.enhancement.category : #NOT_EXTENSIBLE, then
|
|
"define structure {{p}}name { code : abap.char(4); amount : abap.dec(9,2); ... }" (include another structure with
|
|
"include {{p}}other;" when the spec asks). Contract entry: {"type": "STRU", "name": "{{P}}NAME", "fields": [...]}
|
|
(the field names in lower case). The Business rules list every field with data type and length. Reference: the DDL file
|
|
and ONE global own-test class (type CLAS). Hidden tests (one global class): RTTI: cl_abap_typedescr=>describe_by_name(
|
|
'{{P}}NAME' ) cast to cl_abap_structdescr; check components (names, length, decimals, type kind) and the order.""",
|
|
"MSAG": """The model must create a MESSAGE CLASS (type MSAG) as the main contract object. The name has at most 20
|
|
characters including the prefix. A message class has NO source file. In task.json the reference entry has no "file":
|
|
{"type": "MSAG", "name": "{{P}}MSG", "description": "...", "messages": [{"msgno": "001", "text": "Order &1 is blocked"}, ...]}
|
|
and the contract entry is {"type": "MSAG", "name": "{{P}}MSG", "messages": [{"msgno": "001"}, ...]} (the numbers that
|
|
must exist). The Business rules give for each message the number, the exact text with the placeholders &1 to &4 (at
|
|
most 72 characters), and when it is used (error, warning, information). Reference: the message class entry and ONE global
|
|
own-test class (type CLAS) that reads the messages. Hidden tests (one global class): for each message use
|
|
MESSAGE ID '{{P}}MSG' TYPE 'E' NUMBER '001' WITH 'A' 'B' INTO DATA(lv_text) and assert the final text; also check one
|
|
placeholder substitution. The model may also be asked for an exception class that uses the message class (then both
|
|
are contract entries and the exception class is a reference file).""",
|
|
"EXC": """The main contract object is a class-based EXCEPTION class (CLAS, name {{P}}CX_...) that inherits from
|
|
CX_STATIC_CHECK, CX_DYNAMIC_CHECK or CX_NO_CHECK. The spec asks for constants for the error cases, attributes with
|
|
context data, a constructor with these parameters, and a message text (through IF_T100_DYN_MSG / IF_T100_MESSAGE with
|
|
a message class that is a seed object, or through a redefined get_text). Hidden tests raise the exception from a small
|
|
seed class, catch it and check the attributes and the text (get_text( )). Reference: the exception class with its
|
|
testclasses_file or ONE global own-test class (an exception class has few methods).""",
|
|
}
|
|
|
|
SYSTEM = """You write evaluation tasks for an ABAP developer model. Each task is a bundle of files.
|
|
The model gets only spec.md and works on an SAP ABAP Platform 2025 system (SAP_BASIS 816, client 001)
|
|
through ADT tools. The harness installs the seed objects, runs the model, then checks the result with
|
|
hidden ABAP Unit tests, ATC, and abaplint.
|
|
|
|
Rules for the bundle:
|
|
- Use the placeholder {{P}} (upper case) and {{p}} (lower case) at the start of EVERY object name.
|
|
The harness replaces it with a run prefix of 9 characters (for example Z005P001_).
|
|
Name length after replacement: classes, programs, function modules, CDS entities max 30;
|
|
database tables max 16; function groups max 26.
|
|
- Every other ABAP name (methods, test methods, local classes, data, types, constants) has max 30
|
|
characters. Keep test method names short, for example "rejects_zero_qty".
|
|
- Method parameters need a complete type: "TYPE c LENGTH 4" is not allowed in a signature.
|
|
Declare a type first (TYPES ty_zone TYPE c LENGTH 4) and use it.
|
|
- Field names in tables and CDS views are not SQL reserved words (no HOURS, MODE, ORDER, DATE, COUNT ...).
|
|
- CDS: use "define view entity". Parameters have no default value. Use a parameter as
|
|
$parameters.p_name (not :p_name). In a UNION, all branches have the same
|
|
key elements and element names, and the view
|
|
has the annotation @Metadata.ignorePropagatedAnnotations: true. Hidden tests for a view with parameters pass all parameters.
|
|
- Legacy seed code (categories E, I) must activate on SAP_BASIS 816: do not mix old and new syntax in one
|
|
SQL statement (with new syntax, every host variable needs "@").
|
|
- The contract lists only the objects that the hidden tests call. Test classes are never in the contract.
|
|
Every reference class (except exception classes) has a "testclasses_file".
|
|
- A database table seed has type "TABL" and DDL source "define table ...". Type "DDLS" is only for CDS views.
|
|
- All objects are in package $TMP. Do not use transports.
|
|
- Do not use SAP application module data (no FI, SD, MM tables). Use generic business domains and
|
|
only objects that exist in every ABAP Platform system (language, ABAP SQL, CDS, CL_ABAP_*, CL_SALV_TABLE,
|
|
CL_OSQL_TEST_ENVIRONMENT, CL_CDS_TEST_ENVIRONMENT). Seed your own tables and data if needed.
|
|
- No dynpro (CALL SCREEN), no SmartForms, no BAdI, no RAP behavior definitions.
|
|
- spec.md uses Simplified Technical English and these sections in this order:
|
|
1. Goal, 2. Open questions (write "None."), 3. Context, 4. Contract, 5. Business rules,
|
|
6. Constraints (release target, coding standards, out of scope), 7. Acceptance.
|
|
The Contract fixes every public name the hidden tests use: object names, method signatures,
|
|
function module parameters, report parameters, ALV column names, CDS element names.
|
|
The Business rules are complete and unambiguous. Every rule is checked by at least one hidden test.
|
|
Do not tell the model HOW to implement (no table kinds, no SQL). Craft decisions belong to the model.
|
|
- Seed objects: TABL as DDL source ("define table ..."), FUGR without source, FUNC with "functionGroup"
|
|
and full source including the signature in the FUNCTION statement, classes with full source.
|
|
Seed table data: a seed class that implements IF_OO_ADT_CLASSRUN, with "run": true.
|
|
- Hidden tests: one global class, "FOR TESTING DURATION SHORT RISK LEVEL HARMLESS", 5 to 12 test
|
|
methods. Use only the public contract. Function modules: CALL FUNCTION with EXCEPTIONS.
|
|
Reports: SUBMIT ... AND RETURN with cl_salv_bs_runtime_info=>set( display = abap_false
|
|
metadata = abap_false data = abap_true ) and get_data_ref. CDS: cl_cds_test_environment.
|
|
- Reference solution: correct, Clean ABAP, methods below 40 statements, passes all hidden tests,
|
|
no ATC priority 1 or 2 findings (for example: pass large parameters by reference).
|
|
Include the model's expected own tests: for classes a "testclasses_file" (local test classes);
|
|
for reports local test classes inside the program; for function modules and CDS a global test class.
|
|
- task.json keys: id, category, object_type, difficulty, release_target, expected_outcome ("implement"),
|
|
budget {max_tool_calls, max_activations}, seed[], contract[], out_of_scope[], hidden_tests[],
|
|
reference[], craft_checks[]. Contract entries: CLAS {"implements"} optional; FUNC {"functionGroup",
|
|
"params":[{"name","type"}]}; PROG {"parameters":[...]}; DDLS {"fields":[...]}.
|
|
|
|
Return ONLY one JSON object, no markdown fence:
|
|
{"task": <task.json object>, "files": {"<relative path>": "<file content>", ...}}
|
|
"""
|
|
|
|
|
|
def bundle_of(task_dir):
|
|
files = {}
|
|
for base, _, names in os.walk(task_dir):
|
|
for n in names:
|
|
if n.startswith(".") or n == "generation.json":
|
|
continue
|
|
path = os.path.join(base, n)
|
|
rel = os.path.relpath(path, task_dir)
|
|
if rel != "task.json":
|
|
files[rel] = open(path).read()
|
|
return {"task": json.load(open(os.path.join(task_dir, "task.json"))), "files": files}
|
|
|
|
|
|
def chat(model, messages, base_url):
|
|
# max_tokens: one pilot call ran to 393k output tokens (reasoning) and returned no content
|
|
body = {"model": model, "messages": messages, "temperature": 0.7, "max_tokens": MAX_OUT_TOKENS}
|
|
req = urllib.request.Request(f"{base_url}/chat/completions", json.dumps(body).encode(),
|
|
{"Content-Type": "application/json", "Authorization": "Bearer none"})
|
|
for attempt in range(4):
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=1800) as r:
|
|
data = json.loads(r.read().decode())
|
|
add_usage(model, data.get("usage", {}), kind="generate")
|
|
content = data["choices"][0]["message"].get("content") or ""
|
|
if content.strip():
|
|
return content
|
|
# empty content (output limit reached in reasoning): retry once more with the next attempt
|
|
if attempt == 3:
|
|
return ""
|
|
continue
|
|
except Exception as e: # noqa: BLE001
|
|
if getattr(e, "code", 500) < 500 and getattr(e, "code", 500) != 429:
|
|
raise
|
|
time.sleep(15 * (attempt + 1))
|
|
raise RuntimeError("generation request failed")
|
|
|
|
|
|
def parse_bundle(text):
|
|
text = text.strip()
|
|
text = re.sub(r"^```(json)?\s*|\s*```$", "", text)
|
|
# raw_decode stops after the first complete object; text after it (a note, a second object) is ignored
|
|
return json.JSONDecoder().raw_decode(text[text.find("{"):])[0]
|
|
|
|
|
|
NAME_RE = re.compile(r"^\s*(?:CLASS-)?(?:METHODS|METHOD|DATA|TYPES|CONSTANTS|FORM|CLASS|INTERFACE)\b:?\s+(\w+)",
|
|
re.I | re.M)
|
|
# reserved words seen in activation errors, plus common SQL keywords
|
|
RESERVED = {"HOURS", "MODE", "ORDER", "GROUP", "DATE", "TIME", "VALUE", "USER", "KEY", "COUNT", "SUM", "MIN",
|
|
"MAX", "AVG", "SELECT", "FROM", "WHERE", "TABLE", "VIEW", "UNION", "JOIN", "LEVEL", "SIZE", "TYPE",
|
|
"INDEX", "CASE", "WHEN", "THEN", "ELSE", "END", "AS", "ON", "BY", "DESC", "ASC", "DAYS", "MINUTES",
|
|
"SECONDS", "YEAR", "MONTH", "DAY", "LIMIT", "OFFSET", "PARAMETERS"}
|
|
GLOBAL_TEST_CLASS = re.compile(r"\A\s*(?:\*[^\n]*\n|\"[^\n]*\n|\s)*CLASS\s+\S+\s+DEFINITION[^.]*FOR\s+TESTING",
|
|
re.I)
|
|
DDL_KIND = re.compile(r"^\s*define\s+(?:root\s+)?(table|view|structure)\b", re.I | re.M)
|
|
|
|
|
|
def lint_files(b):
|
|
"""Static checks that need no SAP system: identifier length, seed type against DDL source."""
|
|
errs = []
|
|
for rel, src in b.get("files", {}).items():
|
|
if not rel.endswith(".abap"):
|
|
continue
|
|
for name in set(NAME_RE.findall(src.replace("{{P}}", "Z0000000_").replace("{{p}}", "z0000000_"))):
|
|
if len(name) > 30:
|
|
errs.append(f"{rel}: name {name} has {len(name)} characters (ABAP max 30)")
|
|
t = b.get("task", {})
|
|
oos = {n.upper() for n in t.get("out_of_scope", [])}
|
|
if t.get("category") in ("E", "I"):
|
|
seed_names = {o.get("name", "").upper() for o in t.get("seed", [])}
|
|
if not any(c.get("name", "").upper() in seed_names for c in t.get("contract", [])):
|
|
errs.append(f"category {t['category']}: the contract object must be the seed object that the model "
|
|
"changes in place (same name in seed, contract and reference)")
|
|
test_names = {o.get("name", "").upper() for o in t.get("hidden_tests", [])}
|
|
for o in t.get("reference", []):
|
|
name = o.get("name", "")
|
|
if name.upper() in oos:
|
|
errs.append(f"reference object {name} is also in out_of_scope; the solution must change it")
|
|
if o.get("type") == "CLAS" and GLOBAL_TEST_CLASS.search(b.get("files", {}).get(o.get("file", ""), "")):
|
|
test_names.add(name.upper())
|
|
for c in t.get("contract", []):
|
|
if c.get("name", "").upper() in test_names:
|
|
errs.append(f"contract contains the test class {c.get('name')}; only objects that the hidden "
|
|
"tests call belong in the contract")
|
|
for o in t.get("reference", []):
|
|
if (o.get("type") == "CLAS" and o.get("name", "").upper() not in test_names
|
|
and not o.get("testclasses_file")
|
|
and not re.search(r"INHERITING\s+FROM\s+\S*CX_", b.get("files", {}).get(o.get("file", ""), ""),
|
|
re.I)):
|
|
errs.append(f"reference class {o.get('name')} has no testclasses_file (local unit tests)")
|
|
for rel, src in b.get("files", {}).items():
|
|
# G0121, G0122: SAP says only "Can't save due to errors in source" for these
|
|
if re.search(r"^\s*define\s+table\b", src, re.I | re.M):
|
|
for ann in ("tableCategory", "deliveryClass"):
|
|
if not re.search(rf"@AbapCatalog\.{ann}\s*:", src, re.I):
|
|
errs.append(f"{rel}: a table needs the annotation @AbapCatalog.{ann} "
|
|
"(for example #TRANSPARENT, #A)")
|
|
if DDL_KIND.search(src):
|
|
for f in re.findall(r"^\s*(?:key\s+)?(\w+)\s*:", src, re.I | re.M):
|
|
if f.upper() in RESERVED:
|
|
errs.append(f"{rel}: field name {f} is a reserved word; choose another name")
|
|
for rel, src in b.get("files", {}).items():
|
|
if rel.endswith(".asddls") and re.search(r"define\s+(root\s+)?view\s+entity", src, re.I):
|
|
if re.search(r"\bunion\b", src, re.I) and not re.search(r"@Metadata\.ignorePropagatedAnnotations\s*:\s*true",
|
|
src, re.I):
|
|
errs.append(f"{rel}: a view entity with UNION needs @Metadata.ignorePropagatedAnnotations: true")
|
|
for p in sorted(set(re.findall(r"[=<>(,]\s*:(\w+)|\bbetween\s+:(\w+)|\band\s+:(\w+)", src, re.I))):
|
|
name = next(x for x in p if x)
|
|
errs.append(f"{rel}: parameter :{name}; a view entity needs $parameters.{name}")
|
|
for k in ("seed", "reference"):
|
|
for o in t.get(k, []):
|
|
m = DDL_KIND.search(b.get("files", {}).get(o.get("file", ""), ""))
|
|
if not m:
|
|
continue
|
|
want = {"table": "TABL", "structure": "STRU", "view": "DDLS"}[m.group(1).lower()]
|
|
if o.get("type") != want:
|
|
errs.append(f"{k}: {o.get('name')} has type {o.get('type')}, but its source is "
|
|
f"'define {m.group(1)}'; use type {want}")
|
|
return errs
|
|
|
|
|
|
def abaplint_files(b):
|
|
"""Parser errors of ABAP sources (classes, interfaces, programs) found by local abaplint.
|
|
SAP often reports only "save operation failed" for these; abaplint gives the line.
|
|
Only parser_error: check_syntax gives false errors for standard objects abaplint does not know."""
|
|
t, files = b.get("task", {}), b.get("files", {})
|
|
ext = {"CLAS": "clas", "INTF": "intf", "PROG": "prog"}
|
|
version = t.get("release_target") if t.get("release_target") in ABAPLINT_VERSIONS else "v758"
|
|
d = tempfile.mkdtemp(prefix="genlint_")
|
|
origin = {}
|
|
try:
|
|
os.makedirs(os.path.join(d, "src"))
|
|
for k in ("seed", "reference", "hidden_tests"):
|
|
for o in t.get(k, []):
|
|
if o.get("type") not in ext:
|
|
continue
|
|
name = re.sub(r"\W", "_", o.get("name", "").replace("{{P}}", "Z0000000_")).lower()
|
|
for fk, suffix in (("file", ""), ("testclasses_file", ".testclasses")):
|
|
if o.get(fk) in files:
|
|
fn = f"{name}.{ext[o['type']]}{suffix}.abap"
|
|
origin[fn] = o[fk]
|
|
src = files[o[fk]].replace("{{P}}", "Z0000000_").replace("{{p}}", "z0000000_")
|
|
open(os.path.join(d, "src", fn), "w").write(src)
|
|
if not origin:
|
|
return []
|
|
cfg = {"global": {"files": "/src/**/*.*"}, "dependencies": [],
|
|
"syntax": {"version": version, "errorNamespace": "^(Z|Y)"}, "rules": {"parser_error": True}}
|
|
json.dump(cfg, open(os.path.join(d, "abaplint.json"), "w"))
|
|
p = subprocess.run([ABAPLINT, "abaplint.json", "-f", "json"], cwd=d, capture_output=True, text=True,
|
|
timeout=300)
|
|
issues = json.loads(p.stdout or "[]")
|
|
except Exception: # noqa: BLE001 the pre-check is optional; SAP validation follows
|
|
return []
|
|
finally:
|
|
shutil.rmtree(d, ignore_errors=True)
|
|
out = []
|
|
for i in issues:
|
|
f = i.get("file", "")
|
|
fn = os.path.basename(f.get("filename", "") if isinstance(f, dict) else str(f))
|
|
if i.get("key") == "parser_error" and fn in origin:
|
|
out.append(f"{origin[fn]} line {i.get('start', {}).get('row')}: syntax error for release {version} "
|
|
f"(abaplint): {i.get('description')}")
|
|
return out[:20]
|
|
|
|
|
|
def check_bundle(b):
|
|
errs = []
|
|
t, files = b.get("task", {}), b.get("files", {})
|
|
stop = t.get("expected_outcome") == "stop"
|
|
for k in ("seed", "contract", "hidden_tests", "reference"):
|
|
if k not in t:
|
|
errs.append(f"task.json misses '{k}'")
|
|
if stop:
|
|
gap = t.get("expected_gap") or {}
|
|
if not gap.get("description") or not gap.get("keywords"):
|
|
errs.append("stop task: expected_gap needs description and keywords")
|
|
if "spec.md" not in files:
|
|
errs.append("spec.md missing")
|
|
for rel in files: # G0106: a key "seed/" (a directory) crashed write_bundle
|
|
if not rel or rel.endswith("/") or rel.startswith("/") or ".." in rel.split("/"):
|
|
errs.append(f"files: '{rel}' is not a valid relative file path")
|
|
for k in ("seed", "hidden_tests", "reference"):
|
|
for o in t.get(k, []):
|
|
for fk in ("file", "testclasses_file"):
|
|
if fk in o and o[fk] not in files:
|
|
errs.append(f"{k}: file {o[fk]} missing")
|
|
if not o.get("name", "").startswith("{{P}}"):
|
|
errs.append(f"{k}: name {o.get('name')} does not start with {{{{P}}}}")
|
|
limit = {"TABL": 16, "FUGR": 26, "MSAG": 20}.get(o.get("type"), 30)
|
|
if len(o.get("name", "").replace("{{P}}", "Z0000000_")) > limit:
|
|
errs.append(f"{k}: name {o.get('name')} too long (max {limit})")
|
|
for k in ("seed", "reference"):
|
|
for o in t.get(k, []):
|
|
if o.get("type") == "MSAG":
|
|
msgs = o.get("messages") or []
|
|
if k == "reference" and not msgs:
|
|
errs.append(f"reference message class {o.get('name')} has no 'messages' list")
|
|
for m in msgs:
|
|
if not re.fullmatch(r"\d{1,3}", str(m.get("msgno", ""))) or not m.get("text") or len(m["text"]) > 72:
|
|
errs.append(f"message class {o.get('name')}: message {m.get('msgno')} needs a 1 to 3 digit "
|
|
"msgno and a text of at most 72 characters")
|
|
for c in t.get("contract", []):
|
|
if c.get("type") in ("TABL", "STRU") and not c.get("fields"):
|
|
errs.append(f"contract {c.get('name')}: a {c['type']} entry needs the list 'fields'")
|
|
if c.get("type") == "MSAG" and not c.get("messages"):
|
|
errs.append(f"contract {c.get('name')}: a MSAG entry needs the list 'messages'")
|
|
if not t.get("hidden_tests") and not stop:
|
|
errs.append("no hidden test class")
|
|
return errs
|
|
|
|
|
|
def dependency_order(objs, files):
|
|
"""Stable order in which every object comes after the objects that its sources name
|
|
(for example exception classes before the class that raises them). A cycle keeps the given order."""
|
|
def text(o):
|
|
return " ".join(files.get(o.get(k), "") for k in ("file", "testclasses_file")).upper()
|
|
names = [o.get("name", "").upper() for o in objs]
|
|
needs = []
|
|
for o in objs:
|
|
src, own = text(o), o.get("name", "").upper()
|
|
dep = {n for n in names if n != own and re.search(rf"(?<![\w{{}}]){re.escape(n)}(?!\w)", src)}
|
|
if o.get("functionGroup"):
|
|
dep.add(o["functionGroup"].upper())
|
|
needs.append(dep & set(names))
|
|
out, done = [], set()
|
|
while len(out) < len(objs):
|
|
ready = [i for i, o in enumerate(objs) if i not in done and needs[i] <= {names[j] for j in done}]
|
|
i = ready[0] if ready else min(set(range(len(objs))) - done)
|
|
done.add(i)
|
|
out.append(objs[i])
|
|
return out
|
|
|
|
|
|
def floor_budget(task):
|
|
"""The model needs more calls than the oracle (reads, repairs). G0022: budget 12 activations,
|
|
reference alone 13. Minimum: 2 x oracle activations, 3 x oracle tool calls."""
|
|
ref = task.get("reference", [])
|
|
pushes = sum(bool(o.get("file")) + bool(o.get("testclasses_file")) for o in ref)
|
|
calls = len(ref) + pushes
|
|
bud = task.setdefault("budget", {})
|
|
# absolute floor (design default 60): with 40 calls DeepSeek used the budget to read the seed tables and
|
|
# stopped before its own tests (empirical runs G0017, G0024)
|
|
bud["max_activations"] = max(int(bud.get("max_activations", 0)), 2 * pushes, 15)
|
|
bud["max_tool_calls"] = max(int(bud.get("max_tool_calls", 0)), 3 * calls, 60)
|
|
|
|
|
|
KEEP_FILES = {"generation.json", "review.json"}
|
|
|
|
|
|
def write_bundle(b, task_dir, task_id):
|
|
# remove the files of an earlier attempt (G0003 kept an unused seed/course.ddls.abap)
|
|
if os.path.isdir(task_dir):
|
|
for name in os.listdir(task_dir):
|
|
if name not in KEEP_FILES:
|
|
path = os.path.join(task_dir, name)
|
|
shutil.rmtree(path) if os.path.isdir(path) else os.remove(path)
|
|
os.makedirs(task_dir, exist_ok=True)
|
|
b["task"]["id"] = task_id
|
|
for k in ("seed", "reference"):
|
|
b["task"][k] = dependency_order(b["task"].get(k, []), b["files"])
|
|
floor_budget(b["task"])
|
|
json.dump(b["task"], open(os.path.join(task_dir, "task.json"), "w"), indent=2)
|
|
for rel, content in b["files"].items():
|
|
path = os.path.join(task_dir, rel)
|
|
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
open(path, "w").write(content)
|
|
|
|
|
|
def validate(pool_root, task_id, run_base):
|
|
r = Runner(pool_root, os.path.join(ROOT, "runs", "gen"))
|
|
os.makedirs(os.path.join(ROOT, "runs", "gen"), exist_ok=True)
|
|
rep_o, dir_o = r.run(task_id, OracleAgent(), run_base)
|
|
rep_o["_dir"] = dir_o
|
|
rep_n, _ = r.run(task_id, NullAgent(), run_base + 1)
|
|
return rep_o, rep_n
|
|
|
|
|
|
def write_errors(run_dir):
|
|
"""Failed writes and activation messages of the oracle run."""
|
|
out = []
|
|
path = os.path.join(run_dir or "", "trajectory.jsonl")
|
|
if not os.path.exists(path):
|
|
return out
|
|
for line in open(path):
|
|
e = json.loads(line)
|
|
if e.get("tool") in ("sap_create_object", "sap_push_source") and (
|
|
e["is_error"] or '"success":false' in e["result"].replace(" ", "")):
|
|
out.append({"tool": e["tool"], "object": e["args"].get("objectName"),
|
|
"include": e["args"].get("includeType", "main"), "result": e["result"][:1500]})
|
|
return out
|
|
|
|
|
|
def failure_summary(rep):
|
|
out = {"reference_write_errors": write_errors(rep.get("_dir")), "gates": rep.get("gates"),
|
|
"contract_check": rep.get("contract_check"), "score": rep.get("score"), "setup": rep.get("setup"),
|
|
"atc": rep.get("atc"), "abaplint": rep.get("abaplint"), "own_tests": rep.get("own_tests"),
|
|
"hidden_failed": [d for d in rep.get("hidden_tests", {}).get("detail", []) if not d["ok"]],
|
|
"hidden_install": rep.get("hidden_tests", {}).get("install")}
|
|
return json.dumps(out)[:6000]
|
|
|
|
|
|
def generate(task_id, pool, object_type, category, difficulty, model, base_url, run_base, topic=None,
|
|
max_repairs=3, extra_check=None):
|
|
"""extra_check(bundle) -> list of problems; it runs with the static checks (training mode: overlap)."""
|
|
check_budget()
|
|
pool_root = os.path.join(ROOT, "tasks_gen", pool)
|
|
task_dir = os.path.join(pool_root, task_id)
|
|
example = bundle_of(os.path.join(ROOT, "tasks", EXAMPLE_FOR[object_type]))
|
|
note_key = "EXC" if (object_type == "CLAS" and topic and "exception class" in topic.lower()) else object_type
|
|
ask = (f"Write one new task.\nObject type of the main contract object: {object_type}.\n"
|
|
f"Skill category {category}: {CATEGORIES[category]}.\nDifficulty {difficulty} of 3.\n"
|
|
+ (f"Topic idea: {topic}\n" if topic else "Choose a new, realistic business topic.\n")
|
|
+ (("Format notes for this object type:\n" + TYPE_NOTES[note_key] + "\n") if note_key in TYPE_NOTES else "")
|
|
+ "Here is an example bundle of a different task (same format):\n" + json.dumps(example))
|
|
messages = [{"role": "system", "content": SYSTEM}, {"role": "user", "content": ask}]
|
|
log = {"id": task_id, "pool": pool, "object_type": object_type, "category": category, "attempts": []}
|
|
for attempt in range(max_repairs + 1):
|
|
text = chat(model, messages, base_url)
|
|
messages.append({"role": "assistant", "content": text})
|
|
try:
|
|
b = parse_bundle(text)
|
|
b.get("task", {}).setdefault("category", category)
|
|
found = {"structure": check_bundle(b), "static": lint_files(b), "abaplint": abaplint_files(b)}
|
|
if extra_check and not found["structure"]:
|
|
found["extra"] = extra_check(b)
|
|
except Exception as e: # noqa: BLE001
|
|
b, found = None, {"json": [f"invalid JSON: {e}"]}
|
|
errs = [e for v in found.values() for e in v]
|
|
if errs:
|
|
# one entry per source, so a batch shows what the checks found before SAP
|
|
log["attempts"].append({"stage": "bundle", "errors": errs,
|
|
"by_check": {k: len(v) for k, v in found.items() if v}})
|
|
messages.append({"role": "user", "content": "Fix these problems and return the full bundle again:\n"
|
|
+ "\n".join(errs)})
|
|
continue
|
|
b["task"].setdefault("object_type", object_type)
|
|
b["task"].setdefault("category", category)
|
|
write_bundle(b, task_dir, task_id)
|
|
# run numbers per attempt: +0 oracle, +1 null, +2..+6 mutants (RUNS_PER_ATTEMPT)
|
|
base = run_base + RUNS_PER_ATTEMPT * attempt
|
|
rep_o, rep_n = validate(pool_root, task_id, base)
|
|
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
|
|
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
|
|
if category == "H": # oracle stops with the gap: 100; null stops without a report: 30
|
|
if so == 100 and sn == 30:
|
|
log["accepted"] = True
|
|
break
|
|
messages.append({"role": "user", "content":
|
|
f"Stop task check: required oracle 100 and null 30. Result: oracle {so}, null {sn}. "
|
|
f"Oracle report: {failure_summary(rep_o)}\nFix expected_gap (description, keywords) "
|
|
"and return the full bundle again."})
|
|
continue
|
|
if so == 100 and sn == 0:
|
|
mut = check_task(pool_root, task_id, base + 2, keep=True)
|
|
log["attempts"][-1]["mutation"] = {k: mut[k] for k in ("valid", "killed", "ok")}
|
|
if mut["ok"]:
|
|
log["accepted"] = True
|
|
break
|
|
survived = [f"{m['object']} {m['mutant']}" for m in mut["mutants"] if m["status"] == "survived"]
|
|
shutil.rmtree(os.path.join(task_dir, "faulty"), ignore_errors=True)
|
|
messages.append({"role": "user", "content":
|
|
"Oracle 100 and null 0: good. But the hidden tests are too weak. The harness changed "
|
|
"the reference (mutation check) and all hidden tests still passed for these changes:\n"
|
|
+ "\n".join(survived or ["(fewer than 2 changes possible: add more business logic "
|
|
"checks to the hidden tests)"])
|
|
+ "\nAdd or improve hidden tests so that each of these changes makes a test fail. "
|
|
"If a change does not change the behavior, ignore it. Do not change the behavior "
|
|
"of the reference. Return the full bundle again."})
|
|
continue
|
|
messages.append({"role": "user", "content":
|
|
"The harness ran your reference solution (oracle) and an empty solution (null). "
|
|
f"Required: oracle 100, null 0. Result: oracle {so}, null {sn}.\n"
|
|
f"Oracle report: {failure_summary(rep_o)}\n"
|
|
"Fix the bundle (reference, hidden tests, seed, or contract) and return the full "
|
|
"bundle again."})
|
|
else:
|
|
log["accepted"] = False
|
|
json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"),
|
|
indent=1)
|
|
return log
|
|
|
|
|
|
K_STYLES = {
|
|
"free_text": "Rewrite the spec as a short free-text request from a functional consultant: no section "
|
|
"headings, no numbered rules, plain sentences, as in an e-mail. Keep every business rule.",
|
|
"incomplete": "Rewrite the spec as a short free-text request. Leave out the craft hints and the context "
|
|
"that a good ABAP developer can find in the system (for example how the seed table looks). "
|
|
"Keep every business rule; the task must stay solvable without questions.",
|
|
}
|
|
|
|
|
|
def contract_terms(task):
|
|
"""Names that a K spec must keep: object names, FM parameters, report parameters, CDS elements."""
|
|
out = []
|
|
for c in task.get("contract", []):
|
|
out.append(c["name"])
|
|
out += [p["name"] if isinstance(p, dict) else p for p in c.get("params", []) + c.get("parameters", [])]
|
|
out += list(c.get("fields", []))
|
|
return out
|
|
|
|
|
|
def make_k_variant(src_id, new_id, style, model, base_url, run_base, tool_schema="generic_v0", max_repairs=2,
|
|
pool="eval"):
|
|
"""Category K: free-text or incomplete input + other tool schema. Same reference and hidden tests
|
|
as the source task; only spec.md, category and tool_schema change."""
|
|
check_budget()
|
|
pool_root = os.path.join(ROOT, "tasks_gen", pool)
|
|
src_dir, task_dir = os.path.join(pool_root, src_id), os.path.join(pool_root, new_id)
|
|
b = bundle_of(src_dir)
|
|
b["files"] = {k: v for k, v in b["files"].items()
|
|
if not (k.startswith("faulty/") or k in ("mutation.json", "review.json", "empirical.json"))}
|
|
spec = b["files"]["spec.md"]
|
|
terms = contract_terms(b["task"])
|
|
messages = [{"role": "user", "content":
|
|
f"{K_STYLES[style]}\nKeep every contract item: the output form (for example an ALV list "
|
|
"with CL_SALV_TABLE), the selection screen, the signature, and the column names (G0181 lost the "
|
|
"ALV). Keep these names exactly as written (the tests use them): "
|
|
f"{', '.join(terms)}. Keep the placeholder {{{{P}}}} in names. Use Simplified Technical English. "
|
|
f"Return only the new spec text.\n\nSpec:\n{spec}"}]
|
|
log = {"id": new_id, "pool": pool, "category": "K", "base_task": src_id, "style": style,
|
|
"tool_schema": tool_schema, "attempts": []}
|
|
for attempt in range(max_repairs + 1):
|
|
text = chat(model, messages, base_url).strip()
|
|
text = re.sub(r"^```\w*\s*|\s*```$", "", text)
|
|
messages.append({"role": "assistant", "content": text})
|
|
missing = [t for t in terms if t.upper() not in text.upper()]
|
|
if missing:
|
|
log["attempts"].append({"stage": "spec", "missing_names": missing})
|
|
messages.append({"role": "user", "content": "These names are missing: " + ", ".join(missing)
|
|
+ ". Return the full spec again with all names."})
|
|
continue
|
|
b["files"]["spec.md"] = text + "\n"
|
|
b["task"].update(category="K", base_task=src_id, input_style=style)
|
|
if tool_schema: # None: EPOD tool names (training: free-text input only)
|
|
b["task"]["tool_schema"] = tool_schema
|
|
write_bundle(b, task_dir, new_id)
|
|
rep_o, rep_n = validate(pool_root, new_id, run_base + RUNS_PER_ATTEMPT * attempt)
|
|
so, sn = (rep_o.get("score") or {}).get("total"), (rep_n.get("score") or {}).get("total")
|
|
log["attempts"].append({"stage": "validate", "oracle": so, "null": sn})
|
|
log["accepted"] = so == 100 and sn == 0
|
|
break
|
|
else:
|
|
log["accepted"] = False
|
|
json.dump(log, open(os.path.join(task_dir if os.path.isdir(task_dir) else pool_root, "generation.json"), "w"),
|
|
indent=1)
|
|
return log
|
|
|
|
|
|
def main():
|
|
load_env(os.path.join(ROOT, ".env"))
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--id", required=True)
|
|
ap.add_argument("--pool", default="eval", choices=["eval", "train"])
|
|
ap.add_argument("--object-type", choices=sorted(EXAMPLE_FOR))
|
|
ap.add_argument("--category", choices=sorted(CATEGORIES))
|
|
ap.add_argument("--difficulty", type=int, default=2)
|
|
ap.add_argument("--topic")
|
|
ap.add_argument("--model", default="deepseek-v4.1-flash:cloud")
|
|
ap.add_argument("--base-url", default=os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1"))
|
|
ap.add_argument("--run-base", type=int, required=True)
|
|
ap.add_argument("--k-from", help="category K: make a variant of this accepted eval task")
|
|
ap.add_argument("--k-style", default="free_text", choices=sorted(K_STYLES))
|
|
a = ap.parse_args()
|
|
if a.k_from:
|
|
print(json.dumps(make_k_variant(a.k_from, a.id, a.k_style, a.model, a.base_url, a.run_base)))
|
|
return
|
|
os.makedirs(os.path.join(ROOT, "tasks_gen", a.pool), exist_ok=True)
|
|
log = generate(a.id, a.pool, a.object_type, a.category, a.difficulty, a.model, a.base_url,
|
|
a.run_base, a.topic)
|
|
print(json.dumps(log))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|