Object type mix: INTF, TABL, STRU, MSAG, exception tasks (harness G2, mutants, generator notes), balanced generator, kind-deficit job order, dashboard mix card

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-05 18:52:29 +02:00
parent 454db6626c
commit b119f1afac
39 changed files with 2329 additions and 24 deletions

View File

@@ -53,6 +53,8 @@ class OracleAgent:
ident["functionGroup"] = o["functionGroup"]
proxy.call("sap_create_object", dict(ident, packageName="$TMP",
description=o.get("description", o["name"])[:60]))
if o.get("messages"): # message class: messages are written with sap_push_message
proxy.call("sap_push_message", {"objectName": o["name"], "messages": o["messages"]})
if o.get("source"):
proxy.call("sap_push_source", dict(ident, source=o["source"]))
if o.get("testclasses_source"):

View File

@@ -17,6 +17,7 @@ import time
import urllib.request
from .adt_client import load_env
from . import mix
from .ledger import _env_budget, spent
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
@@ -242,6 +243,29 @@ def agg_table(title, evs, rows, key, note=""):
% (E(title), E(key.replace("_", " ")), trs, "<div class=note>%s</div>" % E(note) if note else ""))
def mix_table(rows, evs):
kt = mix.accepted_task_counts()
kr = {}
for r, e in zip(rows, evs):
k = mix.kind_of_task_dir(r["task"])
a = kr.setdefault(k, [0, 0])
a[0] += 1
a[1] += e["accepted"]
tt, tr = max(sum(kt.values()), 1), max(sum(v[1] for v in kr.values()), 1)
ss = sum(mix.TYPE_SHARE.values())
trs = ""
for k, v in mix.TYPE_SHARE.items():
tgt = 100.0 * v / ss
pt, pr = 100.0 * kt.get(k, 0) / tt, 100.0 * kr.get(k, [0, 0])[1] / tr
cls = "g" if abs(pt - tgt) < 4 else ("w" if abs(pt - tgt) < 10 else "r")
trs += ("<tr><td>%s</td><td class=n>%.0f%%</td><td class=n>%d (%.0f%%)</td><td style='width:25%%'>%s</td>"
"<td class=n>%d/%d (%.0f%%)</td></tr>" % (k, tgt, kt.get(k, 0), pt, bar(pt, max(tgt * 2, 1), cls),
kr.get(k, [0, 0])[1], kr.get(k, [0, 0])[0], pr))
return ("<div class=card><h2>Object type mix vs target</h2><div class=b><table><tr><th>kind</th><th class=n>target</th>"
"<th class=n>tasks</th><th>vs target</th><th class=n>trajectories acc/runs</th></tr>%s</table></div>"
"<div class=note>CLAS/INTF 35, CDS 25, FUNC 15, PROG 10, DDIC 10, MSAG + exception 5 (percent of accepted tasks).</div></div>" % trs)
def gen_table(logs, key, title):
agg = {}
for l in logs:
@@ -351,7 +375,7 @@ def render(d):
E("\n".join(d["procs"]))))
logc = "<div class=card><h2>Pipeline log</h2><div class=b><div class='log mono'>%s</div></div></div>" % E("\n".join(d["log_tail"]))
body = (banner + "<div class=tiles>" + tiles + "</div><div class=grid>" + budget + prog + stops + sysc
+ agg_table("Trajectories by category", evs, rows, "category") + agg_table("Trajectories by object type", evs, rows, "object_type")
+ mix_table(rows, evs) + agg_table("Trajectories by category", evs, rows, "category") + agg_table("Trajectories by object type", evs, rows, "object_type")
+ gen_table(logs, "category", "Task generation by category") + gen_table(logs, "object_type", "Task generation by object type")
+ tok + recent + logc + "</div>")
stamp = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(d["now"]))

View File

@@ -24,7 +24,8 @@ ABAPLINT = os.path.join(ROOT, "node_modules", ".bin", "abaplint")
MAX_OUT_TOKENS = 80000
RUNS_PER_ATTEMPT = 10
ABAPLINT_VERSIONS = {"v702", "v740sp05", "v750", "v758"}
EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15"}
EXAMPLE_FOR = {"CLAS": "T01", "INTF": "T01", "FUNC": "T13", "PROG": "T14", "DDLS": "T15", "TABL": "T15",
"STRU": "T15", "MSAG": "T13"}
CATEGORIES = {
"A": "pure logic in a new class (language, OO design, Clean ABAP)",
"B": "database access (ABAP SQL, CDS) with test doubles",
@@ -50,6 +51,57 @@ CATEGORIES = {
"hidden_tests is []. Do not write the gap in Open questions (write \"None.\")",
}
# Extra format notes for object types that the main prompt does not describe (training mode, 2026-10-05).
# The example bundle is of another type; these notes say how this type differs.
TYPE_NOTES = {
"INTF": """The model must create an INTERFACE (type INTF) as the main contract object: the methods, types and constants
named in the spec. Contract entry: {"type": "INTF", "name": "{{P}}IF_X"}. Reference: the interface source
("reference/x.intf.abap") and ONE global own-test class (type CLAS, source with a global class DEFINITION FOR TESTING;
an interface has no test include). Hidden tests: one global test class with a local class that IMPLEMENTS the
interface (it only compiles when every signature matches), calls the methods through the interface, reads the
constants (values), and checks type lengths with RTTI (cl_abap_typedescr=>describe_by_name). The business rules give the
exact constant values, parameter names and types. No other object is needed.""",
"TABL": """The model must create a transparent DATABASE TABLE (type TABL) as the main contract object. The table name has at
most 16 characters including the prefix: use a short name, for example {{P}}ORD (7 characters after the prefix at most).
Source is DDL: annotations @EndUserText.label, @AbapCatalog.enhancement.category : #NOT_EXTENSIBLE,
@AbapCatalog.tableCategory : #TRANSPARENT, @AbapCatalog.deliveryClass : #A, @AbapCatalog.dataMaintenance : #RESTRICTED,
then "define table {{p}}ord { key client : abap.clnt not null; key item_id : abap.char(10) not null; ... }".
Contract entry: {"type": "TABL", "name": "{{P}}ORD", "fields": ["client", "item_id", ...]} (all fields, lower case).
The spec (Business rules) lists every field with its data type and length, the key fields, and not-null fields. Use
only built-in types (abap.char, abap.numc, abap.dec, abap.int4, abap.dats). Avoid abap.curr and abap.quan; if the spec
needs an amount or a quantity, the reference field needs a reference annotation in the form 'tablename.fieldname'
(for example @Semantics.amount.currencyCode : '{{p}}ord.currency_code' on the amount and a field currency_code :
abap.cuky in the same table; @Semantics.quantity.unitOfMeasure : '{{p}}ord.unit' with unit : abap.unit(3)); a
reference without the table name fails with "Annotation with reference to unit code ... is uncomplete". Reference: the DDL file and ONE global own-test class
(type CLAS). Hidden tests (one global class): RTTI on the table: cast cl_abap_typedescr=>describe_by_name( '{{P}}ORD' ) to
cl_abap_structdescr and check get_ddic_field_list( ) (field names, key flags, lengths, decimals, types), and a
cl_osql_test_environment test (create( i_dependency_list = VALUE #( ( '{{P}}ORD' ) ) ), insert, select back; a second
INSERT with the same key gives sy-subrc = 4). No seed is needed.""",
"STRU": """The model must create a DDIC STRUCTURE (type STRU) as the main contract object. DDL source: annotations
@EndUserText.label and @AbapCatalog.enhancement.category : #NOT_EXTENSIBLE, then
"define structure {{p}}name { code : abap.char(4); amount : abap.dec(9,2); ... }" (include another structure with
"include {{p}}other;" when the spec asks). Contract entry: {"type": "STRU", "name": "{{P}}NAME", "fields": [...]}
(the field names in lower case). The Business rules list every field with data type and length. Reference: the DDL file
and ONE global own-test class (type CLAS). Hidden tests (one global class): RTTI: cl_abap_typedescr=>describe_by_name(
'{{P}}NAME' ) cast to cl_abap_structdescr; check components (names, length, decimals, type kind) and the order.""",
"MSAG": """The model must create a MESSAGE CLASS (type MSAG) as the main contract object. The name has at most 20
characters including the prefix. A message class has NO source file. In task.json the reference entry has no "file":
{"type": "MSAG", "name": "{{P}}MSG", "description": "...", "messages": [{"msgno": "001", "text": "Order &1 is blocked"}, ...]}
and the contract entry is {"type": "MSAG", "name": "{{P}}MSG", "messages": [{"msgno": "001"}, ...]} (the numbers that
must exist). The Business rules give for each message the number, the exact text with the placeholders &1 to &4 (at
most 72 characters), and when it is used (error, warning, information). Reference: the message class entry and ONE global
own-test class (type CLAS) that reads the messages. Hidden tests (one global class): for each message use
MESSAGE ID '{{P}}MSG' TYPE 'E' NUMBER '001' WITH 'A' 'B' INTO DATA(lv_text) and assert the final text; also check one
placeholder substitution. The model may also be asked for an exception class that uses the message class (then both
are contract entries and the exception class is a reference file).""",
"EXC": """The main contract object is a class-based EXCEPTION class (CLAS, name {{P}}CX_...) that inherits from
CX_STATIC_CHECK, CX_DYNAMIC_CHECK or CX_NO_CHECK. The spec asks for constants for the error cases, attributes with
context data, a constructor with these parameters, and a message text (through IF_T100_DYN_MSG / IF_T100_MESSAGE with
a message class that is a seed object, or through a redefined get_text). Hidden tests raise the exception from a small
seed class, catch it and check the attributes and the text (get_text( )). Reference: the exception class with its
testclasses_file or ONE global own-test class (an exception class has few methods).""",
}
SYSTEM = """You write evaluation tasks for an ABAP developer model. Each task is a bundle of files.
The model gets only spec.md and works on an SAP ABAP Platform 2025 system (SAP_BASIS 816, client 001)
through ADT tools. The harness installs the seed objects, runs the model, then checks the result with
@@ -220,7 +272,7 @@ def lint_files(b):
m = DDL_KIND.search(b.get("files", {}).get(o.get("file", ""), ""))
if not m:
continue
want = {"table": "TABL", "structure": "TABL", "view": "DDLS"}[m.group(1).lower()]
want = {"table": "TABL", "structure": "STRU", "view": "DDLS"}[m.group(1).lower()]
if o.get("type") != want:
errs.append(f"{k}: {o.get('name')} has type {o.get('type')}, but its source is "
f"'define {m.group(1)}'; use type {want}")
@@ -294,9 +346,24 @@ def check_bundle(b):
errs.append(f"{k}: file {o[fk]} missing")
if not o.get("name", "").startswith("{{P}}"):
errs.append(f"{k}: name {o.get('name')} does not start with {{{{P}}}}")
limit = 16 if o.get("type") == "TABL" else 26 if o.get("type") == "FUGR" else 30
limit = {"TABL": 16, "FUGR": 26, "MSAG": 20}.get(o.get("type"), 30)
if len(o.get("name", "").replace("{{P}}", "Z0000000_")) > limit:
errs.append(f"{k}: name {o.get('name')} too long (max {limit})")
for k in ("seed", "reference"):
for o in t.get(k, []):
if o.get("type") == "MSAG":
msgs = o.get("messages") or []
if k == "reference" and not msgs:
errs.append(f"reference message class {o.get('name')} has no 'messages' list")
for m in msgs:
if not re.fullmatch(r"\d{1,3}", str(m.get("msgno", ""))) or not m.get("text") or len(m["text"]) > 72:
errs.append(f"message class {o.get('name')}: message {m.get('msgno')} needs a 1 to 3 digit "
"msgno and a text of at most 72 characters")
for c in t.get("contract", []):
if c.get("type") in ("TABL", "STRU") and not c.get("fields"):
errs.append(f"contract {c.get('name')}: a {c['type']} entry needs the list 'fields'")
if c.get("type") == "MSAG" and not c.get("messages"):
errs.append(f"contract {c.get('name')}: a MSAG entry needs the list 'messages'")
if not t.get("hidden_tests") and not stop:
errs.append("no hidden test class")
return errs
@@ -399,9 +466,11 @@ def generate(task_id, pool, object_type, category, difficulty, model, base_url,
pool_root = os.path.join(ROOT, "tasks_gen", pool)
task_dir = os.path.join(pool_root, task_id)
example = bundle_of(os.path.join(ROOT, "tasks", EXAMPLE_FOR[object_type]))
note_key = "EXC" if (object_type == "CLAS" and topic and "exception class" in topic.lower()) else object_type
ask = (f"Write one new task.\nObject type of the main contract object: {object_type}.\n"
f"Skill category {category}: {CATEGORIES[category]}.\nDifficulty {difficulty} of 3.\n"
+ (f"Topic idea: {topic}\n" if topic else "Choose a new, realistic business topic.\n")
+ (("Format notes for this object type:\n" + TYPE_NOTES[note_key] + "\n") if note_key in TYPE_NOTES else "")
+ "Here is an example bundle of a different task (same format):\n" + json.dumps(example))
messages = [{"role": "system", "content": SYSTEM}, {"role": "user", "content": ask}]
log = {"id": task_id, "pool": pool, "object_type": object_type, "category": category, "attempts": []}

79
harness/mix.py Normal file
View File

@@ -0,0 +1,79 @@
"""Object type mix of the training data (CLAUDE.md section 1, Opus review 2026-10-05).
Target share of the accepted tasks (and of the accepted trajectories) per "kind":
CLAS/INTF 35 % (CLAS 28, INTF 7), CDS (DDLS) 25 %, FUNC 15 %, PROG 10 %, DDIC 10 % (TABL 8, STRU 2),
MSAG + exception 5 % (MSAG 2.5, EXC 2.5).
A task has the kind of its main contract object; a CLAS whose reference inherits from CX_ is EXC.
DTEL / DOMA tasks are not made yet (the DDIC share is covered by TABL and STRU).
"""
import json
import os
import re
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
POOL = os.path.join(ROOT, "tasks_gen", "train")
TYPE_SHARE = {"CLAS": 28.0, "INTF": 7.0, "DDLS": 25.0, "FUNC": 15.0, "PROG": 10.0, "TABL": 8.0, "STRU": 2.0,
"MSAG": 2.5, "EXC": 2.5}
# categories a kind supports (docs/faz1-tasarim.md 8): the generator picks the one with the biggest deficit
KIND_CATEGORIES = {"CLAS": "ACDEFGHI", "FUNC": "ACDEFGHI", "PROG": "BCEGH", "DDLS": "BEFHI", "INTF": "A",
"TABL": "B", "STRU": "B", "MSAG": "D", "EXC": "D"}
CATEGORY_SHARE = {"A": 10, "B": 15, "C": 15, "D": 10, "E": 15, "F": 10, "G": 10, "I": 5, "H": 10}
_CACHE = {}
def kind_of_task_dir(task_id):
"""Kind of an accepted training task (cached)."""
if task_id in _CACHE:
return _CACHE[task_id]
d = os.path.join(POOL, task_id)
try:
t = json.load(open(os.path.join(d, "task.json")))
except (OSError, ValueError):
return None
otype = t.get("object_type") or "CLAS"
for c in t.get("contract", []): # K variants and old tasks: the contract object is the truth
otype = c.get("type", otype)
break
kind = otype
if otype == "CLAS":
for o in t.get("reference", []):
if o.get("file") and o.get("name") == (t.get("contract") or [{}])[0].get("name"):
try:
src = open(os.path.join(d, o["file"])).read()
except OSError:
src = ""
if re.search(r"INHERITING\s+FROM\s+\S*CX_", src, re.I):
kind = "EXC"
_CACHE[task_id] = kind
return kind
def deficit_pick(counts, share=TYPE_SHARE, allowed=None):
"""Kind with the largest (target - actual) for the next item. counts: {kind: n}."""
total = sum(counts.values()) + 1
best, best_def = None, None
tot_share = sum(share.values())
for k, s in share.items():
if allowed and k not in allowed:
continue
d = total * s / tot_share - counts.get(k, 0)
if best_def is None or d > best_def:
best, best_def = k, d
return best
def accepted_task_counts(logs_dir=None):
"""{kind: accepted tasks} from the generation logs of the training pool (K variants count by their kind)."""
import glob
logs_dir = logs_dir or os.path.join(POOL, "_logs")
out = {}
for f in glob.glob(os.path.join(logs_dir, "G*.json")):
try:
l = json.load(open(f))
except (OSError, ValueError):
continue
if l.get("accepted"):
k = kind_of_task_dir(l["id"])
if k:
out[k] = out.get(k, 0) + 1
return out

View File

@@ -148,6 +148,72 @@ def mutants(src, otype, seed, n=MAX_MUTANTS):
return out
def decl_mutants(src, otype, seed, n=MAX_MUTANTS):
"""Mutants of declarations that carry the behavior of DDIC objects and interfaces (no executable code):
TABL / STRU: field length, decimals, data type, key flag. INTF: constant values, type lengths."""
sites = [] # (kind, start, end, new, line, old)
def add(kind, m, new):
sites.append((kind, m.start(), m.end(), new, src.count("\n", 0, m.start()) + 1, m.group(0)))
if otype in ("TABL", "STRU"):
for m in re.finditer(r"abap\.(char|numc|dec|curr|quan|lang|cuky|unit)\((\d+)(?:,\s*(\d+))?\)", src, re.I):
kind_, ln, dec = m.group(1).lower(), int(m.group(2)), m.group(3)
if kind_ in ("char", "numc") and ln > 1:
add("length", m, f"abap.{kind_}({ln - 1})")
if dec is not None and int(dec) < ln - 1:
add("decimals", m, f"abap.{kind_}({ln},{int(dec) + 1})")
if kind_ == "char" and ln > 1:
add("type", m, f"abap.numc({ln})")
for m in re.finditer(r"abap\.(int4|int8|timestamp|dats|tims)\b", src, re.I):
add("type", m, {"int4": "abap.int8", "int8": "abap.int4", "timestamp": "abap.dats",
"dats": "abap.tims", "tims": "abap.dats"}[m.group(1).lower()])
for m in re.finditer(r"^(\s*)key(\s+)(?!client\b)(\w+\s*:)", src, re.I | re.M):
add("key", m, f"{m.group(1)}{m.group(3)}")
if otype == "INTF":
for m in re.finditer(r"(\bVALUE\s+)(\d+)(?=\s*\.)", src, re.I):
add("const", m, f"{m.group(1)}{int(m.group(2)) + 1}")
for m in re.finditer(r"(\bVALUE\s+)'([^'\n]{1,20})'", src, re.I):
v = m.group(2)
add("lit", m, f"{m.group(1)}'" + ("Z" if v[0] != "Z" else "Y") + v[1:] + "'")
for m in re.finditer(r"(\bVALUE\s+)(abap_true|abap_false)\b", src, re.I):
add("bool", m, m.group(1) + ("abap_false" if m.group(2).lower() == "abap_true" else "abap_true"))
for m in re.finditer(r"\bLENGTH\s+(\d+)", src, re.I):
if int(m.group(1)) > 1:
add("length", m, f"LENGTH {int(m.group(1)) - 1}")
rnd = random.Random(seed)
rnd.shuffle(sites)
chosen, kinds = [], set()
for prefer_new in (True, False):
for st in sites:
if len(chosen) >= n:
break
if st in chosen or (prefer_new and st[0] in kinds):
continue
chosen.append(st)
kinds.add(st[0])
return [(f"line {line}: {old.strip()} -> {new.strip()} ({kind})", src[:a] + new + src[b:])
for kind, a, b, new, line, old in sorted(chosen, key=lambda x: x[1])]
def msag_mutants(messages, seed, n=MAX_MUTANTS):
"""Mutants of a message class: changed text, changed placeholder, a message moved to another number."""
out = []
for i, m in enumerate(messages):
t = m.get("text", "")
mut = [dict(x) for x in messages]
mut[i]["text"] = t + " x" if len(t) < 70 else t[:-1]
out.append((f"message {m['msgno']}: text + ' x' (text)", mut))
if "&1" in t:
mut = [dict(x) for x in messages]
mut[i]["text"] = t.replace("&1", "&2", 1)
out.append((f"message {m['msgno']}: &1 -> &2 (placeholder)", mut))
if len(messages) > 1:
mut = [dict(x) for x in messages]
mut[i]["msgno"] = str(int(m["msgno"]) + 50).zfill(3)
out.append((f"message {m['msgno']}: number + 50 (number)", mut))
random.Random(seed).shuffle(out)
return out[:n]
def check_task(pool_root, task_id, run_base, n=MAX_MUTANTS, keep=False):
"""Run the mutants of one task. Returns a summary dict; writes it to <task>/mutation.json."""
task_dir = os.path.join(pool_root, task_id)
@@ -156,14 +222,22 @@ def check_task(pool_root, task_id, run_base, n=MAX_MUTANTS, keep=False):
def is_test(o):
return o["type"] == "CLAS" and re.search(r"^\s*CLASS\s+\S+\s+DEFINITION[^.]*FOR\s+TESTING",
open(os.path.join(task_dir, o["file"])).read(), re.I | re.M)
targets = [o for o in meta["reference"] if o.get("file") and o["type"] in ("CLAS", "FUNC", "PROG", "DDLS")
targets = [o for o in meta["reference"] if o.get("file") and o["type"] in ("CLAS", "FUNC", "PROG", "DDLS", "INTF",
"TABL", "STRU")
and not is_test(o)]
targets += [o for o in meta["reference"] if o["type"] == "MSAG" and o.get("messages")]
work = os.path.join(ROOT, "runs", "gen", "mut")
os.makedirs(work, exist_ok=True)
runner = Runner(os.path.join(work, "pool"), os.path.join(work, "runs"))
# candidates per object, then round robin: objects without mutation sites (exception classes) give their share
cand = [[(o, d, m) for d, m in mutants(open(os.path.join(task_dir, o["file"])).read(), o["type"],
f"{task_id}:{o['name']}", n)] for o in targets]
def _muts(o):
if o["type"] == "MSAG":
return msag_mutants(o["messages"], f"{task_id}:{o['name']}", n)
src = open(os.path.join(task_dir, o["file"])).read()
if o["type"] in ("TABL", "STRU", "INTF"):
return decl_mutants(src, o["type"], f"{task_id}:{o['name']}", n)
return mutants(src, o["type"], f"{task_id}:{o['name']}", n)
cand = [[(o, d, m) for d, m in _muts(o)] for o in targets]
plan = []
while len(plan) < n and any(cand):
for c in cand:
@@ -174,7 +248,14 @@ def check_task(pool_root, task_id, run_base, n=MAX_MUTANTS, keep=False):
mdir = os.path.join(work, "pool", task_id)
shutil.rmtree(mdir, ignore_errors=True)
shutil.copytree(task_dir, mdir)
open(os.path.join(mdir, o["file"]), "w").write(msrc)
if o["type"] == "MSAG": # the mutant changes the messages of the reference in task.json
tj = json.load(open(os.path.join(mdir, "task.json")))
for r in tj["reference"]:
if r["name"] == o["name"]:
r["messages"] = msrc
json.dump(tj, open(os.path.join(mdir, "task.json"), "w"), indent=2)
else:
open(os.path.join(mdir, o["file"]), "w").write(msrc)
rep, _ = runner.run(task_id, OracleAgent(), run_base + k)
h = rep.get("hidden_tests") or {}
g = rep.get("gates") or {}
@@ -187,7 +268,7 @@ def check_task(pool_root, task_id, run_base, n=MAX_MUTANTS, keep=False):
results.append({"object": o["name"], "mutant": desc, "status": status,
"hidden": f"{h.get('passed')}/{h.get('total')}",
"failed_tests": [d["method"] for d in h.get("detail", []) if not d["ok"]]})
if keep and status == "killed": # candidate faulty reference for own-test scoring
if keep and status == "killed" and o["type"] != "MSAG": # candidate faulty reference for own-test scoring
fdir = os.path.join(task_dir, "faulty")
os.makedirs(fdir, exist_ok=True)
open(os.path.join(fdir, f"m{k}_{os.path.basename(o['file'])}"), "w").write(msrc)

View File

@@ -25,7 +25,7 @@ import threading
import time
from .adt_client import load_env
from . import trainset, trajectories
from . import mix, trainset, trajectories
from .ledger import spent
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
@@ -129,10 +129,21 @@ class Pipeline:
done = {(r["task"], r["attempt"]) for r in rows}
ev0 = {r["task"]: self.evaluate(r) for r in rows if r["attempt"] == 0}
tasks = trajectories.accepted_tasks()
for t in tasks: # first attempt for every task first
if (t, 0) not in done and (t, 0) not in self.inflight:
self.inflight.add((t, 0))
return t, 0
cands = [t for t in tasks if (t, 0) not in done and (t, 0) not in self.inflight]
if cands: # first attempts: the kind with the biggest deficit against the target mix goes first
counts = {}
for r in rows:
if self.evaluate(r)["accepted"]:
k = mix.kind_of_task_dir(r["task"])
counts[k] = counts.get(k, 0) + 1
for (t, _a) in self.inflight:
k = mix.kind_of_task_dir(t)
counts[k] = counts.get(k, 0) + 1
total, tot_share = sum(counts.values()) + 1, sum(mix.TYPE_SHARE.values())
best = max(cands, key=lambda t: (total * mix.TYPE_SHARE.get(mix.kind_of_task_dir(t), 0) / tot_share
- counts.get(mix.kind_of_task_dir(t), 0), -int(t[1:])))
self.inflight.add((best, 0))
return best, 0
for t in tasks: # second attempt: first failed, or accepted without a repair
e = ev0.get(t)
if e and (t, 1) not in done and (t, 1) not in self.inflight and (not e["accepted"] or not e["repair"]):
@@ -275,6 +286,25 @@ class Pipeline:
a[1] += e["accepted"]
a[2] += e["accepted"] and e["repair"]
return "; ".join(f"{k} {v[1]}/{v[0]} (repair {v[2]})" for k, v in sorted(agg.items()))
kinds_t, kinds_r = mix.accepted_task_counts(), {}
for r, e in zip(rows, evs):
k = mix.kind_of_task_dir(r["task"])
a = kinds_r.setdefault(k, [0, 0])
a[0] += 1
a[1] += e["accepted"]
tt, tr_ = max(sum(kinds_t.values()), 1), max(sum(v[1] for v in kinds_r.values()), 1)
share_sum = sum(mix.TYPE_SHARE.values())
kind_line = "; ".join("%s tasks %d (%.0f%%) traj %d/%d (%.0f%%) target %.0f%%" % (
k, kinds_t.get(k, 0), 100.0 * kinds_t.get(k, 0) / tt, kinds_r.get(k, [0, 0])[1], kinds_r.get(k, [0, 0])[0],
100.0 * kinds_r.get(k, [0, 0])[1] / tr_, 100.0 * v / share_sum) for k, v in mix.TYPE_SHARE.items())
ddls = {}
for r, e in zip(rows, evs):
if mix.kind_of_task_dir(r["task"]) == "DDLS" and not e["accepted"]:
for w in (e["reasons"] or ["?"]):
w = w.split("=")[0] if w.startswith(("end_reason", "score")) else w
ddls[w] = ddls.get(w, 0) + 1
ddls_n = sum(1 for r in rows if mix.kind_of_task_dir(r["task"]) == "DDLS")
ddls_acc = sum(1 for r, e in zip(rows, evs) if mix.kind_of_task_dir(r["task"]) == "DDLS" and e["accepted"])
s = spent()
used = (s - BASE_LEDGER) / LEDGER_TO_USAGE
from .ledger import _env_budget
@@ -290,6 +320,8 @@ trajectory runs {len(rows)}, accepted trajectories {len(accepted)} (acceptance {
- By category, accepted/runs: {table('category')}.
- By object type, accepted/runs: {table('object_type')}.
- Tokens of accepted samples (20 tool schemas kept): p50 {pct(0.5)}, p90 {pct(0.9)}, p95 {pct(0.95)}, max {toks[-1] if toks else None}, n {len(toks)}.
- Object type mix (accepted tasks; accepted/runs trajectories; target): {kind_line}.
- DDLS (CDS) runs: {ddls_acc} accepted of {ddls_n}; reject reasons {ddls or 'none'} (activation, hidden tests and ATC rejections are separate gate failures, they show as score reasons).
- Syntax hints (proxy syntaxCheck added): {sum(e['hints'] for e in evs)} in {sum(1 for e in evs if e['hints'])} runs.
- Harness events: {sum(self.events.values())} ({self.events or 'none'}); trajectory workers now {self.max_workers}.
"""

View File

@@ -21,7 +21,7 @@ CLEAN_RULES = {
}
DELETE_ORDER = ["CLAS", "INTF", "PROG", "FUNC", "FUGR", "SRVD", "DDLX", "DCLS", "DDLS",
"TTYP", "TABL", "STRU", "DTEL", "DOMA", "MSAG"]
SOURCE_TYPES = ("CLAS", "INTF", "PROG", "FUNC", "DDLS", "DCLS", "DDLX", "TABL")
SOURCE_TYPES = ("CLAS", "INTF", "PROG", "FUNC", "DDLS", "DCLS", "DDLX", "TABL", "STRU")
def _cds_elements(src):
@@ -40,6 +40,13 @@ def _cds_elements(src):
return out
def _ddl_fields(src):
"""Field names of a DDL table or structure source ('key name : type', 'name : type', 'include x')."""
body = re.sub(r"//[^\n]*|/\*.*?\*/", "", src or "", flags=re.S)
return {m.group(1).upper() for m in re.finditer(r"^\s*(?:key\s+)?(\w+)\s*:", body, re.I | re.M)
if not m.group(1).startswith("@") and m.group(1).lower() not in ("define",)}
def _obj_args(otype, name, fg=None):
a = {"objectType": otype, "objectName": name}
if fg:
@@ -136,6 +143,9 @@ class Runner:
includeType="testclasses",
source=o["testclasses_source"]))
ok = ok and not e3 and (_json(t3) or {}).get("success", False)
if o.get("messages"): # message class (MSAG): created empty, messages written with sap_push_message
e5, t5 = mcp.call("sap_push_message", {"objectName": o["name"], "messages": o["messages"]})
ok = ok and not e5 and (_json(t5) or {}).get("success", False)
if o.get("run"):
e4, t4 = mcp.call("sap_run_class", {"className": o["name"]})
ok = ok and not e4 and (_json(t4) or {}).get("success", False)
@@ -285,9 +295,11 @@ class Runner:
g2_detail = []
for c in contract:
src = sources.get((c["type"], c["name"].upper()), "")
if c.get("implements") and not re.search(rf"INTERFACES\s+{re.escape(c['implements'])}\b", src, re.I):
g2 = False
g2_detail.append(f"{c['name']}: does not implement {c['implements']}")
impl = c.get("implements") or []
for iname in ([impl] if isinstance(impl, str) else impl): # one interface or a list
if not re.search(rf"INTERFACES\s+{re.escape(str(iname))}\b", src, re.I):
g2 = False
g2_detail.append(f"{c['name']}: does not implement {iname}")
if c["type"] == "FUNC": # signature: every parameter with its type in the FUNCTION header
# the FUNCTION statement, not the first statement: local classes can come before it (G0107)
fm = re.search(rf"^\s*FUNCTION\s+{re.escape(c['name'])}\b[^.]*\.", src, re.I | re.M)
@@ -303,15 +315,25 @@ class Runner:
if not re.search(rf"(PARAMETERS|SELECT-OPTIONS)\s*:?[^.]*\b{re.escape(prm)}\b", src, re.I):
g2 = False
g2_detail.append(f"{c['name']}: no selection screen parameter {prm}")
if c["type"] == "DDLS" and c.get("fields") and g["G1_active"]:
_, q = mcp.call("sap_sql_query", {"query": f"SELECT * FROM {c['name']}", "maxRows": 1})
cols = {col.get("name", "").upper() for col in (_json(q) or {}).get("columns", [])}
if c["type"] in ("DDLS", "TABL", "STRU") and c.get("fields") and g["G1_active"]:
cols = set()
if c["type"] != "STRU": # a structure cannot be selected
_, q = mcp.call("sap_sql_query", {"query": f"SELECT * FROM {c['name']}", "maxRows": 1})
cols = {col.get("name", "").upper() for col in (_json(q) or {}).get("columns", [])}
if not cols: # view with parameters: SELECT without parameters fails
cols = _cds_elements(src)
cols = _cds_elements(src) if c["type"] == "DDLS" else _ddl_fields(src)
missing = {f.upper() for f in c["fields"]} - cols
if missing:
g2 = False
g2_detail.append(f"{c['name']}: CDS elements missing: {sorted(missing)}")
g2_detail.append(f"{c['name']}: {'CDS elements' if c['type'] == 'DDLS' else 'fields'} missing: {sorted(missing)}")
if c["type"] == "MSAG" and c.get("messages") and g["G1_active"]: # the numbers must exist (texts: hidden tests)
_, q = mcp.call("sap_sql_query", {"query": "SELECT msgnr FROM t100 WHERE arbgb = '%s' AND sprsl = 'E'"
% c["name"].upper(), "maxRows": 999})
have = {r.get("MSGNR") for r in (_json(q) or {}).get("rows", [])}
missing = {str(m["msgno"]).zfill(3) for m in c["messages"]} - have
if missing:
g2 = False
g2_detail.append(f"{c['name']}: message numbers missing: {sorted(missing)}")
# Categories E and I: the contract object is a seed object (refactor or fix in place). It must
# change (else the null agent passes on the legacy code), and it is not out of scope.
contract_upper = {c["name"].upper() for c in contract}

View File

@@ -19,6 +19,7 @@ from .evalset import SLOTS, RELEASES, accepted_goals
from .generator import ROOT, generate, make_k_variant
from .ledger import BudgetExceeded, spent
from . import overlap
from . import mix
POOL = os.path.join(ROOT, "tasks_gen", "train")
PLAN = os.path.join(POOL, "plan.json")
@@ -186,6 +187,137 @@ def run(part, parts, target, stop_ledger, plan_name="plan", deadline=None):
print(json.dumps(log), flush=True)
BAL_FIRST_ID = 1910
BAL_RUN_BASE = 370000 # 40 per slot; below 466560 (a digit must lead the 4-char base36 run)
BAL_ERROR_KINDS = {"CLAS": ["named-type", "long-names"], "FUNC": ["named-type"], "DDLS": ["reserved-word"],
"TABL": ["reserved-word"], "STRU": ["reserved-word"]}
BAL_HINTS = dict(((c, t), h) for c, t, _, h in SLOTS if h)
def _claims_dir():
d = os.path.join(POOL, "_claims")
os.makedirs(d, exist_ok=True)
return d
def _claim_slot():
"""Next free balanced slot number, claimed with O_EXCL (several workers). Returns (n, claim path)."""
for n in range(BAL_FIRST_ID, BAL_FIRST_ID + 600):
sid = "G%04d" % n
if os.path.exists(os.path.join(POOL, "_logs", sid + ".json")):
continue
path = os.path.join(_claims_dir(), sid + ".json")
try:
fd = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY)
except FileExistsError:
continue
os.close(fd)
return n, path
return None, None
def _kind_stats():
"""({kind: accepted}, {kind: attempted}) from the generation logs; claims of running slots count as attempted."""
acc = mix.accepted_task_counts()
att = {}
for f in glob.glob(os.path.join(POOL, "_logs", "G*.json")):
try:
l = json.load(open(f))
except (OSError, ValueError):
continue
k = l.get("kind") or l.get("object_type")
if k and l.get("category") != "K":
att[k] = att.get(k, 0) + 1
return acc, att
def run_balanced(part, parts, deadline):
"""Generation without a fixed plan: each slot takes the kind with the biggest deficit against mix.TYPE_SHARE.
A kind with 6 or more tries and an acceptance below 20 % is skipped (a harness or prompt problem: do not burn budget)."""
base_url = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
evals = overlap.load_pool("eval")
while True:
while backlog() > BACKLOG_LIMIT and not os.path.exists(STOP_FLAG) and not (deadline and time.time() > deadline):
time.sleep(120)
if os.path.exists(STOP_FLAG):
print("STOP flag", flush=True)
return
if deadline and time.time() > deadline:
print("DEADLINE", flush=True)
return
acc, att = _kind_stats()
running = {}
for f in glob.glob(os.path.join(_claims_dir(), "G*.json")):
try:
k = json.load(open(f)).get("kind")
except (OSError, ValueError):
k = None
if k:
running[k] = running.get(k, 0) + 1
counts = {k: acc.get(k, 0) + running.get(k, 0) for k in set(acc) | set(running) | set(mix.TYPE_SHARE)}
blocked = {k for k in mix.TYPE_SHARE if att.get(k, 0) >= 6 and acc.get(k, 0) < 0.2 * att.get(k, 0)}
if blocked:
print("kinds skipped (low acceptance):", sorted(blocked), flush=True)
kind = mix.deficit_pick(counts, allowed=set(mix.TYPE_SHARE) - blocked)
n, claim = _claim_slot()
if n is None:
print("no free slot", flush=True)
return
sid = "G%04d" % n
json.dump({"kind": kind}, open(claim, "w"))
otype = "CLAS" if kind == "EXC" else kind
cats = mix.KIND_CATEGORIES[kind]
logs = [json.load(open(f)) for f in glob.glob(os.path.join(POOL, "_logs", "G*.json"))]
ccount = {c: sum(1 for l in logs if l.get("accepted") and l.get("category") == c) for c in cats}
tot = sum(ccount.values()) + 1
cat = max(cats, key=lambda c: tot * mix.CATEGORY_SHARE[c] / sum(mix.CATEGORY_SHARE[x] for x in cats) - ccount[c])
idx = n - BAL_FIRST_ID
error_kind = None
if idx % 5 == 2 and kind in BAL_ERROR_KINDS: # 20 % error-targeted slots
error_kind = BAL_ERROR_KINDS[kind][(idx // 5) % len(BAL_ERROR_KINDS[kind])]
cat = ERROR_CATEGORY[error_kind] if kind in ("CLAS", "FUNC") else cat
hard = cat != "H" and idx % 10 in (3, 6, 9) # 30 % hard
topic = None
if error_kind:
topic = dict(ERROR_HINTS)[error_kind]
elif kind == "EXC":
topic = "exception class (CX_...): " + ["a domain exception with context attributes and message texts",
"an exception hierarchy with a common super class",
"an exception that wraps a previous exception"][idx % 3]
elif (cat, otype) in BAL_HINTS:
h = BAL_HINTS[(cat, otype)]
topic = h[idx % len(h)]
if cat == "G":
topic = (topic + "; " if topic else "") + f"release target {RELEASES[idx % 2]}"
avoid = [g for g in accepted_goals() if g][-170:]
full = ((topic + ". ") if topic else "Choose a new, realistic business topic. ") + \
"Do not repeat these existing topics: " + "; ".join(avoid)
pool_now = evals + overlap.load_pool("train")
def extra(b, _pool=pool_now):
hits = overlap.check(overlap.load_bundle(b), _pool)
return [f"Too close to task {i} (similarity spec {sc['spec']:.2f}, rules {sc['core']:.2f}, "
f"names {sc['name']:.2f}). Choose a different business topic and different object names."
for i, sc in hits[:3]]
try:
log = generate(sid, "train", otype, cat, 3 if hard else 2, "deepseek-v4.1-flash:cloud", base_url,
BAL_RUN_BASE + 40 * idx, full, extra_check=extra)
except BudgetExceeded as e:
print("BUDGET", e, flush=True)
os.remove(claim)
return
except Exception as e: # noqa: BLE001
log = {"id": sid, "error": str(e)[:500]}
log.update(kind=kind, error_kind=error_kind, difficulty=3 if hard else 2, spent_total=spent())
os.makedirs(os.path.join(POOL, "_logs"), exist_ok=True)
json.dump(log, open(os.path.join(POOL, "_logs", sid + ".json"), "w"), indent=1)
stray = os.path.join(POOL, "generation.json")
if os.path.exists(stray):
os.remove(stray)
os.remove(claim)
print(json.dumps(log), flush=True)
K_FIRST_ID = 1300
K_RUN_BASE = 41800 # 20 per variant; above the trajectory run numbers (41000-41700)
K_COUNT = 18 # K share of the eval plan: 10 of 110 (9 %); counted inside the 200 accepted tasks
@@ -240,7 +372,7 @@ def main():
ap.add_argument("--part", type=int, default=0)
ap.add_argument("--parts", type=int, default=1)
ap.add_argument("--target", type=int, default=200)
ap.add_argument("--plan", default="plan", help="plan (first 223 slots) or plan2 (hard and error share raised)")
ap.add_argument("--plan", default="plan", help="plan (first 223 slots), plan2 (hard and error share raised) or balanced (kind with the biggest deficit)")
ap.add_argument("--deadline", help="YYYY-MM-DDTHH:MM local time: no new slot after it")
ap.add_argument("--k-count", type=int, default=K_COUNT)
ap.add_argument("--stop-ledger", type=float, default=27.0, help="ledger USD for this phase (10 USD usage = 27)")
@@ -255,6 +387,9 @@ def main():
run_k(a.k_count)
return
dl = time.mktime(time.strptime(a.deadline, "%Y-%m-%dT%H:%M")) if a.deadline else None
if a.plan == "balanced":
run_balanced(a.part, a.parts, dl)
return
run(a.part, a.parts, a.target, a.stop_ledger, a.plan, dl)