stage 2 builder: scrub of other runs' leftover objects, drop of foreign reads; rebuild (36 train + 4 valid), doc generator, HF dataset updated

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-06 06:16:02 +02:00
parent c2c3803b1b
commit c5f1a36df0
3 changed files with 163 additions and 27 deletions

View File

@@ -14,6 +14,7 @@ nothing of the system turn (tool schemas), the user turn or the tool results is
"""
import argparse
import importlib
import re
import json
import os
import random
@@ -33,6 +34,55 @@ from transformers import AutoTokenizer # noqa: E402
POOL = os.path.join(ROOT, "tasks_gen", "train")
START_PREFIX = re.compile(r"^(Z\d[0-9A-Z]{6}_)", re.I)
MID_PREFIX = re.compile(r"^[A-Z]{1,5}_(Z\d[0-9A-Z]{6}_)", re.I)
LIST_TOOLS = ("sap_inactive_objects", "sap_search_object", "sap_usage_references")
READ_TOOLS = ("sap_pull_source", "sap_object_structure", "sap_object_members", "sap_element_info", "sap_run_unit_test", "sap_check_object",
"sap_syntax_check", "sap_atc_run")
def foreign_name(name, own):
n = (name or "").upper()
m = START_PREFIX.match(n) or MID_PREFIX.match(n)
return bool(m) and m.group(1).upper() != own.upper()
def scrub_foreign(msgs, own_prefix):
"""Remove other runs' leftover objects from list results (the proxy hides them since 2026-10-06; the older records still have them).
Returns (messages, removed list entries, number of reads of a foreign object)."""
tool_of = {}
for m in msgs:
if m["role"] == "assistant":
for c in m.get("tool_calls") or []:
a = c["function"].get("arguments") or "{}"
try:
a = json.loads(a) if isinstance(a, str) else a
except ValueError:
a = {}
tool_of[c.get("id")] = (c["function"]["name"], a)
removed = reads = 0
out = []
for m in msgs:
if m["role"] == "tool":
name, args = tool_of.get(m.get("tool_call_id"), ("", {}))
if name in READ_TOOLS and foreign_name(args.get("objectName"), own_prefix):
reads += 1
if name in LIST_TOOLS:
body = m["content"]
prefix = "ERROR: " if body.startswith("ERROR: ") else ""
try:
data = json.loads(body[len(prefix):])
except ValueError:
data = None
if isinstance(data, list):
keep = [d for d in data if not (isinstance(d, dict) and foreign_name(d.get("name"), own_prefix))]
if len(keep) != len(data):
removed += len(data) - len(keep)
m = dict(m, content=prefix + json.dumps(keep))
out.append(m)
return out, removed, reads
def pct(v, p):
v = sorted(v)
return v[min(len(v) - 1, int(p * len(v)))] if v else None
@@ -65,6 +115,8 @@ def main():
ap.add_argument("--min-score", type=float, default=80)
ap.add_argument("--valid-frac", type=float, default=0.10)
ap.add_argument("--seed", type=int, default=20261006)
ap.add_argument("--no-scrub", action="store_true", help="keep other runs' leftover objects in list results (default: removed)")
ap.add_argument("--keep-foreign-reads", action="store_true", help="keep trajectories in which the model read another run's object (default: dropped)")
ap.add_argument("--hook", help="module:function; function(row) -> None to drop, or a float weight (repeat factor), or a dict "
"{'keep': bool, 'weight': float, 'extra': {...}} (for example the own-test mutation score)")
a = ap.parse_args()
@@ -90,7 +142,14 @@ def main():
if not ok:
continue
msgs, removed = acc.trim_loops(rec["messages"])
cand.append({"rec": rec, "msgs": msgs, "removed": removed, "row": r, "repair": acc.has_repair(msgs)})
scrubbed = freads = 0
if not a.no_scrub:
msgs, scrubbed, freads = scrub_foreign(msgs, rec["prefix"])
if freads and not a.keep_foreign_reads:
kind0 = mix.kind_of_task_dir(r["task"])
report["dropped"][kind0]["read_of_another_runs_object"] += 1
continue
cand.append({"rec": rec, "msgs": msgs, "removed": removed, "row": r, "repair": acc.has_repair(msgs), "scrubbed": scrubbed, "freads": freads})
report["steps"]["accepted_trajectories"] = len(cand)
# 2 eval overlap at build time (the task against every eval task)
@@ -127,7 +186,7 @@ def main():
t = c["rec"]["task"]
samples.append({"id": f"{c['row']['task']}_r{c['row']['run']}", "task": c["row"]["task"], "family": family_of(c["row"]["task"]),
"kind": c["kind"], "category": t.get("category"), "object_type": t.get("object_type"),
"attempt": c["row"]["attempt"], "repair": c["repair"], "loop_pairs_removed": c["removed"],
"attempt": c["row"]["attempt"], "repair": c["repair"], "loop_pairs_removed": c["removed"], "foreign_entries_removed": c["scrubbed"],
"score": c["rec"]["metadata"].get("score"), "teacher": c["rec"]["teacher"], "weight": 1.0,
"n_tokens": n, "assistant_tokens": nl, "text": text, "assistant_spans": spans})
report["steps"]["after_token_limit"] = len(samples)
@@ -250,6 +309,7 @@ Samples over {r['settings']['max_tokens']} tokens are dropped, never cut. CLAS i
## Known limits
- Small data (see the table); kinds marked short have fewer than the minimum samples.
- Until 2026-10-06 the test system still held leftover objects of earlier runs (the model's own `ZCL_<prefix>_...` classes). Their names are removed from list results (`sap_inactive_objects`, searches) in these samples; trajectories in which the model read such an object are dropped.
- The proxy added local abaplint messages (`syntaxCheck`) to some failed writes; the EPOD server does not do this itself.
- The teacher sees only the tool results of its own runs; a run that repaired an error is kept with its error (60 to 65 % of the samples).
- Validation split is by task family; do not mix valid samples into the training set.