"""Stage 2 training set builder (Opus item C, 2026-10-06). train/.venv/bin/python train/build_stage2.py [--out runs/stage2_data] [--max-tokens 48000] [--clas-cap 0.35] [--min-per-kind 25] [--valid-frac 0.10] [--hook module:function] Input: runs/traj (DeepSeek trajectories; the local Qwen series is never read). Steps: acceptance filter (score, end reason, no harness text, loop repeats trimmed) -> eval overlap check -> Qwen 3.8 chat template (thinking off) with the tokenizer round trip check -> samples over --max-tokens are dropped (never cut) -> CLAS capped at --clas-cap of the stage 2 samples (the rest goes to `stage2_reserve.jsonl`) -> shortage report for kinds under --min-per-kind -> validation split by task family (K variants and all trajectories of a task on the same side) -> optional hook (own-test score, weights) -> files + data card. Loss mask: `assistant_spans` are character spans of the assistant turns (tool calls and the final report, with <|im_end|>); nothing of the system turn (tool schemas), the user turn or the tool results is in a span. `assistant_tokens` counts the loss tokens. """ import argparse import importlib import re import json import os import random import statistics import sys import time from collections import Counter, defaultdict ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) sys.path.insert(0, ROOT) sys.path.insert(0, os.path.join(ROOT, "train")) import accept as acc # noqa: E402 import to_qwen as tq # noqa: E402 from harness import mix, overlap # noqa: E402 from transformers import AutoTokenizer # noqa: E402 POOL = os.path.join(ROOT, "tasks_gen", "train") START_PREFIX = re.compile(r"^(Z\d[0-9A-Z]{6}_)", re.I) MID_PREFIX = re.compile(r"^[A-Z]{1,5}_(Z\d[0-9A-Z]{6}_)", re.I) LIST_TOOLS = ("sap_inactive_objects", "sap_search_object", "sap_usage_references") READ_TOOLS = ("sap_pull_source", "sap_object_structure", "sap_object_members", "sap_element_info", "sap_run_unit_test", "sap_check_object", "sap_syntax_check", "sap_atc_run") def foreign_name(name, own): n = (name or "").upper() m = START_PREFIX.match(n) or MID_PREFIX.match(n) return bool(m) and m.group(1).upper() != own.upper() def scrub_foreign(msgs, own_prefix): """Remove other runs' leftover objects from list results (the proxy hides them since 2026-10-06; the older records still have them). Returns (messages, removed list entries, number of reads of a foreign object).""" tool_of = {} for m in msgs: if m["role"] == "assistant": for c in m.get("tool_calls") or []: a = c["function"].get("arguments") or "{}" try: a = json.loads(a) if isinstance(a, str) else a except ValueError: a = {} tool_of[c.get("id")] = (c["function"]["name"], a) removed = reads = 0 out = [] for m in msgs: if m["role"] == "tool": name, args = tool_of.get(m.get("tool_call_id"), ("", {})) if name in READ_TOOLS and foreign_name(args.get("objectName"), own_prefix): reads += 1 if name in LIST_TOOLS: body = m["content"] prefix = "ERROR: " if body.startswith("ERROR: ") else "" try: data = json.loads(body[len(prefix):]) except ValueError: data = None if isinstance(data, list): keep = [d for d in data if not (isinstance(d, dict) and foreign_name(d.get("name"), own_prefix))] if len(keep) != len(data): removed += len(data) - len(keep) m = dict(m, content=prefix + json.dumps(keep)) out.append(m) return out, removed, reads def pct(v, p): v = sorted(v) return v[min(len(v) - 1, int(p * len(v)))] if v else None def family_of(task_id): try: t = json.load(open(os.path.join(POOL, task_id, "task.json"))) except OSError: return task_id return t.get("base_task") or task_id def loss_tokens(tok, text, spans): enc = tok(text, add_special_tokens=False, return_offsets_mapping=True) n = 0 for (a, b) in enc["offset_mapping"]: if any(a >= s and b <= e for s, e in spans): n += 1 return len(enc["input_ids"]), n def main(): ap = argparse.ArgumentParser() ap.add_argument("--traj", default=os.path.join(ROOT, "runs", "traj")) ap.add_argument("--out", default=os.path.join(ROOT, "runs", "stage2_data")) ap.add_argument("--max-tokens", type=int, default=48000) ap.add_argument("--clas-cap", type=float, default=0.35) ap.add_argument("--min-per-kind", type=int, default=25) ap.add_argument("--min-score", type=float, default=80) ap.add_argument("--valid-frac", type=float, default=0.10) ap.add_argument("--seed", type=int, default=20261006) ap.add_argument("--no-scrub", action="store_true", help="keep other runs' leftover objects in list results (default: removed)") ap.add_argument("--keep-foreign-reads", action="store_true", help="keep trajectories in which the model read another run's object (default: dropped)") ap.add_argument("--hook", help="module:function; function(row) -> None to drop, or a float weight (repeat factor), or a dict " "{'keep': bool, 'weight': float, 'extra': {...}} (for example the own-test mutation score)") a = ap.parse_args() os.makedirs(a.out, exist_ok=True) hook = None if a.hook: mod, fn = a.hook.split(":") sys.path.insert(0, os.path.join(ROOT, "train")) hook = getattr(importlib.import_module(mod), fn) tok = AutoTokenizer.from_pretrained(os.path.expanduser("~/models/Qwen3.8-27B-4bit")) evals = overlap.load_pool("eval") report = {"built": time.strftime("%F %T"), "settings": vars(a), "dropped": defaultdict(Counter), "steps": {}} # 1 accepted trajectories rows = [json.loads(l) for l in open(os.path.join(a.traj, "summary.jsonl"))] cand = [] for r in rows: p = os.path.join(a.traj, r.get("run_dir") or "-", "record.json") if not os.path.exists(p): continue rec = json.load(open(p)) ok, why = acc.judge(rec, r, a.min_score) if not ok: continue msgs, removed = acc.trim_loops(rec["messages"]) scrubbed = freads = 0 if not a.no_scrub: msgs, scrubbed, freads = scrub_foreign(msgs, rec["prefix"]) if freads and not a.keep_foreign_reads: kind0 = mix.kind_of_task_dir(r["task"]) report["dropped"][kind0]["read_of_another_runs_object"] += 1 continue cand.append({"rec": rec, "msgs": msgs, "removed": removed, "row": r, "repair": acc.has_repair(msgs), "scrubbed": scrubbed, "freads": freads}) report["steps"]["accepted_trajectories"] = len(cand) # 2 eval overlap at build time (the task against every eval task) kept = [] for c in cand: tid = c["row"]["task"] kind = mix.kind_of_task_dir(tid) try: hits = overlap.check(overlap.load_task(os.path.join(POOL, tid)), evals) except (OSError, ValueError): hits = [] if hits: report["dropped"][kind]["eval_overlap"] += 1 continue c["kind"] = kind kept.append(c) report["steps"]["after_eval_overlap"] = len(kept) # 3 Qwen template + tokens + loss tokens; drop over the limit (never cut) samples = [] for c in kept: msgs = tq.to_template_messages(c["msgs"]) tools = c["rec"]["tools"] text = tok.apply_chat_template(msgs, tools=tools, tokenize=False, enable_thinking=False) prefix = tok.apply_chat_template(msgs[:2], tools=tools, tokenize=False, add_generation_prompt=True, enable_thinking=False) if not text.startswith(prefix) or not tq.check_calls(msgs, text): report["dropped"][c["kind"]]["template_roundtrip"] += 1 continue spans = tq.spans(text) n, nl = loss_tokens(tok, text, spans) if n > a.max_tokens: report["dropped"][c["kind"]]["over_%dk" % (a.max_tokens // 1000)] += 1 continue t = c["rec"]["task"] samples.append({"id": f"{c['row']['task']}_r{c['row']['run']}", "task": c["row"]["task"], "family": family_of(c["row"]["task"]), "kind": c["kind"], "category": t.get("category"), "object_type": t.get("object_type"), "attempt": c["row"]["attempt"], "repair": c["repair"], "loop_pairs_removed": c["removed"], "foreign_entries_removed": c["scrubbed"], "score": c["rec"]["metadata"].get("score"), "teacher": c["rec"]["teacher"], "weight": 1.0, "n_tokens": n, "assistant_tokens": nl, "text": text, "assistant_spans": spans}) report["steps"]["after_token_limit"] = len(samples) # 4 CLAS cap: keep the CLAS samples with a repair and the highest score first; the rest is the reserve by_kind = defaultdict(list) for s in samples: by_kind[s["kind"]].append(s) non_clas = sum(len(v) for k, v in by_kind.items() if k != "CLAS") clas = sorted(by_kind.get("CLAS", []), key=lambda s: (not s["repair"], -(s["score"] or 0), s["id"])) limit = int(a.clas_cap / (1 - a.clas_cap) * non_clas) if non_clas else len(clas) reserve = clas[limit:] by_kind["CLAS"] = clas[:limit] report["steps"]["clas_cap"] = {"clas_before": len(clas), "clas_kept": len(by_kind["CLAS"]), "clas_reserve": len(reserve), "limit": limit} pool = [s for v in by_kind.values() for s in v] # 5 shortage report (never filled by copies) report["kinds"] = {} for k in mix.TYPE_SHARE: n = len(by_kind.get(k, [])) report["kinds"][k] = {"samples": n, "min_required": a.min_per_kind, "short": max(0, a.min_per_kind - n), "families": len({s["family"] for s in by_kind.get(k, [])})} # 6 hook (own-test score, weights) if hook: out = [] for s in pool: r = hook(s) if r is None or r is False: report["dropped"][s["kind"]]["hook"] += 1 continue if isinstance(r, dict): if not r.get("keep", True): report["dropped"][s["kind"]]["hook"] += 1 continue s["weight"] = float(r.get("weight", 1.0)) s.update(r.get("extra") or {}) elif isinstance(r, (int, float)) and not isinstance(r, bool): s["weight"] = float(r) out.append(s) pool = out # 7 validation split by family, stratified by kind rnd = random.Random(a.seed) fam_kind = {} for s in pool: fam_kind.setdefault(s["family"], s["kind"]) valid_fam = set() for k in mix.TYPE_SHARE: fams = sorted(f for f, kk in fam_kind.items() if kk == k) rnd.shuffle(fams) nv = round(len(fams) * a.valid_frac) if len(fams) >= 4: nv = max(nv, 1) valid_fam |= set(fams[:nv]) train = [s for s in pool if s["family"] not in valid_fam] valid = [s for s in pool if s["family"] in valid_fam] for name, data in (("stage2_train", train), ("stage2_valid", valid), ("stage2_reserve", reserve)): with open(os.path.join(a.out, name + ".jsonl"), "w") as f: for s in data: f.write(json.dumps(s) + "\n") # 8 report and data card def summary(data): toks = [s["n_tokens"] for s in data] return {"samples": len(data), "tokens": sum(toks), "loss_tokens": sum(s["assistant_tokens"] for s in data), "p50": pct(toks, .5), "p90": pct(toks, .9), "p95": pct(toks, .95), "max": max(toks or [0]), "by_kind": dict(Counter(s["kind"] for s in data)), "by_category": dict(Counter(s["category"] for s in data)), "repair_share": round(sum(s["repair"] for s in data) / max(len(data), 1), 2)} report["train"], report["valid"], report["reserve"] = summary(train), summary(valid), summary(reserve) report["dropped"] = {k: dict(v) for k, v in report["dropped"].items()} s1 = json.load(open(os.path.join(ROOT, "train", "data", "stats.json"))) if os.path.exists(os.path.join(ROOT, "train", "data", "stats.json")) else {} report["stage1_stats"] = s1 json.dump(report, open(os.path.join(a.out, "build_report.json"), "w"), indent=1, default=str) open(os.path.join(a.out, "README.md"), "w").write(data_card(report)) print(json.dumps({k: report[k] for k in ("steps", "train", "valid", "reserve", "dropped")}, indent=1, default=str)[:3500]) print("short kinds:", {k: v["short"] for k, v in report["kinds"].items() if v["short"]}) def data_card(r): t, v = r["train"], r["valid"] kinds = "\n".join(f"| {k} | {x['samples']} | {x['families']} | {x['short'] or ''} |" for k, x in r["kinds"].items()) drop = "\n".join(f"- {k}: {d}" for k, d in r["dropped"].items()) or "- nothing dropped" return f"""--- license: mit task_categories: [text-generation] tags: [abap, sap, agentic, tool-use, sft] private: true --- # ABAP stage 2 agent trajectories (built {r['built']}) Tool-using ABAP development trajectories for supervised fine-tuning of Qwen 3.8 27B. A teacher model solved generated ABAP tasks on a real SAP ABAP Platform system (A4H, SAP_BASIS 816) through ADT tools; only runs that passed the harness gates and scored at least {r['settings']['min_score']:.0f} of 100 are kept. ## Sources and licenses - Teacher: DeepSeek V4.1 Flash (MIT license), via Ollama cloud. No output of Claude or other restricted models. - Tasks: generated by the same teacher (spec, seed objects, hidden ABAP Unit tests, reference), validated on A4H (oracle 100, null 0, mutation check). Eval tasks are not in this data (overlap check at build time). - The local Qwen runs (series A) are not in this data. ## Format One JSON per line: `text` (the whole conversation in the Qwen 3.8 chat template, thinking off, Qwen XML tool call format, the 20 tool schemas in the system turn), `assistant_spans` (character spans that carry the loss: assistant turns with tool calls and the final report; none of the system turn, user turn or tool results), `n_tokens`, `assistant_tokens`, `kind`, `category`, `object_type`, `family` (task family: a K variant has the family of its base task), `repair` (an error followed by a fix), `score`, `weight`. ## Size | split | samples | tokens | loss tokens | p50 | p90 | p95 | max | repair share | |---|---|---|---|---|---|---|---|---| | train | {t['samples']} | {t['tokens']} | {t['loss_tokens']} | {t['p50']} | {t['p90']} | {t['p95']} | {t['max']} | {t['repair_share']} | | valid | {v['samples']} | {v['tokens']} | {v['loss_tokens']} | {v['p50']} | {v['p90']} | {v['p95']} | {v['max']} | {v['repair_share']} | Samples over {r['settings']['max_tokens']} tokens are dropped, never cut. CLAS is capped at {int(100 * r['settings']['clas_cap'])} % of the samples (the rest is in `stage2_reserve.jsonl`). ## Kinds (target share: CLAS 28, INTF 7, CDS 25, FUNC 15, PROG 10, TABL 8, STRU 2, MSAG 2.5, exception 2.5 percent) | kind | samples | families | short of the minimum {r['settings']['min_per_kind']} | |---|---|---|---| {kinds} ## Dropped {drop} ## Known limits - Small data (see the table); kinds marked short have fewer than the minimum samples. - Until 2026-10-06 the test system still held leftover objects of earlier runs (the model's own `ZCL__...` classes). Their names are removed from list results (`sap_inactive_objects`, searches) in these samples; trajectories in which the model read such an object are dropped. - The proxy added local abaplint messages (`syntaxCheck`) to some failed writes; the EPOD server does not do this itself. - The teacher sees only the tool results of its own runs; a run that repaired an error is kept with its error (60 to 65 % of the samples). - Validation split is by task family; do not mix valid samples into the training set. """ if __name__ == "__main__": main()