Balanced generation under the pipeline, per-kind brake; STATE: type mix fix

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-05 19:27:43 +02:00
parent 55f4068330
commit 976274d48a
3 changed files with 42 additions and 8 deletions

View File

@@ -221,7 +221,7 @@ class Pipeline:
def old_generators(self): def old_generators(self):
out = subprocess.run(["pgrep", "-fl", r"harness\.trainset run --part"], capture_output=True, text=True).stdout out = subprocess.run(["pgrep", "-fl", r"harness\.trainset run --part"], capture_output=True, text=True).stdout
return [l for l in out.splitlines() if "--plan plan2" not in l] return [l for l in out.splitlines() if "--plan balanced" not in l and "--plan plan2" not in l]
def supervisor(self): def supervisor(self):
launched = False launched = False
@@ -232,15 +232,14 @@ class Pipeline:
if time.time() > self.gen_deadline or os.path.exists(trainset.STOP_FLAG): if time.time() > self.gen_deadline or os.path.exists(trainset.STOP_FLAG):
continue continue
if not launched and not self.old_generators(): if not launched and not self.old_generators():
trainset.ensure_plan("plan2")
dl = self.a.gen_deadline dl = self.a.gen_deadline
for i in range(3): for i in range(3):
f = open(os.path.join(ROOT, "runs", "gen_train", f"p2_w{i}.log"), "a") f = open(os.path.join(ROOT, "runs", "gen_train", f"p2_w{i}.log"), "a")
self.children.append(subprocess.Popen( self.children.append(subprocess.Popen(
[sys.executable, "-m", "harness.trainset", "run", "--plan", "plan2", "--part", str(i), [sys.executable, "-m", "harness.trainset", "run", "--plan", "balanced", "--part", str(i),
"--parts", "3", "--deadline", dl], cwd=ROOT, stdout=f, stderr=f)) "--parts", "3", "--deadline", dl], cwd=ROOT, stdout=f, stderr=f))
launched = True launched = True
log("plan 2 generators started (3 workers, deadline", dl + ")") log("balanced generators started (3 workers, deadline", dl + ")")
elif launched: elif launched:
for i, c in enumerate(self.children): for i, c in enumerate(self.children):
if c.poll() not in (None, 0) and crashes < 3: if c.poll() not in (None, 0) and crashes < 3:
@@ -248,7 +247,7 @@ class Pipeline:
log("generator", i, "exited with", c.returncode, "- restarted") log("generator", i, "exited with", c.returncode, "- restarted")
f = open(os.path.join(ROOT, "runs", "gen_train", f"p2_w{i}.log"), "a") f = open(os.path.join(ROOT, "runs", "gen_train", f"p2_w{i}.log"), "a")
self.children[i] = subprocess.Popen( self.children[i] = subprocess.Popen(
[sys.executable, "-m", "harness.trainset", "run", "--plan", "plan2", "--part", str(i), [sys.executable, "-m", "harness.trainset", "run", "--plan", "balanced", "--part", str(i),
"--parts", "3", "--deadline", self.a.gen_deadline], cwd=ROOT, stdout=f, stderr=f) "--parts", "3", "--deadline", self.a.gen_deadline], cwd=ROOT, stdout=f, stderr=f)
# K variants: free text, about 10 % of the other accepted training tasks # K variants: free text, about 10 % of the other accepted training tasks
if k_proc is None or k_proc.poll() is not None: if k_proc is None or k_proc.poll() is not None:

View File

@@ -133,6 +133,28 @@ def backlog():
return len(acc - done) return len(acc - done)
def backlog_by_kind():
"""{kind: accepted tasks without a first trajectory run}."""
done = set()
sp = os.path.join(ROOT, "runs", "traj", "summary.jsonl")
if os.path.exists(sp):
done = {json.loads(l)["task"] for l in open(sp) if json.loads(l)["attempt"] == 0}
out = {}
for f in glob.glob(os.path.join(POOL, "G*", "generation.json")):
tid = os.path.basename(os.path.dirname(f))
try:
ok = json.load(open(f)).get("accepted")
except (OSError, ValueError):
ok = False
if ok and tid not in done:
k = mix.kind_of_task_dir(tid)
out[k] = out.get(k, 0) + 1
return out
BAL_KIND_BACKLOG = 8 # a kind with more waiting tasks than this is not generated (its trajectories come first)
def run(part, parts, target, stop_ledger, plan_name="plan", deadline=None): def run(part, parts, target, stop_ledger, plan_name="plan", deadline=None):
plan = ensure_plan(plan_name, stop_ledger, target) plan = ensure_plan(plan_name, stop_ledger, target)
stop_at = plan["ledger_at_start"] + plan["stop_ledger"] stop_at = plan["ledger_at_start"] + plan["stop_ledger"]
@@ -237,8 +259,6 @@ def run_balanced(part, parts, deadline):
base_url = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1") base_url = os.environ.get("LLM_BASE_URL", "http://127.0.0.1:11434/v1")
evals = overlap.load_pool("eval") evals = overlap.load_pool("eval")
while True: while True:
while backlog() > BACKLOG_LIMIT and not os.path.exists(STOP_FLAG) and not (deadline and time.time() > deadline):
time.sleep(120)
if os.path.exists(STOP_FLAG): if os.path.exists(STOP_FLAG):
print("STOP flag", flush=True) print("STOP flag", flush=True)
return return
@@ -258,7 +278,12 @@ def run_balanced(part, parts, deadline):
blocked = {k for k in mix.TYPE_SHARE if att.get(k, 0) >= 6 and acc.get(k, 0) < 0.2 * att.get(k, 0)} blocked = {k for k in mix.TYPE_SHARE if att.get(k, 0) >= 6 and acc.get(k, 0) < 0.2 * att.get(k, 0)}
if blocked: if blocked:
print("kinds skipped (low acceptance):", sorted(blocked), flush=True) print("kinds skipped (low acceptance):", sorted(blocked), flush=True)
kind = mix.deficit_pick(counts, allowed=set(mix.TYPE_SHARE) - blocked) waiting = {k for k, n in backlog_by_kind().items() if n > BAL_KIND_BACKLOG}
allowed = set(mix.TYPE_SHARE) - blocked - waiting
if not allowed:
time.sleep(120) # every kind has a backlog: the trajectories are the slower side
continue
kind = mix.deficit_pick(counts, allowed=allowed)
n, claim = _claim_slot() n, claim = _claim_slot()
if n is None: if n is None:
print("no free slot", flush=True) print("no free slot", flush=True)

View File

@@ -165,3 +165,13 @@ Training runs on HF Jobs with Unsloth, not on the Mac. No `mlx_lm` training.
- **Token length (for the bf16 memory test):** accepted samples, 20 tool schemas kept: p50 20k, p90 39k, **p95 48k, max 72k** (about 10k tokens are the tool schemas). The bf16 memory test must use a **sequence length of 48k** (covers 95 % of the samples; samples over 48k are cut or dropped; 72k is not needed in the test). - **Token length (for the bf16 memory test):** accepted samples, 20 tool schemas kept: p50 20k, p90 39k, **p95 48k, max 72k** (about 10k tokens are the tool schemas). The bf16 memory test must use a **sequence length of 48k** (covers 95 % of the samples; samples over 48k are cut or dropped; 72k is not needed in the test).
- **DDLS reject reasons** (11 CDS runs: 6 accepted, 5 rejected): tool budget 2 (old limit 60, before the 100 budget), empty response 3 (one turn reached the 32k output limit while the model wrote its own CDS test class). Activation 0, hidden tests 0, ATC 0: all 11 runs passed every gate and every hidden test. Failed CDS tasks get the second attempt (pipeline rule). - **DDLS reject reasons** (11 CDS runs: 6 accepted, 5 rejected): tool budget 2 (old limit 60, before the 100 budget), empty response 3 (one turn reached the 32k output limit while the model wrote its own CDS test class). Activation 0, hidden tests 0, ATC 0: all 11 runs passed every gate and every hidden test. Failed CDS tasks get the second attempt (pipeline rule).
- **G1034 lock event (cause not found):** after `sap_push_source includeType=testclasses` the class stayed locked ("User KESELI is currently editing", SM12 entry needed; teardown said "You are already editing"). Not a harness bug: the lock outlived the MCP session and the run. Not reproduced in 3 tests on A4H (alone, 60 writes overlapping 2690 unit test calls of another session, writes with ATC + coverage + member listing + a second session; `push_element` on a test-class method). One case in about 90 runs. Details for the EPOD developer: `docs/epod-lock-leak.md`. The harness treats it as an infrastructure event (run not accepted, worker count 2 to 1, 3 equal events stop the pipeline) and lists the object in the dashboard. - **G1034 lock event (cause not found):** after `sap_push_source includeType=testclasses` the class stayed locked ("User KESELI is currently editing", SM12 entry needed; teardown said "You are already editing"). Not a harness bug: the lock outlived the MCP session and the run. Not reproduced in 3 tests on A4H (alone, 60 writes overlapping 2690 unit test calls of another session, writes with ATC + coverage + member listing + a second session; `push_element` on a test-class method). One case in about 90 runs. Details for the EPOD developer: `docs/epod-lock-leak.md`. The harness treats it as an infrastructure event (run not accepted, worker count 2 to 1, 3 equal events stop the pipeline) and lists the object in the dashboard.
### Object type mix fix (2026-10-05 19:30, Opus review item 1)
- **Cause:** plans 1 and 2 were built from `evalset.SLOTS`, which has only CLAS, FUNC, PROG and DDLS. The mix of CLAUDE.md section 1 (INTF, DDIC, MSAG, exception) was never in a plan, so no slot existed for those types and nothing was rejected. The harness also lacked G2 checks, message writing and mutants for them (`evalset.py` header said so).
- **Before (139 accepted tasks):** CLAS 49 %, FUNC 21 %, DDLS 19 %, PROG 9 %, INTF/TABL/STRU/MSAG 0, exception 2 %. Target: CLAS 28, INTF 7, DDLS 25, FUNC 15, PROG 10, TABL 8, STRU 2, MSAG 2.5, exception 2.5 (percent; `harness/mix.py`).
- **Fixes:** G2 for TABL/STRU fields and MSAG message numbers, `sap_push_message` in the oracle and setup, mutants for DDIC, interfaces, message classes and exception classes, G6 cascade fix (abaplint does not know CX_ super classes), format notes per type in the generator, `implements` may be a list. DTEL/DOMA are not made (DDIC share is TABL 8 + STRU 2).
- **First tasks per new type, all accepted with oracle 100, null 0 and mutation kill rate 100 %:** G1900 INTF (3 tries), G1901 TABL (1), G1902 MSAG (2), G1903 STRU (3), G1904 exception (mutation rerun after the exception mutants). Model mistakes fed back by the generator: unit/currency reference annotation in DDL needs `'table.field'`, RTTI length is in bytes.
- **Generation:** `trainset run --plan balanced` (3 workers, started by the pipeline): every slot takes the kind with the biggest deficit against the target; a kind is skipped with 6+ tries and under 20 % accepted, or with more than 8 accepted tasks waiting for a first trajectory run. 20 % error-targeted, 30 % hard (difficulty 3). K variants stay at 10 %.
- **Trajectories:** the next task is the one whose kind has the biggest deficit (accepted trajectories per kind). Summary now lists the mix and DDLS reject reasons.
- **Incident 19:25:** restarting the controller I started a second one by mistake (wrong `pgrep` pattern) and deleted the objects of a running run. Both controllers were stopped, leftovers cleaned, one controller runs. One table `Z4AJ50UB_PO_HEAD` kept a lock from the interrupted write (SM12 needed). Interrupting a write leaks the lock: never kill a controller during a run without checking.