From 35f0eb0e0cca3283e0b687076a3d6229d4ebd45e Mon Sep 17 00:00:00 2001 From: Kral Date: Mon, 5 Oct 2026 12:04:11 +0200 Subject: [PATCH] Overlap limit for whole spec 0.75, slot logs; STATE and roadmap: stage 2 data progress Co-Authored-By: Claude Sonnet 5.5 --- docs/yol-haritasi.md | 4 +++- harness/overlap.py | 2 +- harness/trainset.py | 9 +++++++-- train/STATE.md | 20 ++++++++++++++++++++ 4 files changed, 31 insertions(+), 4 deletions(-) diff --git a/docs/yol-haritasi.md b/docs/yol-haritasi.md index a872f97..7da1370 100644 --- a/docs/yol-haritasi.md +++ b/docs/yol-haritasi.md @@ -1,6 +1,6 @@ # ABAP Danışman Modeli — Yol Haritası -Durum: v21 · 2026-10-05 +Durum: v22 · 2026-10-05 Şu anki konum: **Adım 1.3** ## Adım 0 — Çerçeve ve kararlar @@ -130,6 +130,8 @@ Bitti sayılır: tek baz model seçildi, gerekçesi yazıldı. ## Adım 3 — Eğitim verisi üretimi +Durum (2026-10-05): 3.1 eğitim modu hazır (`harness/trainset.py`, havuz `tasks_gen/train`, eval ile çakışma kontrolü `harness/overlap.py`, 30 hata-odaklı görev dahil), üretim 12:00'de başladı (hedef 200 kabul, ~10 $ usage). 3.2 için proxy (yerel abaplint ile "save failed" ayrıntısı), yörünge kaydı (`harness/record.py`), Qwen çevirici (`train/to_qwen.py`) ve kabul süzgeci (`train/accept.py`) hazır; yörünge koşuları (`harness/trajectories.py`, 6 worker) üretim bitince. + | Alt adım | İş | |---|---| | 3.1 | Görev üreteci: eval görevleriyle aynı şemada binlerce sentetik görev. Eval görevleriyle çakışma yok. Girdilerin bir kısmı serbest metin; yörüngelerin az bir kısmı farklı tool isim/şemalarıyla. | diff --git a/harness/overlap.py b/harness/overlap.py index 410b37b..f37885a 100644 --- a/harness/overlap.py +++ b/harness/overlap.py @@ -13,7 +13,7 @@ ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) STOP = set("the a an of to in is are be and or for with that this it as on by at from not no if then each " "all any must shall can will when which its into than one two use used using value values " "object objects method methods class classes table tables field fields return returns".split()) -SPEC_LIMIT = 0.55 # cosine of word counts of the whole spec +SPEC_LIMIT = 0.75 # cosine of word counts of the whole spec (same-category specs share boilerplate: 0.63-0.67 seen) GOAL_LIMIT = 0.60 # cosine of the Goal and Business rules sections NAME_LIMIT = 0.60 # Jaccard of name tokens (placeholder and Z prefix removed) diff --git a/harness/trainset.py b/harness/trainset.py index 646a690..28f1aa9 100644 --- a/harness/trainset.py +++ b/harness/trainset.py @@ -104,8 +104,8 @@ def run(part, parts, target, stop_ledger): for idx, s in enumerate(plan["slots"]): if idx % parts != part: continue - if os.path.exists(os.path.join(POOL, s["id"], "generation.json")): - continue + if os.path.exists(os.path.join(POOL, "_logs", s["id"] + ".json")): + continue # done or failed: not tried again (delete the log to retry) if accepted_count() >= plan["target"]: print("TARGET reached", flush=True) break @@ -132,6 +132,11 @@ def run(part, parts, target, stop_ledger): except Exception as e: # noqa: BLE001 log = {"id": s["id"], "error": str(e)[:500]} log.update(error_kind=s.get("error_kind"), spent_total=spent()) + os.makedirs(os.path.join(POOL, "_logs"), exist_ok=True) + json.dump(log, open(os.path.join(POOL, "_logs", s["id"] + ".json"), "w"), indent=1) + stray = os.path.join(POOL, "generation.json") # generate() writes here when no task dir exists + if os.path.exists(stray): + os.remove(stray) print(json.dumps(log), flush=True) diff --git a/train/STATE.md b/train/STATE.md index cd083a7..cfe31ac 100644 --- a/train/STATE.md +++ b/train/STATE.md @@ -120,3 +120,23 @@ Training runs on HF Jobs with Unsloth, not on the Mac. No `mlx_lm` training. - Tool names stay as in the EPOD ABAP MCP server (generic_v0 not used). The ADT "save failed" response cannot be fixed on the server side; the proxy adds a syntax check instead. - HF cleanup done: test, overfit and full adapter repos deleted (step 200 and 400 checkpoints too). Kept: dataset `erhankeseli/abap-stage1-data`. Local adapter folders and `data_overfit` deleted. - `BUDGET_LIMIT_USD` = 161 (ledger 66.97 + 35 usage x 2.7), until 12 October. + +## Stage 2 data, Part B (2026-10-05) + +- Teacher budget: `BUDGET_LIMIT_USD` 161 (ledger). Until 11 October up to 35 usage (= 94.5 ledger) without asking. +- B1 generator training mode: `harness/trainset.py`, pool `tasks_gen/train` (ids G1000+, run numbers 32000+), plan in + `tasks_gen/train/plan.json` (223 slots: category shares as the eval plan, 30 error tasks: named-type 10, reserved-word 10, + long-names 10), overlap check against all eval tasks and earlier training tasks (`harness/overlap.py`: spec cosine 0.75, + Goal + Business rules cosine 0.60, contract name Jaccard 0.60; calibrated on the eval pairs; K variants are the only + eval pairs above 0.75). A too close bundle goes back to the model as a repair message. K variants are not made for + training (they need the generic_v0 schema, which is not used). Started 2026-10-05 12:00 with 3 workers, target 200 + accepted tasks, phase limit 27 ledger USD; logs `runs/gen_train/w*.log`, slot logs `tasks_gen/train/_logs/`. +- B2a proxy: a write that fails with only "An error occured during the save operation" gets `syntaxCheck` messages from local + abaplint parser (line + text). The EPOD syntax check cannot check the rejected source (tested 2026-10-05: it returned + 0 errors for the stored stub), so abaplint is used. The raw result stays in `trajectory.jsonl` (`raw_result`). +- B2b record: `harness/record.py` writes `record.json` (task metadata, exact messages, tool schemas, raw tool results, + metadata: score, cost, end reason, ...) and `reasoning.json` (teacher reasoning, not training data) per run. +- B2c converter: `train/to_qwen.py` (Qwen 3.8 chat template, `enable_thinking=False`, Qwen XML tool call format; tokenizer + round-trip check of every tool call; assistant spans for loss masking). B2d filter: `train/accept.py`. +- Test: scripted fake model on T01 (A4H, no cloud): score 100, 1 syntax hint, record, filter and converter OK. +- B3 trajectory runner: `harness/trajectories.py` (6 workers, DeepSeek V4.1 Flash, output `runs/traj/`). Starts after generation.