diff --git a/train/STATE.md b/train/STATE.md index 2ed9c24..0d4a9b8 100644 --- a/train/STATE.md +++ b/train/STATE.md @@ -80,3 +80,16 @@ Training runs on HF Jobs with Unsloth, not on the Mac. No `mlx_lm` training. Stronger check for later: log the Unsloth step-0 loss, or compare the Mac loss after a longer run (the first 200 steps). - Full run estimate: 748 steps x 30.5 s = about 6.3 h = about 16 USD (range 12-21 USD, step time varies with document length). Not started. Needs Kral's go and the alpha decision. + +## Overfit test and full run (2026-10-04, Opus 5.5 decision: alpha 32, cost limit 25 USD) + +- Overfit test (job 6ac28c41404719ba3764f7fc): one train document (ZCL_DEMO_ABAP_STRUCTURES, 2503 tokens, train row 18), + alpha 32, lr 1e-3, 30 steps, no warmup, 5.8 s/step, about 0.4 USD. Unsloth loss on that document: 0.798 -> 0.000234 (-99.97 %). +- Mac with the converted adapter (scale 2.0): 0.811 (no adapter) -> 0.012 (-98.5 %). Conversion confirmed (a broken + conversion would stay near 0.8). The remaining gap is the base mismatch (bnb nf4 vs MLX affine 4-bit). +- Full run started: job 6ac28e19fbc85ba6823a0eef, a100-large, 748 steps, rank 16, alpha 32, lr 5e-5 cosine, warmup 30, + timeout 8 h (max 20 USD). Valid loss logged at step 0 (log line `EVAL`), checkpoints and valid loss every 200 steps in + `erhankeseli/abap-stage1-adapter-full/step/` (with `eval.json`). Expected about 6.3 h. +- Rule: at step 200 convert the checkpoint (`train/peft_to_mlx.py`), measure the valid loss on the Mac + (`runs/stage1/adapter_test_valid_loss.log` command, about 20 min), compare the relative drop with the GPU value. + Stop the job (`hf jobs cancel`) if they differ. diff --git a/train/hf_train.py b/train/hf_train.py index c4df384..e66fcac 100644 --- a/train/hf_train.py +++ b/train/hf_train.py @@ -19,6 +19,10 @@ ap.add_argument("--epochs", type=float, default=2.0) ap.add_argument("--rank", type=int, default=16) ap.add_argument("--alpha", type=int, default=16) ap.add_argument("--lr", type=float, default=5e-5) +ap.add_argument("--train-file", default="train.jsonl") +ap.add_argument("--eval-file", default="valid.jsonl") +ap.add_argument("--warmup-steps", type=int, default=30) +ap.add_argument("--save-every", type=int, default=0, help="upload the adapter to /step every N steps") ap.add_argument("--max-seq-length", type=int, default=16384) a = ap.parse_args() @@ -35,26 +39,45 @@ model = FastLanguageModel.get_peft_model( "in_proj_qkv", "in_proj_z", "out_proj"], use_gradient_checkpointing="unsloth", random_state=20261003) -ds = load_dataset(a.data, data_files={"train": "train.jsonl", "valid": "valid.jsonl"}, token=tok_hf) +ds = load_dataset(a.data, data_files={"train": a.train_file, "valid": a.eval_file}, token=tok_hf) cfg = SFTConfig( output_dir="out", per_device_train_batch_size=1, per_device_eval_batch_size=1, gradient_accumulation_steps=1, num_train_epochs=a.epochs, max_steps=a.max_steps, - learning_rate=a.lr, lr_scheduler_type="cosine", warmup_steps=min(30, max(1, a.max_steps // 3)) if a.max_steps > 0 else 30, + learning_rate=a.lr, lr_scheduler_type="cosine", warmup_steps=a.warmup_steps, optim="adamw_8bit", weight_decay=0.0, logging_steps=1, eval_strategy="no", save_strategy="no", max_length=a.max_seq_length, dataset_text_field="text", packing=False, seed=20261003, report_to="none") tr = SFTTrainer(model=model, processing_class=tok, train_dataset=ds["train"], eval_dataset=ds["valid"], args=cfg) +from transformers import TrainerCallback # noqa: E402 +from huggingface_hub import HfApi # noqa: E402 +api = HfApi(token=tok_hf) + + +class SaveCb(TrainerCallback): + def on_step_end(self, args, state, control, **kw): + n = state.global_step + if a.save_every and n % a.save_every == 0: + model.save_pretrained(f"ckpt{n}") + api.upload_folder(folder_path=f"ckpt{n}", repo_id=a.out, repo_type="model", path_in_repo=f"step{n}") + ev = tr.evaluate() + print("EVAL", json.dumps({"step": n, "eval_loss": ev.get("eval_loss")}), flush=True) + api.upload_file(path_or_fileobj=json.dumps({"step": n, "eval_loss": ev.get("eval_loss")}).encode(), + path_in_repo=f"step{n}/eval.json", repo_id=a.out, repo_type="model") + + +tr.add_callback(SaveCb()) +ev0 = tr.evaluate() +print("EVAL", json.dumps({"step": 0, "eval_loss": ev0.get("eval_loss")}), flush=True) t0 = time.time() tr.train() train_s = time.time() - t0 steps = tr.state.global_step ev = tr.evaluate() # valid loss of the adapter, for the conversion check -info = {"steps": steps, "train_seconds": round(train_s, 1), "sec_per_step": round(train_s / max(steps, 1), 2), +info = {"eval_loss_step0": ev0.get("eval_loss"), "steps": steps, "train_seconds": round(train_s, 1), "sec_per_step": round(train_s / max(steps, 1), 2), "eval_loss": ev.get("eval_loss"), "rank": a.rank, "alpha": a.alpha, "lr": a.lr, "peak_gpu_gb": round(__import__("torch").cuda.max_memory_allocated() / 2**30, 1)} print("RESULT", json.dumps(info)) model.save_pretrained("adapter") json.dump(info, open("adapter/job_result.json", "w")) -from huggingface_hub import HfApi # noqa: E402 -HfApi(token=tok_hf).upload_folder(folder_path="adapter", repo_id=a.out, repo_type="model") +api.upload_folder(folder_path="adapter", repo_id=a.out, repo_type="model") print("PUSHED", a.out)