From 4002d889d50fa5216d5c764d857355e545ac9114 Mon Sep 17 00:00:00 2001 From: Kral Date: Tue, 6 Oct 2026 06:16:29 +0200 Subject: [PATCH] own-test report and chain script; work list --- scripts_probe/after_owntests.sh | 22 +++++++++++++++++ train/own_test_report.py | 42 +++++++++++++++++++++++++++++++++ 2 files changed, 64 insertions(+) create mode 100755 scripts_probe/after_owntests.sh create mode 100644 train/own_test_report.py diff --git a/scripts_probe/after_owntests.sh b/scripts_probe/after_owntests.sh new file mode 100755 index 0000000..df6fab5 --- /dev/null +++ b/scripts_probe/after_owntests.sh @@ -0,0 +1,22 @@ +#!/bin/sh +# Waits for harness.owntests to end, then: report, rebuild the stage 2 set with the own-test score as metadata, update the HF dataset, commit. +cd "$HOME/projects/abap-llm/harness" || exit 1 +PID="$1" +while kill -0 "$PID" 2>/dev/null; do sleep 60; done +python3 train/own_test_report.py > runs/own_test_report.log 2>&1 +train/.venv/bin/python train/build_stage2.py --out runs/stage2_data --hook hooks_example:own_test_weight > runs/build_after_owntests.log 2>&1 +python3 train/build_doc.py >> runs/build_after_owntests.log 2>&1 +set -a; . ./.env; set +a +train/.venv/bin/python - >> runs/build_after_owntests.log 2>&1 <<'PY' +import os +from huggingface_hub import HfApi +api = HfApi(token=os.environ["HF_TOKEN"]) +for f in ("stage2_train.jsonl", "stage2_valid.jsonl", "stage2_reserve.jsonl", "build_report.json", "README.md"): + api.upload_file(path_or_fileobj="runs/stage2_data/" + f, path_in_repo=f, repo_id="erhankeseli/abap-stage2-data", repo_type="dataset") +print("uploaded") +PY +export GIT_AUTHOR_NAME=Kral GIT_AUTHOR_EMAIL=kral@local GIT_COMMITTER_NAME=Kral GIT_COMMITTER_EMAIL=kral@local +git add docs train && git commit -q -m "D: own-test mutation scores of the accepted trajectories (metadata only), stage 2 set rebuilt with the score + +Co-Authored-By: Claude Sonnet 5.5 " +echo "after_owntests done $(date)" >> runs/own_test_report.log diff --git a/train/own_test_report.py b/train/own_test_report.py new file mode 100644 index 0000000..1e20595 --- /dev/null +++ b/train/own_test_report.py @@ -0,0 +1,42 @@ +"""Summary of the own-test mutation scores (item D): docs/own-test-mutation.md. Run after `python3 -m harness.owntests`.""" +import glob +import json +import os +import statistics +from collections import Counter, defaultdict + +ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +res = [json.load(open(p)) for p in glob.glob(os.path.join(ROOT, "runs", "traj", "*", "own_test_mutation.json"))] +by = Counter(r.get("status") for r in res) +scored = [r for r in res if r.get("status") == "scored"] +ref_ok = [r for r in scored if r.get("tests_pass_on_reference")] +per_kind = defaultdict(list) +for r in ref_ok: + if r.get("score") is not None: + per_kind[r["kind"]].append(r["score"]) +sc = [r["score"] for r in ref_ok if r.get("score") is not None] +bins = Counter("1.0" if s == 1 else ">=0.75" if s >= 0.75 else ">=0.5" if s >= 0.5 else "<0.5" for s in sc) +low = sorted([r for r in ref_ok if r.get("score") is not None and r["score"] < 0.75], key=lambda r: r["score"]) +doc = f"""# Own-test mutation score of the accepted trajectories (item D, metadata only) + +Module `harness/owntests.py`, results `runs/traj//own_test_mutation.json` (not in git), report by `train/own_test_report.py`. +The model's own unit tests (testclasses include or global test classes) are run against the faulty references of the task (`faulty/`, mutants that the hidden tests kill). +The correct reference must pass the model's tests first (otherwise the tests encode model specific behavior). score = killed / (killed + survived). +**Metadata only: the acceptance filter is not changed.** The builder takes the score through `--hook hooks_example:own_test_weight` (field `own_test_mutation`). + +## Result +{len(res)} accepted trajectories looked at: {dict(by)}. +- Scored: {len(scored)}; the model's tests also passed on the correct reference: {len(ref_ok)} (the others are not reliable: a test that fails on the correct solution kills every mutant). +- Mean score over the reliable ones: {statistics.mean(sc):.2f}; median {statistics.median(sc):.2f}; distribution {dict(bins)} (n = {len(sc)}). +- By kind (mean, n): {{{", ".join(f"{k}: {statistics.mean(v):.2f} ({len(v)})" for k, v in sorted(per_kind.items()))}}} +- `no_own_tests`: {by.get('no_own_tests', 0)} trajectories were accepted without any own test (at most 85 points); `no_mutants`: {by.get('no_mutants', 0)}; `not_supported` (PROG: tests are inside the program): {by.get('not_supported', 0)}. + +## Weakest (score under 0.75, tests pass on the reference) +""" + "\n".join(f"- {r['task']} run {r['run']} {r['kind']}: score {r['score']} ({r['killed']} of {r['valid']} valid mutants killed)" for r in low[:15]) + """ + +## Use +Not used for filtering yet. Candidates for a later rule (Kral + Opus decide): drop or down-weight trajectories with a reliable score under 0.5 or with `tests_pass_on_reference` false; +prefer trajectories without own tests last. The hook example shows where the weight goes (`train/hooks_example.py`). +""" +open(os.path.join(ROOT, "docs", "own-test-mutation.md"), "w").write(doc) +print(doc[:1500])