own-test report and chain script; work list
This commit is contained in:
22
scripts_probe/after_owntests.sh
Executable file
22
scripts_probe/after_owntests.sh
Executable file
@@ -0,0 +1,22 @@
|
||||
#!/bin/sh
|
||||
# Waits for harness.owntests to end, then: report, rebuild the stage 2 set with the own-test score as metadata, update the HF dataset, commit.
|
||||
cd "$HOME/projects/abap-llm/harness" || exit 1
|
||||
PID="$1"
|
||||
while kill -0 "$PID" 2>/dev/null; do sleep 60; done
|
||||
python3 train/own_test_report.py > runs/own_test_report.log 2>&1
|
||||
train/.venv/bin/python train/build_stage2.py --out runs/stage2_data --hook hooks_example:own_test_weight > runs/build_after_owntests.log 2>&1
|
||||
python3 train/build_doc.py >> runs/build_after_owntests.log 2>&1
|
||||
set -a; . ./.env; set +a
|
||||
train/.venv/bin/python - >> runs/build_after_owntests.log 2>&1 <<'PY'
|
||||
import os
|
||||
from huggingface_hub import HfApi
|
||||
api = HfApi(token=os.environ["HF_TOKEN"])
|
||||
for f in ("stage2_train.jsonl", "stage2_valid.jsonl", "stage2_reserve.jsonl", "build_report.json", "README.md"):
|
||||
api.upload_file(path_or_fileobj="runs/stage2_data/" + f, path_in_repo=f, repo_id="erhankeseli/abap-stage2-data", repo_type="dataset")
|
||||
print("uploaded")
|
||||
PY
|
||||
export GIT_AUTHOR_NAME=Kral GIT_AUTHOR_EMAIL=kral@local GIT_COMMITTER_NAME=Kral GIT_COMMITTER_EMAIL=kral@local
|
||||
git add docs train && git commit -q -m "D: own-test mutation scores of the accepted trajectories (metadata only), stage 2 set rebuilt with the score
|
||||
|
||||
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>"
|
||||
echo "after_owntests done $(date)" >> runs/own_test_report.log
|
||||
42
train/own_test_report.py
Normal file
42
train/own_test_report.py
Normal file
@@ -0,0 +1,42 @@
|
||||
"""Summary of the own-test mutation scores (item D): docs/own-test-mutation.md. Run after `python3 -m harness.owntests`."""
|
||||
import glob
|
||||
import json
|
||||
import os
|
||||
import statistics
|
||||
from collections import Counter, defaultdict
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
res = [json.load(open(p)) for p in glob.glob(os.path.join(ROOT, "runs", "traj", "*", "own_test_mutation.json"))]
|
||||
by = Counter(r.get("status") for r in res)
|
||||
scored = [r for r in res if r.get("status") == "scored"]
|
||||
ref_ok = [r for r in scored if r.get("tests_pass_on_reference")]
|
||||
per_kind = defaultdict(list)
|
||||
for r in ref_ok:
|
||||
if r.get("score") is not None:
|
||||
per_kind[r["kind"]].append(r["score"])
|
||||
sc = [r["score"] for r in ref_ok if r.get("score") is not None]
|
||||
bins = Counter("1.0" if s == 1 else ">=0.75" if s >= 0.75 else ">=0.5" if s >= 0.5 else "<0.5" for s in sc)
|
||||
low = sorted([r for r in ref_ok if r.get("score") is not None and r["score"] < 0.75], key=lambda r: r["score"])
|
||||
doc = f"""# Own-test mutation score of the accepted trajectories (item D, metadata only)
|
||||
|
||||
Module `harness/owntests.py`, results `runs/traj/<run>/own_test_mutation.json` (not in git), report by `train/own_test_report.py`.
|
||||
The model's own unit tests (testclasses include or global test classes) are run against the faulty references of the task (`faulty/`, mutants that the hidden tests kill).
|
||||
The correct reference must pass the model's tests first (otherwise the tests encode model specific behavior). score = killed / (killed + survived).
|
||||
**Metadata only: the acceptance filter is not changed.** The builder takes the score through `--hook hooks_example:own_test_weight` (field `own_test_mutation`).
|
||||
|
||||
## Result
|
||||
{len(res)} accepted trajectories looked at: {dict(by)}.
|
||||
- Scored: {len(scored)}; the model's tests also passed on the correct reference: {len(ref_ok)} (the others are not reliable: a test that fails on the correct solution kills every mutant).
|
||||
- Mean score over the reliable ones: {statistics.mean(sc):.2f}; median {statistics.median(sc):.2f}; distribution {dict(bins)} (n = {len(sc)}).
|
||||
- By kind (mean, n): {{{", ".join(f"{k}: {statistics.mean(v):.2f} ({len(v)})" for k, v in sorted(per_kind.items()))}}}
|
||||
- `no_own_tests`: {by.get('no_own_tests', 0)} trajectories were accepted without any own test (at most 85 points); `no_mutants`: {by.get('no_mutants', 0)}; `not_supported` (PROG: tests are inside the program): {by.get('not_supported', 0)}.
|
||||
|
||||
## Weakest (score under 0.75, tests pass on the reference)
|
||||
""" + "\n".join(f"- {r['task']} run {r['run']} {r['kind']}: score {r['score']} ({r['killed']} of {r['valid']} valid mutants killed)" for r in low[:15]) + """
|
||||
|
||||
## Use
|
||||
Not used for filtering yet. Candidates for a later rule (Kral + Opus decide): drop or down-weight trajectories with a reliable score under 0.5 or with `tests_pass_on_reference` false;
|
||||
prefer trajectories without own tests last. The hook example shows where the weight goes (`train/hooks_example.py`).
|
||||
"""
|
||||
open(os.path.join(ROOT, "docs", "own-test-mutation.md"), "w").write(doc)
|
||||
print(doc[:1500])
|
||||
Reference in New Issue
Block a user