B: stage 2 data analysis (repair taxonomy, teacher vs Qwen behaviors, duplicates, empty_response); empty_response fix (cap 24000, retry temperature, stream guard)

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Kral
2026-10-06 05:56:23 +02:00
parent 1ddb9a7c65
commit 542fc8ec31
6 changed files with 406 additions and 9 deletions

View File

@@ -28,7 +28,20 @@ POOL = os.path.join(ROOT, "tasks_gen", "train")
OUT = os.path.join(ROOT, "runs", "traj")
RUN_BASE = 200000 # 200000 + (task number - 1000) * 3 + attempt (a digit must lead the 4-char base36 run: < 466560)
MODEL = "deepseek-v4.1-flash:cloud"
CDS_CALLS = 100 # tool-call budget for tasks with a CDS contract object (eval keeps 60; Kral 2026-10-05)
# empty_response fix (2026-10-06, docs/empty-response.md): 13 of 119 runs ended with one turn that used the whole output limit on
# reasoning. Cap per turn 24000 (only 3 of 1997 turns were legitimately longer), one retry at another temperature (the same
# sample ran away again in 7 of 13 runs), and the stream guard (a streamed turn with only reasoning is cut after that many reasoning
# tokens) which stays off until it is verified on the cloud model (env STREAM_GUARD, for example 9000).
TEACHER_MAX_TOKENS = 24000
EMPTY_RETRIES = 1
RETRY_TEMPERATURE = 0.8
STREAM_GUARD = int(os.environ["STREAM_GUARD"]) if os.environ.get("STREAM_GUARD") else None
CDS_CALLS = 100
def new_agent():
return LlmAgent(MODEL, loop_guard=3, max_tokens=TEACHER_MAX_TOKENS, empty_retries=EMPTY_RETRIES,
retry_temperature=RETRY_TEMPERATURE, stream_guard=STREAM_GUARD) # tool-call budget for tasks with a CDS contract object (eval keeps 60; Kral 2026-10-05)
LOCK = threading.Lock()
@@ -74,7 +87,7 @@ def one(task_id, attempt, stop):
print("BUDGET", e, flush=True)
return
run_no = RUN_BASE + (int("".join(c for c in task_id if c.isdigit())) - 1000) * 3 + attempt
agent = LlmAgent(MODEL, loop_guard=3)
agent = new_agent()
runner = Runner(POOL, OUT)
t0 = time.time()
row = {"task": task_id, "attempt": attempt, "run": run_no, "model": MODEL}
@@ -97,7 +110,7 @@ def one(task_id, attempt, stop):
for d in glob.glob(os.path.join(OUT, f"{run_no}_{task_id}_*")):
os.makedirs(os.path.join(OUT, "_aborted"), exist_ok=True)
os.rename(d, os.path.join(OUT, "_aborted", os.path.basename(d) + "_" + str(int(time.time()))))
agent = LlmAgent(MODEL, loop_guard=3)
agent = new_agent()
score = (rep.get("score") or {}).get("total")
rec = os.path.exists(os.path.join(run_dir, "record.json"))
row.update(score=score, setup_failed=bool(rep.get("setup_failed")), end_reason=rep.get("end_reason"),