B: stage 2 data analysis (repair taxonomy, teacher vs Qwen behaviors, duplicates, empty_response); empty_response fix (cap 24000, retry temperature, stream guard)
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -28,7 +28,20 @@ POOL = os.path.join(ROOT, "tasks_gen", "train")
|
||||
OUT = os.path.join(ROOT, "runs", "traj")
|
||||
RUN_BASE = 200000 # 200000 + (task number - 1000) * 3 + attempt (a digit must lead the 4-char base36 run: < 466560)
|
||||
MODEL = "deepseek-v4.1-flash:cloud"
|
||||
CDS_CALLS = 100 # tool-call budget for tasks with a CDS contract object (eval keeps 60; Kral 2026-10-05)
|
||||
# empty_response fix (2026-10-06, docs/empty-response.md): 13 of 119 runs ended with one turn that used the whole output limit on
|
||||
# reasoning. Cap per turn 24000 (only 3 of 1997 turns were legitimately longer), one retry at another temperature (the same
|
||||
# sample ran away again in 7 of 13 runs), and the stream guard (a streamed turn with only reasoning is cut after that many reasoning
|
||||
# tokens) which stays off until it is verified on the cloud model (env STREAM_GUARD, for example 9000).
|
||||
TEACHER_MAX_TOKENS = 24000
|
||||
EMPTY_RETRIES = 1
|
||||
RETRY_TEMPERATURE = 0.8
|
||||
STREAM_GUARD = int(os.environ["STREAM_GUARD"]) if os.environ.get("STREAM_GUARD") else None
|
||||
CDS_CALLS = 100
|
||||
|
||||
|
||||
def new_agent():
|
||||
return LlmAgent(MODEL, loop_guard=3, max_tokens=TEACHER_MAX_TOKENS, empty_retries=EMPTY_RETRIES,
|
||||
retry_temperature=RETRY_TEMPERATURE, stream_guard=STREAM_GUARD) # tool-call budget for tasks with a CDS contract object (eval keeps 60; Kral 2026-10-05)
|
||||
LOCK = threading.Lock()
|
||||
|
||||
|
||||
@@ -74,7 +87,7 @@ def one(task_id, attempt, stop):
|
||||
print("BUDGET", e, flush=True)
|
||||
return
|
||||
run_no = RUN_BASE + (int("".join(c for c in task_id if c.isdigit())) - 1000) * 3 + attempt
|
||||
agent = LlmAgent(MODEL, loop_guard=3)
|
||||
agent = new_agent()
|
||||
runner = Runner(POOL, OUT)
|
||||
t0 = time.time()
|
||||
row = {"task": task_id, "attempt": attempt, "run": run_no, "model": MODEL}
|
||||
@@ -97,7 +110,7 @@ def one(task_id, attempt, stop):
|
||||
for d in glob.glob(os.path.join(OUT, f"{run_no}_{task_id}_*")):
|
||||
os.makedirs(os.path.join(OUT, "_aborted"), exist_ok=True)
|
||||
os.rename(d, os.path.join(OUT, "_aborted", os.path.basename(d) + "_" + str(int(time.time()))))
|
||||
agent = LlmAgent(MODEL, loop_guard=3)
|
||||
agent = new_agent()
|
||||
score = (rep.get("score") or {}).get("total")
|
||||
rec = os.path.exists(os.path.join(run_dir, "record.json"))
|
||||
row.update(score=score, setup_failed=bool(rep.get("setup_failed")), end_reason=rep.get("end_reason"),
|
||||
|
||||
Reference in New Issue
Block a user