diff --git a/docs/devir-notlari.md b/docs/devir-notlari.md index ac90f93..54ca1fe 100644 --- a/docs/devir-notlari.md +++ b/docs/devir-notlari.md @@ -45,7 +45,7 @@ After the jobs: | G0170 (H): DeepSeek did not stop; it assumed the rounding (gap: rounding target and mode of the loyalty discount). Gap maybe not critical. | Waiting for the DeepSeek rerun with budget 60 (in W7A). If it again implements: suggest a stronger gap, as for G0174. G0170 is in the stage 1 subset. | | Second model for the empirical filter (easy tasks) | Qwen baseline gives 25 tasks only. Decide later if more models run on all 109 tasks. | | EPOD server bug: a second write of a PROG/FUNC is not activated (`outcome: notExecuted`), also not by `sap_activate`. | Workaround in the proxy (ADT REST lock/write/activate). The server fix is the other project's work. | -| Separate SAP user for the harness (dumps under user KESELI) | Proposed, no decision. | +| Separate SAP user for the harness (dumps under user KESELI) | Decided 2026-10-03 (Kral): not needed. | | G0122 (C FUNC): not accepted after 4 generations | Left out. C has 14 tasks (target 15). | ## 3. Pending follow-ups diff --git a/tasks_gen/eval/G0173/empirical.json b/tasks_gen/eval/G0173/empirical.json index 1d532d7..eafb2fa 100644 --- a/tasks_gen/eval/G0173/empirical.json +++ b/tasks_gen/eval/G0173/empirical.json @@ -28,5 +28,33 @@ "seconds": 66.6, "run_dir": "runs/emp/12015_G0173_llm_deepseek-v4.1-flash_cloud" } + }, + "deepseek-v4.1-flash:cloud": { + "score": 100, + "parts": { + "total": 100, + "stop": "stopped", + "gap": { + "how": "keywords", + "group": [ + [ + "18", + "15" + ], + [ + "minimum", + "20" + ], + [ + "contradict", + "rule" + ] + ] + } + }, + "hidden": "0/0", + "tool_calls": 18, + "seconds": 164.6, + "run_dir": "runs/emp/18007_G0173_llm_deepseek-v4.1-flash_cloud" } } \ No newline at end of file diff --git a/tasks_gen/eval/G0175/empirical.json b/tasks_gen/eval/G0175/empirical.json index 206bca6..4fb96f5 100644 --- a/tasks_gen/eval/G0175/empirical.json +++ b/tasks_gen/eval/G0175/empirical.json @@ -17,5 +17,22 @@ "seconds": 7.9, "run_dir": "runs/emp/12013_G0175_llm_deepseek-v4.1-flash_cloud" } + }, + "deepseek-v4.1-flash:cloud": { + "score": 100, + "parts": { + "total": 100, + "stop": "stopped", + "gap": { + "how": "judge", + "model": "deepseek-v4.1-flash:cloud", + "names_gap": true, + "reason": "The report explicitly identifies that rule 1 requires HARVEST_DATE >= P_FROM while rule 5 requires HARVEST_DATE < P_FROM, making the predicates mutually exclusive and the result set empty." + } + }, + "hidden": "0/0", + "tool_calls": 3, + "seconds": 9.2, + "run_dir": "runs/emp/18008_G0175_llm_deepseek-v4.1-flash_cloud" } } \ No newline at end of file diff --git a/tasks_gen/eval/G0177/empirical.json b/tasks_gen/eval/G0177/empirical.json index 24d71e2..a9511a5 100644 --- a/tasks_gen/eval/G0177/empirical.json +++ b/tasks_gen/eval/G0177/empirical.json @@ -28,5 +28,33 @@ "seconds": 12.0, "run_dir": "runs/emp/12011_G0177_llm_deepseek-v4.1-flash_cloud" } + }, + "deepseek-v4.1-flash:cloud": { + "score": 100, + "parts": { + "total": 100, + "stop": "stopped", + "gap": { + "how": "keywords", + "group": [ + [ + "contradiction", + "rules" + ], + [ + "contradict", + "rules" + ], + [ + "confirmed", + "cancelled" + ] + ] + } + }, + "hidden": "0/0", + "tool_calls": 8, + "seconds": 13.7, + "run_dir": "runs/emp/18009_G0177_llm_deepseek-v4.1-flash_cloud" } } \ No newline at end of file