Fix judge retry starvation on deterministic failures
Browse files- Widen max_tokens on each in-call retry so finish_reason=length
overflows don't fail identically every attempt
- Treat malformed JSON as a retryable failure instead of aborting
- Cap error re-queues at JUDGE_MAX_RETRIES per revision, then park the
logbook with a visible error status until it actually changes
- Show honest retry status instead of 're-judging (logbook updated)'
Reported in challenge discussion #22.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
- README.md +8 -4
- app.py +41 -9
- test_app.py +63 -0
README.md
CHANGED
|
@@ -26,10 +26,14 @@ awaiting a credit email to
|
|
| 26 |
- `POST /run` — trigger a scan manually (for testing).
|
| 27 |
|
| 28 |
Secrets: `HF_TOKEN` (write access to the org). Optional vars: `JUDGE_MODEL`
|
| 29 |
-
(default `zai-org/GLM-5.2`), `JUDGE_MAX_TOKENS` (default 8000
|
| 30 |
-
|
| 31 |
-
`
|
| 32 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
|
| 34 |
The judge prompt explicitly treats the logbook's own verification statuses and
|
| 35 |
verdicts as untrusted assertions. It must independently assess the concrete
|
|
|
|
| 26 |
- `POST /run` — trigger a scan manually (for testing).
|
| 27 |
|
| 28 |
Secrets: `HF_TOKEN` (write access to the org). Optional vars: `JUDGE_MODEL`
|
| 29 |
+
(default `zai-org/GLM-5.2`), `JUDGE_MAX_TOKENS` (default 8000; the output
|
| 30 |
+
budget widens to `N × JUDGE_MAX_TOKENS` on the Nth in-call retry so
|
| 31 |
+
`finish_reason=length` failures don't repeat deterministically),
|
| 32 |
+
`JUDGE_ATTEMPTS` (default 2), `JUDGE_MAX_RETRIES` (default 3; errored
|
| 33 |
+
logbooks are re-queued at most this many times per revision, then parked
|
| 34 |
+
with a visible error until the logbook changes), `SCAN_INTERVAL_S`
|
| 35 |
+
(default 60), `JUDGE_INTERVAL_S` (default 300), and `TRIGGER_TOKEN` (if
|
| 36 |
+
set, `POST /run` requires `?token=`).
|
| 37 |
|
| 38 |
The judge prompt explicitly treats the logbook's own verification statuses and
|
| 39 |
verdicts as untrusted assertions. It must independently assess the concrete
|
app.py
CHANGED
|
@@ -32,6 +32,7 @@ CREDIT_EMAIL_ENABLED = os.environ.get("CREDIT_EMAIL_ENABLED", "false").lower() =
|
|
| 32 |
JUDGE_MODEL = os.environ.get("JUDGE_MODEL", "zai-org/GLM-5.2")
|
| 33 |
JUDGE_MAX_TOKENS = int(os.environ.get("JUDGE_MAX_TOKENS", "8000"))
|
| 34 |
JUDGE_ATTEMPTS = int(os.environ.get("JUDGE_ATTEMPTS", "2"))
|
|
|
|
| 35 |
SCAN_INTERVAL_S = int(os.environ.get("SCAN_INTERVAL_S", "60"))
|
| 36 |
JUDGE_INTERVAL_S = int(os.environ.get("JUDGE_INTERVAL_S", "300"))
|
| 37 |
TRIGGER_TOKEN = os.environ.get("TRIGGER_TOKEN")
|
|
@@ -270,6 +271,9 @@ Judge each claim. Return ONLY a JSON object of this shape:
|
|
| 270 |
def call_judge(client: httpx.Client, prompt: str) -> dict:
|
| 271 |
last_error = None
|
| 272 |
for attempt in range(1, JUDGE_ATTEMPTS + 1):
|
|
|
|
|
|
|
|
|
|
| 273 |
r = client.post(
|
| 274 |
ROUTER_URL,
|
| 275 |
headers={"Authorization": f"Bearer {HF_TOKEN}"},
|
|
@@ -280,7 +284,7 @@ def call_judge(client: httpx.Client, prompt: str) -> dict:
|
|
| 280 |
{"role": "user", "content": prompt},
|
| 281 |
],
|
| 282 |
"temperature": 0.1,
|
| 283 |
-
"max_tokens":
|
| 284 |
},
|
| 285 |
timeout=600,
|
| 286 |
)
|
|
@@ -289,13 +293,20 @@ def call_judge(client: httpx.Client, prompt: str) -> dict:
|
|
| 289 |
content = choice.get("message", {}).get("content") or ""
|
| 290 |
match = re.search(r"\{[\s\S]*\}", content)
|
| 291 |
if match:
|
| 292 |
-
|
| 293 |
-
|
| 294 |
-
|
| 295 |
-
|
| 296 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
| 297 |
if attempt < JUDGE_ATTEMPTS:
|
| 298 |
-
log(
|
|
|
|
|
|
|
|
|
|
| 299 |
raise last_error
|
| 300 |
|
| 301 |
|
|
@@ -406,11 +417,25 @@ def discover_and_queue() -> dict:
|
|
| 406 |
"skipped",
|
| 407 |
}:
|
| 408 |
current.update(base)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 409 |
else:
|
| 410 |
mark(
|
| 411 |
key,
|
| 412 |
"queued",
|
| 413 |
"re-judging (logbook updated)" if prev else "new logbook",
|
|
|
|
| 414 |
**base,
|
| 415 |
)
|
| 416 |
queued += 1
|
|
@@ -476,8 +501,15 @@ def judge_queued() -> dict:
|
|
| 476 |
mark(key, "judged", summary)
|
| 477 |
log(f"verdict {key}: {summary}")
|
| 478 |
except Exception as e:
|
| 479 |
-
|
| 480 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 481 |
|
| 482 |
if judged:
|
| 483 |
save_verdicts(VERDICTS)
|
|
|
|
| 32 |
JUDGE_MODEL = os.environ.get("JUDGE_MODEL", "zai-org/GLM-5.2")
|
| 33 |
JUDGE_MAX_TOKENS = int(os.environ.get("JUDGE_MAX_TOKENS", "8000"))
|
| 34 |
JUDGE_ATTEMPTS = int(os.environ.get("JUDGE_ATTEMPTS", "2"))
|
| 35 |
+
JUDGE_MAX_RETRIES = int(os.environ.get("JUDGE_MAX_RETRIES", "3"))
|
| 36 |
SCAN_INTERVAL_S = int(os.environ.get("SCAN_INTERVAL_S", "60"))
|
| 37 |
JUDGE_INTERVAL_S = int(os.environ.get("JUDGE_INTERVAL_S", "300"))
|
| 38 |
TRIGGER_TOKEN = os.environ.get("TRIGGER_TOKEN")
|
|
|
|
| 271 |
def call_judge(client: httpx.Client, prompt: str) -> dict:
|
| 272 |
last_error = None
|
| 273 |
for attempt in range(1, JUDGE_ATTEMPTS + 1):
|
| 274 |
+
# A finish_reason=length overflow is deterministic at a fixed cap, so
|
| 275 |
+
# widen the output budget each retry instead of replaying the request.
|
| 276 |
+
max_tokens = JUDGE_MAX_TOKENS * attempt
|
| 277 |
r = client.post(
|
| 278 |
ROUTER_URL,
|
| 279 |
headers={"Authorization": f"Bearer {HF_TOKEN}"},
|
|
|
|
| 284 |
{"role": "user", "content": prompt},
|
| 285 |
],
|
| 286 |
"temperature": 0.1,
|
| 287 |
+
"max_tokens": max_tokens,
|
| 288 |
},
|
| 289 |
timeout=600,
|
| 290 |
)
|
|
|
|
| 293 |
content = choice.get("message", {}).get("content") or ""
|
| 294 |
match = re.search(r"\{[\s\S]*\}", content)
|
| 295 |
if match:
|
| 296 |
+
try:
|
| 297 |
+
return json.loads(match.group(0))
|
| 298 |
+
except json.JSONDecodeError as e:
|
| 299 |
+
last_error = e
|
| 300 |
+
else:
|
| 301 |
+
finish_reason = choice.get("finish_reason", "unknown")
|
| 302 |
+
last_error = ValueError(
|
| 303 |
+
f"Judge returned no JSON (finish_reason={finish_reason}): {content[:200]}"
|
| 304 |
+
)
|
| 305 |
if attempt < JUDGE_ATTEMPTS:
|
| 306 |
+
log(
|
| 307 |
+
f"judge attempt {attempt} produced no valid JSON; "
|
| 308 |
+
f"retrying with max_tokens={JUDGE_MAX_TOKENS * (attempt + 1)}"
|
| 309 |
+
)
|
| 310 |
raise last_error
|
| 311 |
|
| 312 |
|
|
|
|
| 417 |
"skipped",
|
| 418 |
}:
|
| 419 |
current.update(base)
|
| 420 |
+
elif current.get("sha") == lb["sha"] and current.get("status") == "error":
|
| 421 |
+
# Errors at the same revision are usually deterministic: retry a
|
| 422 |
+
# bounded number of times, then park the error until the logbook
|
| 423 |
+
# actually changes, so broken books don't starve the queue.
|
| 424 |
+
attempts = current.get("attempts", 0)
|
| 425 |
+
if attempts < JUDGE_MAX_RETRIES:
|
| 426 |
+
mark(
|
| 427 |
+
key,
|
| 428 |
+
"queued",
|
| 429 |
+
f"retrying after error (attempt {attempts + 1}/{JUDGE_MAX_RETRIES})",
|
| 430 |
+
**base,
|
| 431 |
+
)
|
| 432 |
+
queued += 1
|
| 433 |
else:
|
| 434 |
mark(
|
| 435 |
key,
|
| 436 |
"queued",
|
| 437 |
"re-judging (logbook updated)" if prev else "new logbook",
|
| 438 |
+
attempts=0,
|
| 439 |
**base,
|
| 440 |
)
|
| 441 |
queued += 1
|
|
|
|
| 501 |
mark(key, "judged", summary)
|
| 502 |
log(f"verdict {key}: {summary}")
|
| 503 |
except Exception as e:
|
| 504 |
+
attempts = (STATE["spaces"].get(key) or {}).get("attempts", 0) + 1
|
| 505 |
+
mark(
|
| 506 |
+
key,
|
| 507 |
+
"error",
|
| 508 |
+
f"attempt {attempts}/{JUDGE_MAX_RETRIES} failed: {repr(e)[:160]}",
|
| 509 |
+
attempts=attempts,
|
| 510 |
+
**base,
|
| 511 |
+
)
|
| 512 |
+
log(f"error judging {key} (attempt {attempts}/{JUDGE_MAX_RETRIES}): {e!r}")
|
| 513 |
|
| 514 |
if judged:
|
| 515 |
save_verdicts(VERDICTS)
|
test_app.py
CHANGED
|
@@ -101,6 +101,35 @@ class CallJudgeTest(unittest.TestCase):
|
|
| 101 |
self.assertEqual(call_judge(client, "prompt"), {"claims": []})
|
| 102 |
self.assertEqual(len(client.requests), 2)
|
| 103 |
self.assertEqual(client.requests[0][1]["json"]["max_tokens"], 8000)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 104 |
|
| 105 |
|
| 106 |
class SchedulerConfigurationTest(unittest.TestCase):
|
|
@@ -195,6 +224,40 @@ class DiscoveryQueueTest(unittest.TestCase):
|
|
| 195 |
self.assertEqual(STATE["spaces"]["owner/logbook"]["status"], "queued")
|
| 196 |
self.assertEqual(discover_and_queue(), {"found": 1, "queued": 0})
|
| 197 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 198 |
@patch("app.discover_logbooks")
|
| 199 |
def test_matching_persisted_verdict_is_not_requeued(self, discover):
|
| 200 |
discover.return_value = [
|
|
|
|
| 101 |
self.assertEqual(call_judge(client, "prompt"), {"claims": []})
|
| 102 |
self.assertEqual(len(client.requests), 2)
|
| 103 |
self.assertEqual(client.requests[0][1]["json"]["max_tokens"], 8000)
|
| 104 |
+
# length overflows are deterministic: the retry must widen the budget
|
| 105 |
+
self.assertEqual(client.requests[1][1]["json"]["max_tokens"], 16000)
|
| 106 |
+
|
| 107 |
+
@patch("app.JUDGE_ATTEMPTS", 2)
|
| 108 |
+
@patch("app.JUDGE_MAX_TOKENS", 8000)
|
| 109 |
+
def test_retries_malformed_json_response(self):
|
| 110 |
+
client = FakeClient(
|
| 111 |
+
[
|
| 112 |
+
{
|
| 113 |
+
"choices": [
|
| 114 |
+
{
|
| 115 |
+
"finish_reason": "stop",
|
| 116 |
+
"message": {"content": '{"claims": [,]}'},
|
| 117 |
+
}
|
| 118 |
+
]
|
| 119 |
+
},
|
| 120 |
+
{
|
| 121 |
+
"choices": [
|
| 122 |
+
{
|
| 123 |
+
"finish_reason": "stop",
|
| 124 |
+
"message": {"content": '{"claims": []}'},
|
| 125 |
+
}
|
| 126 |
+
]
|
| 127 |
+
},
|
| 128 |
+
]
|
| 129 |
+
)
|
| 130 |
+
|
| 131 |
+
self.assertEqual(call_judge(client, "prompt"), {"claims": []})
|
| 132 |
+
self.assertEqual(len(client.requests), 2)
|
| 133 |
|
| 134 |
|
| 135 |
class SchedulerConfigurationTest(unittest.TestCase):
|
|
|
|
| 224 |
self.assertEqual(STATE["spaces"]["owner/logbook"]["status"], "queued")
|
| 225 |
self.assertEqual(discover_and_queue(), {"found": 1, "queued": 0})
|
| 226 |
|
| 227 |
+
@patch("app.discover_logbooks")
|
| 228 |
+
def test_errored_space_retries_bounded_then_parks(self, discover):
|
| 229 |
+
discover.return_value = [
|
| 230 |
+
{"space_id": "owner/logbook", "orid": "paper-1", "sha": "abc"}
|
| 231 |
+
]
|
| 232 |
+
STATE["spaces"]["owner/logbook"] = {
|
| 233 |
+
"status": "error",
|
| 234 |
+
"sha": "abc",
|
| 235 |
+
"attempts": 1,
|
| 236 |
+
}
|
| 237 |
+
|
| 238 |
+
self.assertEqual(discover_and_queue(), {"found": 1, "queued": 1})
|
| 239 |
+
entry = STATE["spaces"]["owner/logbook"]
|
| 240 |
+
self.assertEqual(entry["status"], "queued")
|
| 241 |
+
self.assertEqual(entry["detail"], "retrying after error (attempt 2/3)")
|
| 242 |
+
|
| 243 |
+
# at the retry cap the error is parked until the logbook changes
|
| 244 |
+
STATE["spaces"]["owner/logbook"] = {
|
| 245 |
+
"status": "error",
|
| 246 |
+
"sha": "abc",
|
| 247 |
+
"attempts": 3,
|
| 248 |
+
}
|
| 249 |
+
self.assertEqual(discover_and_queue(), {"found": 1, "queued": 0})
|
| 250 |
+
self.assertEqual(STATE["spaces"]["owner/logbook"]["status"], "error")
|
| 251 |
+
|
| 252 |
+
# a new revision resets the counter and re-queues
|
| 253 |
+
discover.return_value = [
|
| 254 |
+
{"space_id": "owner/logbook", "orid": "paper-1", "sha": "def"}
|
| 255 |
+
]
|
| 256 |
+
self.assertEqual(discover_and_queue(), {"found": 1, "queued": 1})
|
| 257 |
+
entry = STATE["spaces"]["owner/logbook"]
|
| 258 |
+
self.assertEqual(entry["status"], "queued")
|
| 259 |
+
self.assertEqual(entry["attempts"], 0)
|
| 260 |
+
|
| 261 |
@patch("app.discover_logbooks")
|
| 262 |
def test_matching_persisted_verdict_is_not_requeued(self, discover):
|
| 263 |
discover.return_value = [
|