abidlabs HF Staff Claude Fable 5 commited on
Commit
3639761
·
1 Parent(s): 377e852

Fix judge retry starvation on deterministic failures

Browse files

- Widen max_tokens on each in-call retry so finish_reason=length
overflows don't fail identically every attempt
- Treat malformed JSON as a retryable failure instead of aborting
- Cap error re-queues at JUDGE_MAX_RETRIES per revision, then park the
logbook with a visible error status until it actually changes
- Show honest retry status instead of 're-judging (logbook updated)'

Reported in challenge discussion #22.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

Files changed (3) hide show
  1. README.md +8 -4
  2. app.py +41 -9
  3. test_app.py +63 -0
README.md CHANGED
@@ -26,10 +26,14 @@ awaiting a credit email to
26
  - `POST /run` — trigger a scan manually (for testing).
27
 
28
  Secrets: `HF_TOKEN` (write access to the org). Optional vars: `JUDGE_MODEL`
29
- (default `zai-org/GLM-5.2`), `JUDGE_MAX_TOKENS` (default 8000),
30
- `JUDGE_ATTEMPTS` (default 2), `SCAN_INTERVAL_S` (default 60),
31
- `JUDGE_INTERVAL_S` (default 300), and `TRIGGER_TOKEN` (if set, `POST /run`
32
- requires `?token=`).
 
 
 
 
33
 
34
  The judge prompt explicitly treats the logbook's own verification statuses and
35
  verdicts as untrusted assertions. It must independently assess the concrete
 
26
  - `POST /run` — trigger a scan manually (for testing).
27
 
28
  Secrets: `HF_TOKEN` (write access to the org). Optional vars: `JUDGE_MODEL`
29
+ (default `zai-org/GLM-5.2`), `JUDGE_MAX_TOKENS` (default 8000; the output
30
+ budget widens to `N × JUDGE_MAX_TOKENS` on the Nth in-call retry so
31
+ `finish_reason=length` failures don't repeat deterministically),
32
+ `JUDGE_ATTEMPTS` (default 2), `JUDGE_MAX_RETRIES` (default 3; errored
33
+ logbooks are re-queued at most this many times per revision, then parked
34
+ with a visible error until the logbook changes), `SCAN_INTERVAL_S`
35
+ (default 60), `JUDGE_INTERVAL_S` (default 300), and `TRIGGER_TOKEN` (if
36
+ set, `POST /run` requires `?token=`).
37
 
38
  The judge prompt explicitly treats the logbook's own verification statuses and
39
  verdicts as untrusted assertions. It must independently assess the concrete
app.py CHANGED
@@ -32,6 +32,7 @@ CREDIT_EMAIL_ENABLED = os.environ.get("CREDIT_EMAIL_ENABLED", "false").lower() =
32
  JUDGE_MODEL = os.environ.get("JUDGE_MODEL", "zai-org/GLM-5.2")
33
  JUDGE_MAX_TOKENS = int(os.environ.get("JUDGE_MAX_TOKENS", "8000"))
34
  JUDGE_ATTEMPTS = int(os.environ.get("JUDGE_ATTEMPTS", "2"))
 
35
  SCAN_INTERVAL_S = int(os.environ.get("SCAN_INTERVAL_S", "60"))
36
  JUDGE_INTERVAL_S = int(os.environ.get("JUDGE_INTERVAL_S", "300"))
37
  TRIGGER_TOKEN = os.environ.get("TRIGGER_TOKEN")
@@ -270,6 +271,9 @@ Judge each claim. Return ONLY a JSON object of this shape:
270
  def call_judge(client: httpx.Client, prompt: str) -> dict:
271
  last_error = None
272
  for attempt in range(1, JUDGE_ATTEMPTS + 1):
 
 
 
273
  r = client.post(
274
  ROUTER_URL,
275
  headers={"Authorization": f"Bearer {HF_TOKEN}"},
@@ -280,7 +284,7 @@ def call_judge(client: httpx.Client, prompt: str) -> dict:
280
  {"role": "user", "content": prompt},
281
  ],
282
  "temperature": 0.1,
283
- "max_tokens": JUDGE_MAX_TOKENS,
284
  },
285
  timeout=600,
286
  )
@@ -289,13 +293,20 @@ def call_judge(client: httpx.Client, prompt: str) -> dict:
289
  content = choice.get("message", {}).get("content") or ""
290
  match = re.search(r"\{[\s\S]*\}", content)
291
  if match:
292
- return json.loads(match.group(0))
293
- finish_reason = choice.get("finish_reason", "unknown")
294
- last_error = ValueError(
295
- f"Judge returned no JSON (finish_reason={finish_reason}): {content[:200]}"
296
- )
 
 
 
 
297
  if attempt < JUDGE_ATTEMPTS:
298
- log(f"judge attempt {attempt} returned no JSON; retrying")
 
 
 
299
  raise last_error
300
 
301
 
@@ -406,11 +417,25 @@ def discover_and_queue() -> dict:
406
  "skipped",
407
  }:
408
  current.update(base)
 
 
 
 
 
 
 
 
 
 
 
 
 
409
  else:
410
  mark(
411
  key,
412
  "queued",
413
  "re-judging (logbook updated)" if prev else "new logbook",
 
414
  **base,
415
  )
416
  queued += 1
@@ -476,8 +501,15 @@ def judge_queued() -> dict:
476
  mark(key, "judged", summary)
477
  log(f"verdict {key}: {summary}")
478
  except Exception as e:
479
- mark(key, "error", repr(e)[:200])
480
- log(f"error judging {key}: {e!r}")
 
 
 
 
 
 
 
481
 
482
  if judged:
483
  save_verdicts(VERDICTS)
 
32
  JUDGE_MODEL = os.environ.get("JUDGE_MODEL", "zai-org/GLM-5.2")
33
  JUDGE_MAX_TOKENS = int(os.environ.get("JUDGE_MAX_TOKENS", "8000"))
34
  JUDGE_ATTEMPTS = int(os.environ.get("JUDGE_ATTEMPTS", "2"))
35
+ JUDGE_MAX_RETRIES = int(os.environ.get("JUDGE_MAX_RETRIES", "3"))
36
  SCAN_INTERVAL_S = int(os.environ.get("SCAN_INTERVAL_S", "60"))
37
  JUDGE_INTERVAL_S = int(os.environ.get("JUDGE_INTERVAL_S", "300"))
38
  TRIGGER_TOKEN = os.environ.get("TRIGGER_TOKEN")
 
271
  def call_judge(client: httpx.Client, prompt: str) -> dict:
272
  last_error = None
273
  for attempt in range(1, JUDGE_ATTEMPTS + 1):
274
+ # A finish_reason=length overflow is deterministic at a fixed cap, so
275
+ # widen the output budget each retry instead of replaying the request.
276
+ max_tokens = JUDGE_MAX_TOKENS * attempt
277
  r = client.post(
278
  ROUTER_URL,
279
  headers={"Authorization": f"Bearer {HF_TOKEN}"},
 
284
  {"role": "user", "content": prompt},
285
  ],
286
  "temperature": 0.1,
287
+ "max_tokens": max_tokens,
288
  },
289
  timeout=600,
290
  )
 
293
  content = choice.get("message", {}).get("content") or ""
294
  match = re.search(r"\{[\s\S]*\}", content)
295
  if match:
296
+ try:
297
+ return json.loads(match.group(0))
298
+ except json.JSONDecodeError as e:
299
+ last_error = e
300
+ else:
301
+ finish_reason = choice.get("finish_reason", "unknown")
302
+ last_error = ValueError(
303
+ f"Judge returned no JSON (finish_reason={finish_reason}): {content[:200]}"
304
+ )
305
  if attempt < JUDGE_ATTEMPTS:
306
+ log(
307
+ f"judge attempt {attempt} produced no valid JSON; "
308
+ f"retrying with max_tokens={JUDGE_MAX_TOKENS * (attempt + 1)}"
309
+ )
310
  raise last_error
311
 
312
 
 
417
  "skipped",
418
  }:
419
  current.update(base)
420
+ elif current.get("sha") == lb["sha"] and current.get("status") == "error":
421
+ # Errors at the same revision are usually deterministic: retry a
422
+ # bounded number of times, then park the error until the logbook
423
+ # actually changes, so broken books don't starve the queue.
424
+ attempts = current.get("attempts", 0)
425
+ if attempts < JUDGE_MAX_RETRIES:
426
+ mark(
427
+ key,
428
+ "queued",
429
+ f"retrying after error (attempt {attempts + 1}/{JUDGE_MAX_RETRIES})",
430
+ **base,
431
+ )
432
+ queued += 1
433
  else:
434
  mark(
435
  key,
436
  "queued",
437
  "re-judging (logbook updated)" if prev else "new logbook",
438
+ attempts=0,
439
  **base,
440
  )
441
  queued += 1
 
501
  mark(key, "judged", summary)
502
  log(f"verdict {key}: {summary}")
503
  except Exception as e:
504
+ attempts = (STATE["spaces"].get(key) or {}).get("attempts", 0) + 1
505
+ mark(
506
+ key,
507
+ "error",
508
+ f"attempt {attempts}/{JUDGE_MAX_RETRIES} failed: {repr(e)[:160]}",
509
+ attempts=attempts,
510
+ **base,
511
+ )
512
+ log(f"error judging {key} (attempt {attempts}/{JUDGE_MAX_RETRIES}): {e!r}")
513
 
514
  if judged:
515
  save_verdicts(VERDICTS)
test_app.py CHANGED
@@ -101,6 +101,35 @@ class CallJudgeTest(unittest.TestCase):
101
  self.assertEqual(call_judge(client, "prompt"), {"claims": []})
102
  self.assertEqual(len(client.requests), 2)
103
  self.assertEqual(client.requests[0][1]["json"]["max_tokens"], 8000)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
104
 
105
 
106
  class SchedulerConfigurationTest(unittest.TestCase):
@@ -195,6 +224,40 @@ class DiscoveryQueueTest(unittest.TestCase):
195
  self.assertEqual(STATE["spaces"]["owner/logbook"]["status"], "queued")
196
  self.assertEqual(discover_and_queue(), {"found": 1, "queued": 0})
197
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
198
  @patch("app.discover_logbooks")
199
  def test_matching_persisted_verdict_is_not_requeued(self, discover):
200
  discover.return_value = [
 
101
  self.assertEqual(call_judge(client, "prompt"), {"claims": []})
102
  self.assertEqual(len(client.requests), 2)
103
  self.assertEqual(client.requests[0][1]["json"]["max_tokens"], 8000)
104
+ # length overflows are deterministic: the retry must widen the budget
105
+ self.assertEqual(client.requests[1][1]["json"]["max_tokens"], 16000)
106
+
107
+ @patch("app.JUDGE_ATTEMPTS", 2)
108
+ @patch("app.JUDGE_MAX_TOKENS", 8000)
109
+ def test_retries_malformed_json_response(self):
110
+ client = FakeClient(
111
+ [
112
+ {
113
+ "choices": [
114
+ {
115
+ "finish_reason": "stop",
116
+ "message": {"content": '{"claims": [,]}'},
117
+ }
118
+ ]
119
+ },
120
+ {
121
+ "choices": [
122
+ {
123
+ "finish_reason": "stop",
124
+ "message": {"content": '{"claims": []}'},
125
+ }
126
+ ]
127
+ },
128
+ ]
129
+ )
130
+
131
+ self.assertEqual(call_judge(client, "prompt"), {"claims": []})
132
+ self.assertEqual(len(client.requests), 2)
133
 
134
 
135
  class SchedulerConfigurationTest(unittest.TestCase):
 
224
  self.assertEqual(STATE["spaces"]["owner/logbook"]["status"], "queued")
225
  self.assertEqual(discover_and_queue(), {"found": 1, "queued": 0})
226
 
227
+ @patch("app.discover_logbooks")
228
+ def test_errored_space_retries_bounded_then_parks(self, discover):
229
+ discover.return_value = [
230
+ {"space_id": "owner/logbook", "orid": "paper-1", "sha": "abc"}
231
+ ]
232
+ STATE["spaces"]["owner/logbook"] = {
233
+ "status": "error",
234
+ "sha": "abc",
235
+ "attempts": 1,
236
+ }
237
+
238
+ self.assertEqual(discover_and_queue(), {"found": 1, "queued": 1})
239
+ entry = STATE["spaces"]["owner/logbook"]
240
+ self.assertEqual(entry["status"], "queued")
241
+ self.assertEqual(entry["detail"], "retrying after error (attempt 2/3)")
242
+
243
+ # at the retry cap the error is parked until the logbook changes
244
+ STATE["spaces"]["owner/logbook"] = {
245
+ "status": "error",
246
+ "sha": "abc",
247
+ "attempts": 3,
248
+ }
249
+ self.assertEqual(discover_and_queue(), {"found": 1, "queued": 0})
250
+ self.assertEqual(STATE["spaces"]["owner/logbook"]["status"], "error")
251
+
252
+ # a new revision resets the counter and re-queues
253
+ discover.return_value = [
254
+ {"space_id": "owner/logbook", "orid": "paper-1", "sha": "def"}
255
+ ]
256
+ self.assertEqual(discover_and_queue(), {"found": 1, "queued": 1})
257
+ entry = STATE["spaces"]["owner/logbook"]
258
+ self.assertEqual(entry["status"], "queued")
259
+ self.assertEqual(entry["attempts"], 0)
260
+
261
  @patch("app.discover_logbooks")
262
  def test_matching_persisted_verdict_is_not_requeued(self, discover):
263
  discover.return_value = [