Deploy from GitLab 80e62a80
Browse files
agent/skills/pediatry_wiki/scripts/generate_wiki_response.py
CHANGED
|
@@ -106,6 +106,9 @@ _JUDGE_MODEL_MAX_TOKENS = max_completion_tokens_for(JUDGE_MODEL_ID)
|
|
| 106 |
_JUDGE_CALL_TIMEOUT_SECONDS_DEFAULT = 11.0
|
| 107 |
_JUDGE_CALL_TIMEOUT_OVERRIDES_SECONDS: dict[str, float | None] = {
|
| 108 |
"google/gemma-4-26b-a4b-it": None,
|
|
|
|
|
|
|
|
|
|
| 109 |
}
|
| 110 |
JUDGE_CALL_TIMEOUT_SECONDS = _JUDGE_CALL_TIMEOUT_OVERRIDES_SECONDS.get(
|
| 111 |
JUDGE_MODEL_ID.lower(), _JUDGE_CALL_TIMEOUT_SECONDS_DEFAULT
|
|
|
|
| 106 |
_JUDGE_CALL_TIMEOUT_SECONDS_DEFAULT = 11.0
|
| 107 |
_JUDGE_CALL_TIMEOUT_OVERRIDES_SECONDS: dict[str, float | None] = {
|
| 108 |
"google/gemma-4-26b-a4b-it": None,
|
| 109 |
+
# Prod traces showed Qwen's own judge reasoning regularly running past
|
| 110 |
+
# 11s on a normal, successful call — not just on genuinely slow requests.
|
| 111 |
+
"qwen/qwen3.8-27b": 30.0,
|
| 112 |
}
|
| 113 |
JUDGE_CALL_TIMEOUT_SECONDS = _JUDGE_CALL_TIMEOUT_OVERRIDES_SECONDS.get(
|
| 114 |
JUDGE_MODEL_ID.lower(), _JUDGE_CALL_TIMEOUT_SECONDS_DEFAULT
|