Resume retries after cooldown instead of giving up for the day

Previously 3 failed attempts at 06:00 (e.g. llamaswap model cold/starting)
permanently blocked generation until the next day. Now the retry loop
pauses SLOW_RETRY_MINUTES (default 60) after the burst budget is spent,
then resumes — one attempt per hour until jokes exist.
This commit is contained in:
2026-09-30 07:55:40 +00:00
parent 7e60343ed5
commit 6bc9fa0e8d
4 changed files with 27 additions and 9 deletions
+3
View File
@@ -33,6 +33,9 @@ class Settings:
retry_minutes: int = field(
default_factory=lambda: int(os.environ.get("RETRY_MINUTES", "15"))
)
slow_retry_minutes: int = field(
default_factory=lambda: int(os.environ.get("SLOW_RETRY_MINUTES", "60"))
)
wikipedia_lang: str = field(
default_factory=lambda: os.environ.get("WIKIPEDIA_LANG", "en")
)
+18 -7
View File
@@ -13,6 +13,7 @@ log = logging.getLogger("jokes.generator")
# In-process retry bookkeeping (resets on container restart; cold-start
# generation covers the restart case anyway).
_attempts_today: dict[str, int] = {}
_last_failure: dict[str, datetime] = {}
def today_local() -> date:
@@ -24,7 +25,17 @@ def _attempt_key(day: date) -> str:
def generation_exhausted(day: date) -> bool:
return _attempts_today.get(_attempt_key(day), 0) >= settings.max_attempts
"""True while in cooldown: max attempts hit and the last failure is
younger than SLOW_RETRY_MINUTES. After the cooldown passes, the daily
retry loop resumes at the slow interval — so a cold/down LLM endpoint
at 06:00 no longer kills the whole day."""
if _attempts_today.get(_attempt_key(day), 0) < settings.max_attempts:
return False
last = _last_failure.get(_attempt_key(day))
if last is None:
return True
age = datetime.now(timezone.utc) - last
return age < timedelta(minutes=settings.slow_retry_minutes)
def run_generation() -> bool:
@@ -34,12 +45,11 @@ def run_generation() -> bool:
log.info("Jokes for %s already exist, skipping", day)
return True
attempts = _attempts_today.get(_attempt_key(day), 0)
if attempts >= settings.max_attempts:
log.warning("Generation for %s already exhausted (%d attempts)", day, attempts)
if generation_exhausted(day):
# In cooldown after max attempts; will resume after slow_retry_minutes.
return False
_attempts_today[_attempt_key(day)] = attempts + 1
_attempts_today[_attempt_key(day)] = _attempts_today.get(_attempt_key(day), 0) + 1
try:
article = get_featured_article(day)
log.info("Fetched featured article: %s", article.title)
@@ -49,9 +59,10 @@ def run_generation() -> bool:
return True
except Exception as exc: # noqa: BLE001 — job must never crash the scheduler
used = _attempts_today[_attempt_key(day)]
_last_failure[_attempt_key(day)] = datetime.now(timezone.utc)
log.error(
"Generation attempt %d/%d for %s failed: %s",
used, settings.max_attempts, day, exc,
"Generation attempt %d for %s failed: %s (cooldown %d min before next try)",
used, day, exc, settings.slow_retry_minutes,
)
return False