From 6bc9fa0e8d8e7e7fedf8fb3353be7a3464ff39a8 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Wed, 30 Sep 2026 07:55:40 +0000 Subject: [PATCH] Resume retries after cooldown instead of giving up for the day MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Previously 3 failed attempts at 06:00 (e.g. llamaswap model cold/starting) permanently blocked generation until the next day. Now the retry loop pauses SLOW_RETRY_MINUTES (default 60) after the burst budget is spent, then resumes — one attempt per hour until jokes exist. --- .env.example | 5 ++++- README.md | 3 ++- app/config.py | 3 +++ app/generator.py | 25 ++++++++++++++++++------- 4 files changed, 27 insertions(+), 9 deletions(-) diff --git a/.env.example b/.env.example index 82bdc65..c81c7b3 100644 --- a/.env.example +++ b/.env.example @@ -9,9 +9,12 @@ OPENAI_API_KEY=*** GENERATE_HOUR=6 GENERATE_MINUTE=0 -# Failure handling: retry every RETRY_MINUTES, up to MAX_ATTEMPTS per day +# Failure handling: retry every RETRY_MINUTES, up to MAX_ATTEMPTS per day. +# After that, retries pause for SLOW_RETRY_MINUTES, then resume (so a cold +# LLM endpoint at 06:00 doesn't kill the whole day). MAX_ATTEMPTS=3 RETRY_MINUTES=15 +SLOW_RETRY_MINUTES=60 # Wikipedia language for the featured-article feed WIKIPEDIA_LANG=en diff --git a/README.md b/README.md index 1ee6978..2a65b6e 100644 --- a/README.md +++ b/README.md @@ -28,8 +28,9 @@ docker compose up -d --build | `OPENAI_MODEL` | `gpt-4o-mini` | Model name | | `OPENAI_API_KEY` | *(empty)* | Bearer token (optional for llamaswap) | | `GENERATE_HOUR` / `GENERATE_MINUTE` | `6` / `0` | Daily generation time | -| `MAX_ATTEMPTS` | `3` | Retry budget per day | +| `MAX_ATTEMPTS` | `3` | Retry budget per burst | | `RETRY_MINUTES` | `15` | Retry interval | +| `SLOW_RETRY_MINUTES` | `60` | Cooldown after budget exhausted, then retries resume | | `WIKIPEDIA_LANG` | `en` | Wikipedia language for the feed | | `DB_PATH` | `/data/jokes.db` | SQLite location (volume-mounted) | diff --git a/app/config.py b/app/config.py index 0c35af4..66f1a2b 100644 --- a/app/config.py +++ b/app/config.py @@ -33,6 +33,9 @@ class Settings: retry_minutes: int = field( default_factory=lambda: int(os.environ.get("RETRY_MINUTES", "15")) ) + slow_retry_minutes: int = field( + default_factory=lambda: int(os.environ.get("SLOW_RETRY_MINUTES", "60")) + ) wikipedia_lang: str = field( default_factory=lambda: os.environ.get("WIKIPEDIA_LANG", "en") ) diff --git a/app/generator.py b/app/generator.py index 09116e6..43043ac 100644 --- a/app/generator.py +++ b/app/generator.py @@ -13,6 +13,7 @@ log = logging.getLogger("jokes.generator") # In-process retry bookkeeping (resets on container restart; cold-start # generation covers the restart case anyway). _attempts_today: dict[str, int] = {} +_last_failure: dict[str, datetime] = {} def today_local() -> date: @@ -24,7 +25,17 @@ def _attempt_key(day: date) -> str: def generation_exhausted(day: date) -> bool: - return _attempts_today.get(_attempt_key(day), 0) >= settings.max_attempts + """True while in cooldown: max attempts hit and the last failure is + younger than SLOW_RETRY_MINUTES. After the cooldown passes, the daily + retry loop resumes at the slow interval — so a cold/down LLM endpoint + at 06:00 no longer kills the whole day.""" + if _attempts_today.get(_attempt_key(day), 0) < settings.max_attempts: + return False + last = _last_failure.get(_attempt_key(day)) + if last is None: + return True + age = datetime.now(timezone.utc) - last + return age < timedelta(minutes=settings.slow_retry_minutes) def run_generation() -> bool: @@ -34,12 +45,11 @@ def run_generation() -> bool: log.info("Jokes for %s already exist, skipping", day) return True - attempts = _attempts_today.get(_attempt_key(day), 0) - if attempts >= settings.max_attempts: - log.warning("Generation for %s already exhausted (%d attempts)", day, attempts) + if generation_exhausted(day): + # In cooldown after max attempts; will resume after slow_retry_minutes. return False - _attempts_today[_attempt_key(day)] = attempts + 1 + _attempts_today[_attempt_key(day)] = _attempts_today.get(_attempt_key(day), 0) + 1 try: article = get_featured_article(day) log.info("Fetched featured article: %s", article.title) @@ -49,9 +59,10 @@ def run_generation() -> bool: return True except Exception as exc: # noqa: BLE001 — job must never crash the scheduler used = _attempts_today[_attempt_key(day)] + _last_failure[_attempt_key(day)] = datetime.now(timezone.utc) log.error( - "Generation attempt %d/%d for %s failed: %s", - used, settings.max_attempts, day, exc, + "Generation attempt %d for %s failed: %s (cooldown %d min before next try)", + used, day, exc, settings.slow_retry_minutes, ) return False