Resume retries after cooldown instead of giving up for the day
Previously 3 failed attempts at 06:00 (e.g. llamaswap model cold/starting) permanently blocked generation until the next day. Now the retry loop pauses SLOW_RETRY_MINUTES (default 60) after the burst budget is spent, then resumes — one attempt per hour until jokes exist.
This commit is contained in:
+4
-1
@@ -9,9 +9,12 @@ OPENAI_API_KEY=***
|
|||||||
GENERATE_HOUR=6
|
GENERATE_HOUR=6
|
||||||
GENERATE_MINUTE=0
|
GENERATE_MINUTE=0
|
||||||
|
|
||||||
# Failure handling: retry every RETRY_MINUTES, up to MAX_ATTEMPTS per day
|
# Failure handling: retry every RETRY_MINUTES, up to MAX_ATTEMPTS per day.
|
||||||
|
# After that, retries pause for SLOW_RETRY_MINUTES, then resume (so a cold
|
||||||
|
# LLM endpoint at 06:00 doesn't kill the whole day).
|
||||||
MAX_ATTEMPTS=3
|
MAX_ATTEMPTS=3
|
||||||
RETRY_MINUTES=15
|
RETRY_MINUTES=15
|
||||||
|
SLOW_RETRY_MINUTES=60
|
||||||
|
|
||||||
# Wikipedia language for the featured-article feed
|
# Wikipedia language for the featured-article feed
|
||||||
WIKIPEDIA_LANG=en
|
WIKIPEDIA_LANG=en
|
||||||
|
|||||||
@@ -28,8 +28,9 @@ docker compose up -d --build
|
|||||||
| `OPENAI_MODEL` | `gpt-4o-mini` | Model name |
|
| `OPENAI_MODEL` | `gpt-4o-mini` | Model name |
|
||||||
| `OPENAI_API_KEY` | *(empty)* | Bearer token (optional for llamaswap) |
|
| `OPENAI_API_KEY` | *(empty)* | Bearer token (optional for llamaswap) |
|
||||||
| `GENERATE_HOUR` / `GENERATE_MINUTE` | `6` / `0` | Daily generation time |
|
| `GENERATE_HOUR` / `GENERATE_MINUTE` | `6` / `0` | Daily generation time |
|
||||||
| `MAX_ATTEMPTS` | `3` | Retry budget per day |
|
| `MAX_ATTEMPTS` | `3` | Retry budget per burst |
|
||||||
| `RETRY_MINUTES` | `15` | Retry interval |
|
| `RETRY_MINUTES` | `15` | Retry interval |
|
||||||
|
| `SLOW_RETRY_MINUTES` | `60` | Cooldown after budget exhausted, then retries resume |
|
||||||
| `WIKIPEDIA_LANG` | `en` | Wikipedia language for the feed |
|
| `WIKIPEDIA_LANG` | `en` | Wikipedia language for the feed |
|
||||||
| `DB_PATH` | `/data/jokes.db` | SQLite location (volume-mounted) |
|
| `DB_PATH` | `/data/jokes.db` | SQLite location (volume-mounted) |
|
||||||
|
|
||||||
|
|||||||
@@ -33,6 +33,9 @@ class Settings:
|
|||||||
retry_minutes: int = field(
|
retry_minutes: int = field(
|
||||||
default_factory=lambda: int(os.environ.get("RETRY_MINUTES", "15"))
|
default_factory=lambda: int(os.environ.get("RETRY_MINUTES", "15"))
|
||||||
)
|
)
|
||||||
|
slow_retry_minutes: int = field(
|
||||||
|
default_factory=lambda: int(os.environ.get("SLOW_RETRY_MINUTES", "60"))
|
||||||
|
)
|
||||||
wikipedia_lang: str = field(
|
wikipedia_lang: str = field(
|
||||||
default_factory=lambda: os.environ.get("WIKIPEDIA_LANG", "en")
|
default_factory=lambda: os.environ.get("WIKIPEDIA_LANG", "en")
|
||||||
)
|
)
|
||||||
|
|||||||
+18
-7
@@ -13,6 +13,7 @@ log = logging.getLogger("jokes.generator")
|
|||||||
# In-process retry bookkeeping (resets on container restart; cold-start
|
# In-process retry bookkeeping (resets on container restart; cold-start
|
||||||
# generation covers the restart case anyway).
|
# generation covers the restart case anyway).
|
||||||
_attempts_today: dict[str, int] = {}
|
_attempts_today: dict[str, int] = {}
|
||||||
|
_last_failure: dict[str, datetime] = {}
|
||||||
|
|
||||||
|
|
||||||
def today_local() -> date:
|
def today_local() -> date:
|
||||||
@@ -24,7 +25,17 @@ def _attempt_key(day: date) -> str:
|
|||||||
|
|
||||||
|
|
||||||
def generation_exhausted(day: date) -> bool:
|
def generation_exhausted(day: date) -> bool:
|
||||||
return _attempts_today.get(_attempt_key(day), 0) >= settings.max_attempts
|
"""True while in cooldown: max attempts hit and the last failure is
|
||||||
|
younger than SLOW_RETRY_MINUTES. After the cooldown passes, the daily
|
||||||
|
retry loop resumes at the slow interval — so a cold/down LLM endpoint
|
||||||
|
at 06:00 no longer kills the whole day."""
|
||||||
|
if _attempts_today.get(_attempt_key(day), 0) < settings.max_attempts:
|
||||||
|
return False
|
||||||
|
last = _last_failure.get(_attempt_key(day))
|
||||||
|
if last is None:
|
||||||
|
return True
|
||||||
|
age = datetime.now(timezone.utc) - last
|
||||||
|
return age < timedelta(minutes=settings.slow_retry_minutes)
|
||||||
|
|
||||||
|
|
||||||
def run_generation() -> bool:
|
def run_generation() -> bool:
|
||||||
@@ -34,12 +45,11 @@ def run_generation() -> bool:
|
|||||||
log.info("Jokes for %s already exist, skipping", day)
|
log.info("Jokes for %s already exist, skipping", day)
|
||||||
return True
|
return True
|
||||||
|
|
||||||
attempts = _attempts_today.get(_attempt_key(day), 0)
|
if generation_exhausted(day):
|
||||||
if attempts >= settings.max_attempts:
|
# In cooldown after max attempts; will resume after slow_retry_minutes.
|
||||||
log.warning("Generation for %s already exhausted (%d attempts)", day, attempts)
|
|
||||||
return False
|
return False
|
||||||
|
|
||||||
_attempts_today[_attempt_key(day)] = attempts + 1
|
_attempts_today[_attempt_key(day)] = _attempts_today.get(_attempt_key(day), 0) + 1
|
||||||
try:
|
try:
|
||||||
article = get_featured_article(day)
|
article = get_featured_article(day)
|
||||||
log.info("Fetched featured article: %s", article.title)
|
log.info("Fetched featured article: %s", article.title)
|
||||||
@@ -49,9 +59,10 @@ def run_generation() -> bool:
|
|||||||
return True
|
return True
|
||||||
except Exception as exc: # noqa: BLE001 — job must never crash the scheduler
|
except Exception as exc: # noqa: BLE001 — job must never crash the scheduler
|
||||||
used = _attempts_today[_attempt_key(day)]
|
used = _attempts_today[_attempt_key(day)]
|
||||||
|
_last_failure[_attempt_key(day)] = datetime.now(timezone.utc)
|
||||||
log.error(
|
log.error(
|
||||||
"Generation attempt %d/%d for %s failed: %s",
|
"Generation attempt %d for %s failed: %s (cooldown %d min before next try)",
|
||||||
used, settings.max_attempts, day, exc,
|
used, day, exc, settings.slow_retry_minutes,
|
||||||
)
|
)
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user