Wiki Jokes: daily LLM jokes from Wikipedia featured article

- FastAPI + APScheduler + SQLite, single container
- Daily generation at 06:00 Europe/Copenhagen + cold-start generation
- Retry with backoff (3 attempts/15 min), stale fallback with banner
- OpenAI-compatible endpoint via env (OPENAI_BASE_URL/MODEL/API_KEY)
- docker-compose with named volume for joke persistence
This commit is contained in:
2026-09-29 19:38:01 +00:00
commit 075b72ec6c
14 changed files with 601 additions and 0 deletions
View File
+41
View File
@@ -0,0 +1,41 @@
import os
from dataclasses import dataclass, field
@dataclass
class Settings:
openai_base_url: str = field(
default_factory=lambda: os.environ.get(
"OPENAI_BASE_URL", "https://api.openai.com/v1"
).rstrip("/")
)
openai_model: str = field(
default_factory=lambda: os.environ.get("OPENAI_MODEL", "gpt-4o-mini")
)
openai_api_key: str = field(
default_factory=lambda: os.environ.get("OPENAI_API_KEY", "")
)
db_path: str = field(
default_factory=lambda: os.environ.get("DB_PATH", "/data/jokes.db")
)
timezone: str = field(
default_factory=lambda: os.environ.get("TZ", "Europe/Copenhagen")
)
generate_hour: int = field(
default_factory=lambda: int(os.environ.get("GENERATE_HOUR", "6"))
)
generate_minute: int = field(
default_factory=lambda: int(os.environ.get("GENERATE_MINUTE", "0"))
)
max_attempts: int = field(
default_factory=lambda: int(os.environ.get("MAX_ATTEMPTS", "3"))
)
retry_minutes: int = field(
default_factory=lambda: int(os.environ.get("RETRY_MINUTES", "15"))
)
wikipedia_lang: str = field(
default_factory=lambda: os.environ.get("WIKIPEDIA_LANG", "en")
)
settings = Settings()
+86
View File
@@ -0,0 +1,86 @@
import sqlite3
from contextlib import contextmanager
from datetime import date
from .config import settings
SCHEMA = """
CREATE TABLE IF NOT EXISTS jokes (
day TEXT PRIMARY KEY,
article_title TEXT NOT NULL,
article_url TEXT NOT NULL,
jokes TEXT NOT NULL,
generated_at TEXT NOT NULL
);
"""
def init_db() -> None:
with _conn() as conn:
conn.executescript(SCHEMA)
@contextmanager
def _conn():
conn = sqlite3.connect(settings.db_path)
try:
yield conn
conn.commit()
finally:
conn.close()
def save_batch(day: date, article_title: str, article_url: str, jokes: list[str]) -> None:
import json
from datetime import datetime, timezone
with _conn() as conn:
conn.execute(
"INSERT OR REPLACE INTO jokes (day, article_title, article_url, jokes, generated_at) "
"VALUES (?, ?, ?, ?, ?)",
(
day.isoformat(),
article_title,
article_url,
json.dumps(jokes, ensure_ascii=False),
datetime.now(timezone.utc).isoformat(timespec="seconds"),
),
)
def get_day(day: date) -> dict | None:
import json
with _conn() as conn:
row = conn.execute(
"SELECT day, article_title, article_url, jokes, generated_at FROM jokes WHERE day = ?",
(day.isoformat(),),
).fetchone()
if row is None:
return None
return {
"day": row[0],
"article_title": row[1],
"article_url": row[2],
"jokes": json.loads(row[3]),
"generated_at": row[4],
}
def get_latest() -> dict | None:
import json
with _conn() as conn:
row = conn.execute(
"SELECT day, article_title, article_url, jokes, generated_at "
"FROM jokes ORDER BY day DESC LIMIT 1"
).fetchone()
if row is None:
return None
return {
"day": row[0],
"article_title": row[1],
"article_url": row[2],
"jokes": json.loads(row[3]),
"generated_at": row[4],
}
+60
View File
@@ -0,0 +1,60 @@
"""Daily joke generation job with retry/backoff and stale-fallback semantics."""
import logging
from datetime import date, datetime, timedelta, timezone
from zoneinfo import ZoneInfo
from . import db
from .config import settings
from .llm_client import generate_jokes
from .wiki_client import get_featured_article
log = logging.getLogger("jokes.generator")
# In-process retry bookkeeping (resets on container restart; cold-start
# generation covers the restart case anyway).
_attempts_today: dict[str, int] = {}
def today_local() -> date:
return datetime.now(ZoneInfo(settings.timezone)).date()
def _attempt_key(day: date) -> str:
return day.isoformat()
def generation_exhausted(day: date) -> bool:
return _attempts_today.get(_attempt_key(day), 0) >= settings.max_attempts
def run_generation() -> bool:
"""Try to generate today's jokes. Returns True on success."""
day = today_local()
if db.get_day(day) is not None:
log.info("Jokes for %s already exist, skipping", day)
return True
attempts = _attempts_today.get(_attempt_key(day), 0)
if attempts >= settings.max_attempts:
log.warning("Generation for %s already exhausted (%d attempts)", day, attempts)
return False
_attempts_today[_attempt_key(day)] = attempts + 1
try:
article = get_featured_article(day)
log.info("Fetched featured article: %s", article.title)
jokes = generate_jokes(article.title, article.extract)
db.save_batch(day, article.title, article.url, jokes)
log.info("Saved %d jokes for %s", len(jokes), day)
return True
except Exception as exc: # noqa: BLE001 — job must never crash the scheduler
used = _attempts_today[_attempt_key(day)]
log.error(
"Generation attempt %d/%d for %s failed: %s",
used, settings.max_attempts, day, exc,
)
return False
def next_retry_delay() -> timedelta:
return timedelta(minutes=settings.retry_minutes)
+66
View File
@@ -0,0 +1,66 @@
"""Call an OpenAI-compatible chat completions endpoint to generate jokes."""
import json
import re
import httpx
from .config import settings
SYSTEM_PROMPT = (
"You are a comedian. You will receive the text of Wikipedia's featured article "
"of the day. Write exactly 5 short, clean, family-friendly jokes inspired by "
"facts from the article. Vary the style (one-liners, puns, observational). "
"The jokes must be understandable on their own without reading the article. "
"Respond with ONLY a JSON array of 5 strings and nothing else. "
"Example: [\"joke one\", \"joke two\", \"joke three\", \"joke four\", \"joke five\"]"
)
def _extract_json_array(text: str) -> list[str]:
"""Pull a JSON array of strings out of a possibly messy LLM response."""
text = text.strip()
# Strip markdown fences if present.
fence = re.search(r"```(?:json)?\s*(.*?)```", text, re.DOTALL)
if fence:
text = fence.group(1).strip()
# Find the outermost [...] span.
start = text.find("[")
end = text.rfind("]")
if start == -1 or end == -1 or end <= start:
raise ValueError(f"No JSON array found in LLM response: {text[:200]!r}")
data = json.loads(text[start : end + 1])
if not isinstance(data, list):
raise ValueError("Parsed JSON is not a list")
jokes = [str(j).strip() for j in data if str(j).strip()]
if len(jokes) < 5:
raise ValueError(f"Only {len(jokes)} jokes returned, expected 5")
return jokes[:5]
def generate_jokes(article_title: str, article_extract: str) -> list[str]:
url = f"{settings.openai_base_url}/chat/completions"
headers = {"Content-Type": "application/json"}
if settings.openai_api_key:
headers["Authorization"] = f"Bearer {settings.openai_api_key}"
user_prompt = (
f"Today's Wikipedia featured article: \"{article_title}\"\n\n"
f"Article text:\n{article_extract}\n\n"
"Now write exactly 5 jokes as a JSON array of 5 strings."
)
body = {
"model": settings.openai_model,
"messages": [
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": user_prompt},
],
"temperature": 0.9,
"max_tokens": 800,
}
resp = httpx.post(url, headers=headers, json=body, timeout=120)
resp.raise_for_status()
payload = resp.json()
content = payload["choices"][0]["message"]["content"]
return _extract_json_array(content)
+111
View File
@@ -0,0 +1,111 @@
import logging
from contextlib import asynccontextmanager
from datetime import date
from apscheduler.schedulers.asyncio import AsyncIOScheduler
from apscheduler.triggers.cron import CronTrigger
from apscheduler.triggers.interval import IntervalTrigger
from fastapi import FastAPI
from fastapi.responses import HTMLResponse, JSONResponse
from zoneinfo import ZoneInfo
from . import db
from .config import settings
from .generator import generation_exhausted, run_generation, today_local
from .templates import PAGE_TEMPLATE
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(levelname)s %(name)s: %(message)s",
)
log = logging.getLogger("jokes.app")
scheduler = AsyncIOScheduler(timezone=ZoneInfo(settings.timezone))
async def tick() -> None:
"""Runs every RETRY_MINUTES: generate today's jokes if missing,
unless we've exhausted retries for today."""
day = today_local()
if db.get_day(day) is not None:
return
if generation_exhausted(day):
return
run_generation()
@asynccontextmanager
async def lifespan(app: FastAPI):
db.init_db()
# Cold start: generate immediately if today's jokes are missing.
if db.get_day(today_local()) is None:
log.info("Cold start: generating today's jokes now")
run_generation()
# Daily schedule.
scheduler.add_job(
tick,
CronTrigger(
hour=settings.generate_hour,
minute=settings.generate_minute,
timezone=ZoneInfo(settings.timezone),
),
id="daily-generation",
)
# Retry loop (no-ops when today's jokes exist or retries are exhausted).
scheduler.add_job(
tick,
IntervalTrigger(minutes=settings.retry_minutes),
id="retry-generation",
)
scheduler.start()
yield
scheduler.shutdown(wait=False)
app = FastAPI(title="Wiki Jokes", lifespan=lifespan)
@app.get("/", response_class=HTMLResponse)
def index():
today = today_local()
batch = db.get_day(today)
stale = False
if batch is None:
batch = db.get_latest()
stale = True
if batch is None:
return HTMLResponse(
"<h1>Wiki Jokes</h1><p>No jokes yet — the daily generator is "
"working on it. Check back soon!</p>",
status_code=503,
)
items = "\n".join(
f' <li class="joke">{j}</li>' for j in batch["jokes"]
)
stale_note = (
'<p class="stale">⚠️ Today\'s jokes are still being prepared — '
"showing the latest available batch.</p>"
if stale
else ""
)
return PAGE_TEMPLATE.format(
day=batch["day"],
stale_note=stale_note,
jokes=items,
article_title=batch["article_title"],
article_url=batch["article_url"],
)
@app.get("/health")
def health():
today = today_local()
has_today = db.get_day(today) is not None
return JSONResponse(
{
"status": "ok",
"today": today.isoformat(),
"has_todays_jokes": has_today,
"retries_exhausted": generation_exhausted(today),
}
)
+54
View File
@@ -0,0 +1,54 @@
PAGE_TEMPLATE = """<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Wiki Jokes — {day}</title>
<style>
:root {{ color-scheme: light dark; }}
body {{
font-family: Georgia, 'Times New Roman', serif;
max-width: 720px; margin: 0 auto; padding: 2rem 1rem;
background: #fdfcf8; color: #222;
}}
@media (prefers-color-scheme: dark) {{
body {{ background: #1a1a1e; color: #e8e6e1; }}
a {{ color: #8ab4f8; }}
.joke {{ background: #26262c; border-color: #3a3a42; }}
header h1 {{ color: #f0ede6; }}
}}
header h1 {{ font-size: 2rem; margin-bottom: .25rem; }}
.date {{ color: #777; margin-top: 0; }}
.stale {{
background: #fff3cd; color: #665100; padding: .5rem .75rem;
border-radius: 6px; font-size: .9rem;
}}
@media (prefers-color-scheme: dark) {{
.stale {{ background: #4a3c0a; color: #ffd970; }}
}}
ul.jokes {{ list-style: none; padding: 0; }}
li.joke {{
background: #fff; border: 1px solid #e2ddd2; border-radius: 10px;
padding: 1rem 1.25rem; margin: .9rem 0; font-size: 1.15rem;
line-height: 1.5; box-shadow: 0 1px 3px rgba(0,0,0,.06);
}}
li.joke::before {{ content: "😄 "; }}
footer {{ margin-top: 2rem; font-size: .9rem; color: #888; }}
</style>
</head>
<body>
<header>
<h1>Wiki Jokes</h1>
<p class="date">{day}</p>
</header>
{stale_note}
<ul class="jokes">
{jokes}
</ul>
<footer>
Inspired by today's Wikipedia featured article:
<a href="{article_url}" rel="noopener">{article_title}</a>
</footer>
</body>
</html>
"""
+79
View File
@@ -0,0 +1,79 @@
"""Fetch Wikipedia's 'Today's Featured Article' via the REST API feed."""
import re
from dataclasses import dataclass
from datetime import date
import httpx
from .config import settings
FEED_URL = (
"https://{lang}.wikipedia.org/api/rest_v1/feed/featured/{year:04d}/{month:02d}/{day:02d}"
)
@dataclass
class FeaturedArticle:
title: str
url: str
extract: str # plain-text extract, HTML stripped
def _strip_html(html: str) -> str:
text = re.sub(r"<[^>]+>", " ", html)
text = (
text.replace("&amp;", "&")
.replace("&quot;", '"')
.replace("&#39;", "'")
.replace("&lt;", "<")
.replace("&gt;", ">")
.replace("&nbsp;", " ")
)
return re.sub(r"\s+", " ", text).strip()
def get_featured_article(day: date) -> FeaturedArticle:
url = FEED_URL.format(
lang=settings.wikipedia_lang, year=day.year, month=day.month, day=day.day
)
resp = httpx.get(
url,
headers={
"User-Agent": "wiki-jokes/1.0 (daily joke generator; contact: admin@example.com)",
"Accept": "application/json",
},
timeout=30,
follow_redirects=True,
)
resp.raise_for_status()
data = resp.json()
tfa = data.get("tfa")
if not tfa:
raise ValueError(f"No featured article (tfa) in feed for {day.isoformat()}")
# Prefer the REST "extracts.plain.text" if present, else the summary HTML.
extract = ""
extracts = tfa.get("extracts") or {}
plain = extracts.get("plain") or {}
if plain.get("text"):
extract = plain["text"]
elif tfa.get("extract"):
extract = tfa["extract"]
elif tfa.get("summary"):
extract = _strip_html(tfa["summary"])
extract = _strip_html(extract) if "<" in extract else extract
if not extract:
raise ValueError(f"Featured article for {day.isoformat()} has no extract text")
# Cap the extract so the prompt stays bounded (~4000 chars).
extract = extract[:4000]
display = (tfa.get("displaytitle") or tfa.get("title", "Unknown article"))
display = _strip_html(display).replace("_", " ")
return FeaturedArticle(
title=display,
url=tfa.get("content_urls", {}).get("desktop", {}).get("page", url),
extract=extract,
)