Wiki Jokes: daily LLM jokes from Wikipedia featured article

- FastAPI + APScheduler + SQLite, single container
- Daily generation at 06:00 Europe/Copenhagen + cold-start generation
- Retry with backoff (3 attempts/15 min), stale fallback with banner
- OpenAI-compatible endpoint via env (OPENAI_BASE_URL/MODEL/API_KEY)
- docker-compose with named volume for joke persistence
This commit is contained in:
2026-09-29 19:38:01 +00:00
commit 075b72ec6c
14 changed files with 601 additions and 0 deletions
+79
View File
@@ -0,0 +1,79 @@
"""Fetch Wikipedia's 'Today's Featured Article' via the REST API feed."""
import re
from dataclasses import dataclass
from datetime import date
import httpx
from .config import settings
FEED_URL = (
"https://{lang}.wikipedia.org/api/rest_v1/feed/featured/{year:04d}/{month:02d}/{day:02d}"
)
@dataclass
class FeaturedArticle:
title: str
url: str
extract: str # plain-text extract, HTML stripped
def _strip_html(html: str) -> str:
text = re.sub(r"<[^>]+>", " ", html)
text = (
text.replace("&amp;", "&")
.replace("&quot;", '"')
.replace("&#39;", "'")
.replace("&lt;", "<")
.replace("&gt;", ">")
.replace("&nbsp;", " ")
)
return re.sub(r"\s+", " ", text).strip()
def get_featured_article(day: date) -> FeaturedArticle:
url = FEED_URL.format(
lang=settings.wikipedia_lang, year=day.year, month=day.month, day=day.day
)
resp = httpx.get(
url,
headers={
"User-Agent": "wiki-jokes/1.0 (daily joke generator; contact: admin@example.com)",
"Accept": "application/json",
},
timeout=30,
follow_redirects=True,
)
resp.raise_for_status()
data = resp.json()
tfa = data.get("tfa")
if not tfa:
raise ValueError(f"No featured article (tfa) in feed for {day.isoformat()}")
# Prefer the REST "extracts.plain.text" if present, else the summary HTML.
extract = ""
extracts = tfa.get("extracts") or {}
plain = extracts.get("plain") or {}
if plain.get("text"):
extract = plain["text"]
elif tfa.get("extract"):
extract = tfa["extract"]
elif tfa.get("summary"):
extract = _strip_html(tfa["summary"])
extract = _strip_html(extract) if "<" in extract else extract
if not extract:
raise ValueError(f"Featured article for {day.isoformat()} has no extract text")
# Cap the extract so the prompt stays bounded (~4000 chars).
extract = extract[:4000]
display = (tfa.get("displaytitle") or tfa.get("title", "Unknown article"))
display = _strip_html(display).replace("_", " ")
return FeaturedArticle(
title=display,
url=tfa.get("content_urls", {}).get("desktop", {}).get("page", url),
extract=extract,
)