Files
hermes 075b72ec6c Wiki Jokes: daily LLM jokes from Wikipedia featured article
- FastAPI + APScheduler + SQLite, single container
- Daily generation at 06:00 Europe/Copenhagen + cold-start generation
- Retry with backoff (3 attempts/15 min), stale fallback with banner
- OpenAI-compatible endpoint via env (OPENAI_BASE_URL/MODEL/API_KEY)
- docker-compose with named volume for joke persistence
2026-09-29 19:38:01 +00:00

80 lines
2.2 KiB
Python

"""Fetch Wikipedia's 'Today's Featured Article' via the REST API feed."""
import re
from dataclasses import dataclass
from datetime import date
import httpx
from .config import settings
FEED_URL = (
"https://{lang}.wikipedia.org/api/rest_v1/feed/featured/{year:04d}/{month:02d}/{day:02d}"
)
@dataclass
class FeaturedArticle:
title: str
url: str
extract: str # plain-text extract, HTML stripped
def _strip_html(html: str) -> str:
text = re.sub(r"<[^>]+>", " ", html)
text = (
text.replace("&amp;", "&")
.replace("&quot;", '"')
.replace("&#39;", "'")
.replace("&lt;", "<")
.replace("&gt;", ">")
.replace("&nbsp;", " ")
)
return re.sub(r"\s+", " ", text).strip()
def get_featured_article(day: date) -> FeaturedArticle:
url = FEED_URL.format(
lang=settings.wikipedia_lang, year=day.year, month=day.month, day=day.day
)
resp = httpx.get(
url,
headers={
"User-Agent": "wiki-jokes/1.0 (daily joke generator; contact: admin@example.com)",
"Accept": "application/json",
},
timeout=30,
follow_redirects=True,
)
resp.raise_for_status()
data = resp.json()
tfa = data.get("tfa")
if not tfa:
raise ValueError(f"No featured article (tfa) in feed for {day.isoformat()}")
# Prefer the REST "extracts.plain.text" if present, else the summary HTML.
extract = ""
extracts = tfa.get("extracts") or {}
plain = extracts.get("plain") or {}
if plain.get("text"):
extract = plain["text"]
elif tfa.get("extract"):
extract = tfa["extract"]
elif tfa.get("summary"):
extract = _strip_html(tfa["summary"])
extract = _strip_html(extract) if "<" in extract else extract
if not extract:
raise ValueError(f"Featured article for {day.isoformat()} has no extract text")
# Cap the extract so the prompt stays bounded (~4000 chars).
extract = extract[:4000]
display = (tfa.get("displaytitle") or tfa.get("title", "Unknown article"))
display = _strip_html(display).replace("_", " ")
return FeaturedArticle(
title=display,
url=tfa.get("content_urls", {}).get("desktop", {}).get("page", url),
extract=extract,
)