- FastAPI + APScheduler + SQLite, single container - Daily generation at 06:00 Europe/Copenhagen + cold-start generation - Retry with backoff (3 attempts/15 min), stale fallback with banner - OpenAI-compatible endpoint via env (OPENAI_BASE_URL/MODEL/API_KEY) - docker-compose with named volume for joke persistence
80 lines
2.2 KiB
Python
80 lines
2.2 KiB
Python
"""Fetch Wikipedia's 'Today's Featured Article' via the REST API feed."""
|
|
import re
|
|
from dataclasses import dataclass
|
|
from datetime import date
|
|
|
|
import httpx
|
|
|
|
from .config import settings
|
|
|
|
FEED_URL = (
|
|
"https://{lang}.wikipedia.org/api/rest_v1/feed/featured/{year:04d}/{month:02d}/{day:02d}"
|
|
)
|
|
|
|
|
|
@dataclass
|
|
class FeaturedArticle:
|
|
title: str
|
|
url: str
|
|
extract: str # plain-text extract, HTML stripped
|
|
|
|
|
|
def _strip_html(html: str) -> str:
|
|
text = re.sub(r"<[^>]+>", " ", html)
|
|
text = (
|
|
text.replace("&", "&")
|
|
.replace(""", '"')
|
|
.replace("'", "'")
|
|
.replace("<", "<")
|
|
.replace(">", ">")
|
|
.replace(" ", " ")
|
|
)
|
|
return re.sub(r"\s+", " ", text).strip()
|
|
|
|
|
|
def get_featured_article(day: date) -> FeaturedArticle:
|
|
url = FEED_URL.format(
|
|
lang=settings.wikipedia_lang, year=day.year, month=day.month, day=day.day
|
|
)
|
|
resp = httpx.get(
|
|
url,
|
|
headers={
|
|
"User-Agent": "wiki-jokes/1.0 (daily joke generator; contact: admin@example.com)",
|
|
"Accept": "application/json",
|
|
},
|
|
timeout=30,
|
|
follow_redirects=True,
|
|
)
|
|
resp.raise_for_status()
|
|
data = resp.json()
|
|
|
|
tfa = data.get("tfa")
|
|
if not tfa:
|
|
raise ValueError(f"No featured article (tfa) in feed for {day.isoformat()}")
|
|
|
|
# Prefer the REST "extracts.plain.text" if present, else the summary HTML.
|
|
extract = ""
|
|
extracts = tfa.get("extracts") or {}
|
|
plain = extracts.get("plain") or {}
|
|
if plain.get("text"):
|
|
extract = plain["text"]
|
|
elif tfa.get("extract"):
|
|
extract = tfa["extract"]
|
|
elif tfa.get("summary"):
|
|
extract = _strip_html(tfa["summary"])
|
|
|
|
extract = _strip_html(extract) if "<" in extract else extract
|
|
if not extract:
|
|
raise ValueError(f"Featured article for {day.isoformat()} has no extract text")
|
|
|
|
# Cap the extract so the prompt stays bounded (~4000 chars).
|
|
extract = extract[:4000]
|
|
|
|
display = (tfa.get("displaytitle") or tfa.get("title", "Unknown article"))
|
|
display = _strip_html(display).replace("_", " ")
|
|
return FeaturedArticle(
|
|
title=display,
|
|
url=tfa.get("content_urls", {}).get("desktop", {}).get("page", url),
|
|
extract=extract,
|
|
)
|