Wiki Jokes: daily LLM jokes from Wikipedia featured article
- FastAPI + APScheduler + SQLite, single container - Daily generation at 06:00 Europe/Copenhagen + cold-start generation - Retry with backoff (3 attempts/15 min), stale fallback with banner - OpenAI-compatible endpoint via env (OPENAI_BASE_URL/MODEL/API_KEY) - docker-compose with named volume for joke persistence
This commit is contained in:
@@ -0,0 +1,79 @@
|
||||
"""Fetch Wikipedia's 'Today's Featured Article' via the REST API feed."""
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from datetime import date
|
||||
|
||||
import httpx
|
||||
|
||||
from .config import settings
|
||||
|
||||
FEED_URL = (
|
||||
"https://{lang}.wikipedia.org/api/rest_v1/feed/featured/{year:04d}/{month:02d}/{day:02d}"
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class FeaturedArticle:
|
||||
title: str
|
||||
url: str
|
||||
extract: str # plain-text extract, HTML stripped
|
||||
|
||||
|
||||
def _strip_html(html: str) -> str:
|
||||
text = re.sub(r"<[^>]+>", " ", html)
|
||||
text = (
|
||||
text.replace("&", "&")
|
||||
.replace(""", '"')
|
||||
.replace("'", "'")
|
||||
.replace("<", "<")
|
||||
.replace(">", ">")
|
||||
.replace(" ", " ")
|
||||
)
|
||||
return re.sub(r"\s+", " ", text).strip()
|
||||
|
||||
|
||||
def get_featured_article(day: date) -> FeaturedArticle:
|
||||
url = FEED_URL.format(
|
||||
lang=settings.wikipedia_lang, year=day.year, month=day.month, day=day.day
|
||||
)
|
||||
resp = httpx.get(
|
||||
url,
|
||||
headers={
|
||||
"User-Agent": "wiki-jokes/1.0 (daily joke generator; contact: admin@example.com)",
|
||||
"Accept": "application/json",
|
||||
},
|
||||
timeout=30,
|
||||
follow_redirects=True,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
|
||||
tfa = data.get("tfa")
|
||||
if not tfa:
|
||||
raise ValueError(f"No featured article (tfa) in feed for {day.isoformat()}")
|
||||
|
||||
# Prefer the REST "extracts.plain.text" if present, else the summary HTML.
|
||||
extract = ""
|
||||
extracts = tfa.get("extracts") or {}
|
||||
plain = extracts.get("plain") or {}
|
||||
if plain.get("text"):
|
||||
extract = plain["text"]
|
||||
elif tfa.get("extract"):
|
||||
extract = tfa["extract"]
|
||||
elif tfa.get("summary"):
|
||||
extract = _strip_html(tfa["summary"])
|
||||
|
||||
extract = _strip_html(extract) if "<" in extract else extract
|
||||
if not extract:
|
||||
raise ValueError(f"Featured article for {day.isoformat()} has no extract text")
|
||||
|
||||
# Cap the extract so the prompt stays bounded (~4000 chars).
|
||||
extract = extract[:4000]
|
||||
|
||||
display = (tfa.get("displaytitle") or tfa.get("title", "Unknown article"))
|
||||
display = _strip_html(display).replace("_", " ")
|
||||
return FeaturedArticle(
|
||||
title=display,
|
||||
url=tfa.get("content_urls", {}).get("desktop", {}).get("page", url),
|
||||
extract=extract,
|
||||
)
|
||||
Reference in New Issue
Block a user