"""Fetch Wikipedia's 'Today's Featured Article' via the REST API feed.""" import re from dataclasses import dataclass from datetime import date import httpx from .config import settings FEED_URL = ( "https://{lang}.wikipedia.org/api/rest_v1/feed/featured/{year:04d}/{month:02d}/{day:02d}" ) @dataclass class FeaturedArticle: title: str url: str extract: str # plain-text extract, HTML stripped def _strip_html(html: str) -> str: text = re.sub(r"<[^>]+>", " ", html) text = ( text.replace("&", "&") .replace(""", '"') .replace("'", "'") .replace("<", "<") .replace(">", ">") .replace(" ", " ") ) return re.sub(r"\s+", " ", text).strip() def get_featured_article(day: date) -> FeaturedArticle: url = FEED_URL.format( lang=settings.wikipedia_lang, year=day.year, month=day.month, day=day.day ) resp = httpx.get( url, headers={ "User-Agent": "wiki-jokes/1.0 (daily joke generator; contact: admin@example.com)", "Accept": "application/json", }, timeout=30, follow_redirects=True, ) resp.raise_for_status() data = resp.json() tfa = data.get("tfa") if not tfa: raise ValueError(f"No featured article (tfa) in feed for {day.isoformat()}") # Prefer the REST "extracts.plain.text" if present, else the summary HTML. extract = "" extracts = tfa.get("extracts") or {} plain = extracts.get("plain") or {} if plain.get("text"): extract = plain["text"] elif tfa.get("extract"): extract = tfa["extract"] elif tfa.get("summary"): extract = _strip_html(tfa["summary"]) extract = _strip_html(extract) if "<" in extract else extract if not extract: raise ValueError(f"Featured article for {day.isoformat()} has no extract text") # Cap the extract so the prompt stays bounded (~4000 chars). extract = extract[:4000] display = (tfa.get("displaytitle") or tfa.get("title", "Unknown article")) display = _strip_html(display).replace("_", " ") return FeaturedArticle( title=display, url=tfa.get("content_urls", {}).get("desktop", {}).get("page", url), extract=extract, )