commit 075b72ec6c16a8f0ee8a7bd89418d6c035a95172 Author: Hermes Agent Date: Tue Sep 29 19:38:01 2026 +0000 Wiki Jokes: daily LLM jokes from Wikipedia featured article - FastAPI + APScheduler + SQLite, single container - Daily generation at 06:00 Europe/Copenhagen + cold-start generation - Retry with backoff (3 attempts/15 min), stale fallback with banner - OpenAI-compatible endpoint via env (OPENAI_BASE_URL/MODEL/API_KEY) - docker-compose with named volume for joke persistence diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..82bdc65 --- /dev/null +++ b/.env.example @@ -0,0 +1,17 @@ +# Copy to .env and adjust. docker compose reads this automatically. + +# Any OpenAI-compatible endpoint (llamaswap example below) +OPENAI_BASE_URL=http://swap.obahan.xyz/v1 +OPENAI_MODEL=halogen-qwen3.8-flash-next +OPENAI_API_KEY=*** + +# Daily generation time (in TZ below) +GENERATE_HOUR=6 +GENERATE_MINUTE=0 + +# Failure handling: retry every RETRY_MINUTES, up to MAX_ATTEMPTS per day +MAX_ATTEMPTS=3 +RETRY_MINUTES=15 + +# Wikipedia language for the featured-article feed +WIKIPEDIA_LANG=en diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..a98520d --- /dev/null +++ b/.gitignore @@ -0,0 +1,4 @@ +.venv/ +__pycache__/ +*.pyc +.env diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..c28f629 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,16 @@ +FROM python:3.12-slim + +WORKDIR /srv/wiki-jokes + +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY app/ ./app/ + +# Run as non-root; /data is the SQLite volume mount point. +RUN useradd -m appuser && mkdir -p /data && chown -R appuser:appuser /data /srv +USER appuser + +EXPOSE 8000 + +CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/README.md b/README.md new file mode 100644 index 0000000..1ee6978 --- /dev/null +++ b/README.md @@ -0,0 +1,39 @@ +# Wiki Jokes + +5 jokes every day, generated by an LLM from Wikipedia's *Today's Featured Article*. + +## How it works + +- A scheduled job (06:00 Europe/Copenhagen, plus immediately on cold start) fetches + the day's featured article from the Wikipedia REST feed and asks any + OpenAI-compatible endpoint for exactly 5 jokes as JSON. +- Jokes are stored in SQLite (`/data/jokes.db`) and served instantly — no LLM call + at request time. +- On failure: retries every 15 min, up to 3 attempts per day. If still failing, + the page keeps showing the latest batch with a "stale" note. + +## Run + +```bash +cp .env.example .env # set OPENAI_BASE_URL / OPENAI_MODEL / OPENAI_API_KEY +docker compose up -d --build +# open http://localhost:8080 +``` + +## Configuration (env vars) + +| Var | Default | Purpose | +|---|---|---| +| `OPENAI_BASE_URL` | `https://api.openai.com/v1` | OpenAI-compatible endpoint (e.g. llamaswap) | +| `OPENAI_MODEL` | `gpt-4o-mini` | Model name | +| `OPENAI_API_KEY` | *(empty)* | Bearer token (optional for llamaswap) | +| `GENERATE_HOUR` / `GENERATE_MINUTE` | `6` / `0` | Daily generation time | +| `MAX_ATTEMPTS` | `3` | Retry budget per day | +| `RETRY_MINUTES` | `15` | Retry interval | +| `WIKIPEDIA_LANG` | `en` | Wikipedia language for the feed | +| `DB_PATH` | `/data/jokes.db` | SQLite location (volume-mounted) | + +## Endpoints + +- `GET /` — today's 5 jokes + link to the featured article +- `GET /health` — JSON: whether today's jokes exist / retries exhausted diff --git a/app/__init__.py b/app/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/app/config.py b/app/config.py new file mode 100644 index 0000000..0c35af4 --- /dev/null +++ b/app/config.py @@ -0,0 +1,41 @@ +import os +from dataclasses import dataclass, field + + +@dataclass +class Settings: + openai_base_url: str = field( + default_factory=lambda: os.environ.get( + "OPENAI_BASE_URL", "https://api.openai.com/v1" + ).rstrip("/") + ) + openai_model: str = field( + default_factory=lambda: os.environ.get("OPENAI_MODEL", "gpt-4o-mini") + ) + openai_api_key: str = field( + default_factory=lambda: os.environ.get("OPENAI_API_KEY", "") + ) + db_path: str = field( + default_factory=lambda: os.environ.get("DB_PATH", "/data/jokes.db") + ) + timezone: str = field( + default_factory=lambda: os.environ.get("TZ", "Europe/Copenhagen") + ) + generate_hour: int = field( + default_factory=lambda: int(os.environ.get("GENERATE_HOUR", "6")) + ) + generate_minute: int = field( + default_factory=lambda: int(os.environ.get("GENERATE_MINUTE", "0")) + ) + max_attempts: int = field( + default_factory=lambda: int(os.environ.get("MAX_ATTEMPTS", "3")) + ) + retry_minutes: int = field( + default_factory=lambda: int(os.environ.get("RETRY_MINUTES", "15")) + ) + wikipedia_lang: str = field( + default_factory=lambda: os.environ.get("WIKIPEDIA_LANG", "en") + ) + + +settings = Settings() diff --git a/app/db.py b/app/db.py new file mode 100644 index 0000000..8c7b2ce --- /dev/null +++ b/app/db.py @@ -0,0 +1,86 @@ +import sqlite3 +from contextlib import contextmanager +from datetime import date + +from .config import settings + +SCHEMA = """ +CREATE TABLE IF NOT EXISTS jokes ( + day TEXT PRIMARY KEY, + article_title TEXT NOT NULL, + article_url TEXT NOT NULL, + jokes TEXT NOT NULL, + generated_at TEXT NOT NULL +); +""" + + +def init_db() -> None: + with _conn() as conn: + conn.executescript(SCHEMA) + + +@contextmanager +def _conn(): + conn = sqlite3.connect(settings.db_path) + try: + yield conn + conn.commit() + finally: + conn.close() + + +def save_batch(day: date, article_title: str, article_url: str, jokes: list[str]) -> None: + import json + from datetime import datetime, timezone + + with _conn() as conn: + conn.execute( + "INSERT OR REPLACE INTO jokes (day, article_title, article_url, jokes, generated_at) " + "VALUES (?, ?, ?, ?, ?)", + ( + day.isoformat(), + article_title, + article_url, + json.dumps(jokes, ensure_ascii=False), + datetime.now(timezone.utc).isoformat(timespec="seconds"), + ), + ) + + +def get_day(day: date) -> dict | None: + import json + + with _conn() as conn: + row = conn.execute( + "SELECT day, article_title, article_url, jokes, generated_at FROM jokes WHERE day = ?", + (day.isoformat(),), + ).fetchone() + if row is None: + return None + return { + "day": row[0], + "article_title": row[1], + "article_url": row[2], + "jokes": json.loads(row[3]), + "generated_at": row[4], + } + + +def get_latest() -> dict | None: + import json + + with _conn() as conn: + row = conn.execute( + "SELECT day, article_title, article_url, jokes, generated_at " + "FROM jokes ORDER BY day DESC LIMIT 1" + ).fetchone() + if row is None: + return None + return { + "day": row[0], + "article_title": row[1], + "article_url": row[2], + "jokes": json.loads(row[3]), + "generated_at": row[4], + } diff --git a/app/generator.py b/app/generator.py new file mode 100644 index 0000000..09116e6 --- /dev/null +++ b/app/generator.py @@ -0,0 +1,60 @@ +"""Daily joke generation job with retry/backoff and stale-fallback semantics.""" +import logging +from datetime import date, datetime, timedelta, timezone +from zoneinfo import ZoneInfo + +from . import db +from .config import settings +from .llm_client import generate_jokes +from .wiki_client import get_featured_article + +log = logging.getLogger("jokes.generator") + +# In-process retry bookkeeping (resets on container restart; cold-start +# generation covers the restart case anyway). +_attempts_today: dict[str, int] = {} + + +def today_local() -> date: + return datetime.now(ZoneInfo(settings.timezone)).date() + + +def _attempt_key(day: date) -> str: + return day.isoformat() + + +def generation_exhausted(day: date) -> bool: + return _attempts_today.get(_attempt_key(day), 0) >= settings.max_attempts + + +def run_generation() -> bool: + """Try to generate today's jokes. Returns True on success.""" + day = today_local() + if db.get_day(day) is not None: + log.info("Jokes for %s already exist, skipping", day) + return True + + attempts = _attempts_today.get(_attempt_key(day), 0) + if attempts >= settings.max_attempts: + log.warning("Generation for %s already exhausted (%d attempts)", day, attempts) + return False + + _attempts_today[_attempt_key(day)] = attempts + 1 + try: + article = get_featured_article(day) + log.info("Fetched featured article: %s", article.title) + jokes = generate_jokes(article.title, article.extract) + db.save_batch(day, article.title, article.url, jokes) + log.info("Saved %d jokes for %s", len(jokes), day) + return True + except Exception as exc: # noqa: BLE001 — job must never crash the scheduler + used = _attempts_today[_attempt_key(day)] + log.error( + "Generation attempt %d/%d for %s failed: %s", + used, settings.max_attempts, day, exc, + ) + return False + + +def next_retry_delay() -> timedelta: + return timedelta(minutes=settings.retry_minutes) diff --git a/app/llm_client.py b/app/llm_client.py new file mode 100644 index 0000000..d79dcae --- /dev/null +++ b/app/llm_client.py @@ -0,0 +1,66 @@ +"""Call an OpenAI-compatible chat completions endpoint to generate jokes.""" +import json +import re + +import httpx + +from .config import settings + +SYSTEM_PROMPT = ( + "You are a comedian. You will receive the text of Wikipedia's featured article " + "of the day. Write exactly 5 short, clean, family-friendly jokes inspired by " + "facts from the article. Vary the style (one-liners, puns, observational). " + "The jokes must be understandable on their own without reading the article. " + "Respond with ONLY a JSON array of 5 strings and nothing else. " + "Example: [\"joke one\", \"joke two\", \"joke three\", \"joke four\", \"joke five\"]" +) + + +def _extract_json_array(text: str) -> list[str]: + """Pull a JSON array of strings out of a possibly messy LLM response.""" + text = text.strip() + # Strip markdown fences if present. + fence = re.search(r"```(?:json)?\s*(.*?)```", text, re.DOTALL) + if fence: + text = fence.group(1).strip() + # Find the outermost [...] span. + start = text.find("[") + end = text.rfind("]") + if start == -1 or end == -1 or end <= start: + raise ValueError(f"No JSON array found in LLM response: {text[:200]!r}") + data = json.loads(text[start : end + 1]) + if not isinstance(data, list): + raise ValueError("Parsed JSON is not a list") + jokes = [str(j).strip() for j in data if str(j).strip()] + if len(jokes) < 5: + raise ValueError(f"Only {len(jokes)} jokes returned, expected 5") + return jokes[:5] + + +def generate_jokes(article_title: str, article_extract: str) -> list[str]: + url = f"{settings.openai_base_url}/chat/completions" + headers = {"Content-Type": "application/json"} + if settings.openai_api_key: + headers["Authorization"] = f"Bearer {settings.openai_api_key}" + + user_prompt = ( + f"Today's Wikipedia featured article: \"{article_title}\"\n\n" + f"Article text:\n{article_extract}\n\n" + "Now write exactly 5 jokes as a JSON array of 5 strings." + ) + + body = { + "model": settings.openai_model, + "messages": [ + {"role": "system", "content": SYSTEM_PROMPT}, + {"role": "user", "content": user_prompt}, + ], + "temperature": 0.9, + "max_tokens": 800, + } + + resp = httpx.post(url, headers=headers, json=body, timeout=120) + resp.raise_for_status() + payload = resp.json() + content = payload["choices"][0]["message"]["content"] + return _extract_json_array(content) diff --git a/app/main.py b/app/main.py new file mode 100644 index 0000000..fbc2f03 --- /dev/null +++ b/app/main.py @@ -0,0 +1,111 @@ +import logging +from contextlib import asynccontextmanager +from datetime import date + +from apscheduler.schedulers.asyncio import AsyncIOScheduler +from apscheduler.triggers.cron import CronTrigger +from apscheduler.triggers.interval import IntervalTrigger +from fastapi import FastAPI +from fastapi.responses import HTMLResponse, JSONResponse +from zoneinfo import ZoneInfo + +from . import db +from .config import settings +from .generator import generation_exhausted, run_generation, today_local +from .templates import PAGE_TEMPLATE + +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s %(levelname)s %(name)s: %(message)s", +) +log = logging.getLogger("jokes.app") + +scheduler = AsyncIOScheduler(timezone=ZoneInfo(settings.timezone)) + + +async def tick() -> None: + """Runs every RETRY_MINUTES: generate today's jokes if missing, + unless we've exhausted retries for today.""" + day = today_local() + if db.get_day(day) is not None: + return + if generation_exhausted(day): + return + run_generation() + + +@asynccontextmanager +async def lifespan(app: FastAPI): + db.init_db() + # Cold start: generate immediately if today's jokes are missing. + if db.get_day(today_local()) is None: + log.info("Cold start: generating today's jokes now") + run_generation() + # Daily schedule. + scheduler.add_job( + tick, + CronTrigger( + hour=settings.generate_hour, + minute=settings.generate_minute, + timezone=ZoneInfo(settings.timezone), + ), + id="daily-generation", + ) + # Retry loop (no-ops when today's jokes exist or retries are exhausted). + scheduler.add_job( + tick, + IntervalTrigger(minutes=settings.retry_minutes), + id="retry-generation", + ) + scheduler.start() + yield + scheduler.shutdown(wait=False) + + +app = FastAPI(title="Wiki Jokes", lifespan=lifespan) + + +@app.get("/", response_class=HTMLResponse) +def index(): + today = today_local() + batch = db.get_day(today) + stale = False + if batch is None: + batch = db.get_latest() + stale = True + if batch is None: + return HTMLResponse( + "

Wiki Jokes

No jokes yet — the daily generator is " + "working on it. Check back soon!

", + status_code=503, + ) + items = "\n".join( + f'
  • {j}
  • ' for j in batch["jokes"] + ) + stale_note = ( + '

    ⚠️ Today\'s jokes are still being prepared — ' + "showing the latest available batch.

    " + if stale + else "" + ) + return PAGE_TEMPLATE.format( + day=batch["day"], + stale_note=stale_note, + jokes=items, + article_title=batch["article_title"], + article_url=batch["article_url"], + ) + + +@app.get("/health") +def health(): + today = today_local() + has_today = db.get_day(today) is not None + return JSONResponse( + { + "status": "ok", + "today": today.isoformat(), + "has_todays_jokes": has_today, + "retries_exhausted": generation_exhausted(today), + } + ) diff --git a/app/templates.py b/app/templates.py new file mode 100644 index 0000000..b6d351c --- /dev/null +++ b/app/templates.py @@ -0,0 +1,54 @@ +PAGE_TEMPLATE = """ + + + + +Wiki Jokes — {day} + + + +
    +

    Wiki Jokes

    +

    {day}

    +
    + {stale_note} + + + + +""" diff --git a/app/wiki_client.py b/app/wiki_client.py new file mode 100644 index 0000000..5b57d39 --- /dev/null +++ b/app/wiki_client.py @@ -0,0 +1,79 @@ +"""Fetch Wikipedia's 'Today's Featured Article' via the REST API feed.""" +import re +from dataclasses import dataclass +from datetime import date + +import httpx + +from .config import settings + +FEED_URL = ( + "https://{lang}.wikipedia.org/api/rest_v1/feed/featured/{year:04d}/{month:02d}/{day:02d}" +) + + +@dataclass +class FeaturedArticle: + title: str + url: str + extract: str # plain-text extract, HTML stripped + + +def _strip_html(html: str) -> str: + text = re.sub(r"<[^>]+>", " ", html) + text = ( + text.replace("&", "&") + .replace(""", '"') + .replace("'", "'") + .replace("<", "<") + .replace(">", ">") + .replace(" ", " ") + ) + return re.sub(r"\s+", " ", text).strip() + + +def get_featured_article(day: date) -> FeaturedArticle: + url = FEED_URL.format( + lang=settings.wikipedia_lang, year=day.year, month=day.month, day=day.day + ) + resp = httpx.get( + url, + headers={ + "User-Agent": "wiki-jokes/1.0 (daily joke generator; contact: admin@example.com)", + "Accept": "application/json", + }, + timeout=30, + follow_redirects=True, + ) + resp.raise_for_status() + data = resp.json() + + tfa = data.get("tfa") + if not tfa: + raise ValueError(f"No featured article (tfa) in feed for {day.isoformat()}") + + # Prefer the REST "extracts.plain.text" if present, else the summary HTML. + extract = "" + extracts = tfa.get("extracts") or {} + plain = extracts.get("plain") or {} + if plain.get("text"): + extract = plain["text"] + elif tfa.get("extract"): + extract = tfa["extract"] + elif tfa.get("summary"): + extract = _strip_html(tfa["summary"]) + + extract = _strip_html(extract) if "<" in extract else extract + if not extract: + raise ValueError(f"Featured article for {day.isoformat()} has no extract text") + + # Cap the extract so the prompt stays bounded (~4000 chars). + extract = extract[:4000] + + display = (tfa.get("displaytitle") or tfa.get("title", "Unknown article")) + display = _strip_html(display).replace("_", " ") + return FeaturedArticle( + title=display, + url=tfa.get("content_urls", {}).get("desktop", {}).get("page", url), + extract=extract, + ) diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..442479c --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,24 @@ +services: + wiki-jokes: + build: . + image: wiki-jokes:latest + container_name: wiki-jokes + restart: unless-stopped + ports: + - "8080:8000" + environment: + TZ: Europe/Copenhagen + OPENAI_BASE_URL: ${OPENAI_BASE_URL:-https://api.openai.com/v1} + OPENAI_MODEL: ${OPENAI_MODEL:-gpt-4o-mini} + OPENAI_API_KEY: ${OPENAI_API_KEY:-} + GENERATE_HOUR: ${GENERATE_HOUR:-6} + GENERATE_MINUTE: ${GENERATE_MINUTE:-0} + MAX_ATTEMPTS: ${MAX_ATTEMPTS:-3} + RETRY_MINUTES: ${RETRY_MINUTES:-15} + WIKIPEDIA_LANG: ${WIKIPEDIA_LANG:-en} + DB_PATH: /data/jokes.db + volumes: + - jokes-data:/data + +volumes: + jokes-data: diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..5c53266 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,4 @@ +fastapi>=0.111 +uvicorn[standard]>=0.30 +httpx>=0.27 +apscheduler>=3.10,<4