Raise LLM read timeout to 300s; prefer LAN endpoint in env example

Midnight failures were read timeouts against the public URL — hairpin NAT
and/or the request queuing behind other traffic on the shared model.
LAN endpoint + longer timeout covers both.
This commit is contained in:
2026-09-30 08:06:18 +00:00
parent 6bc9fa0e8d
commit 7dc32a50e2
2 changed files with 6 additions and 3 deletions
+3 -2
View File
@@ -1,7 +1,8 @@
# Copy to .env and adjust. docker compose reads this automatically.
# Any OpenAI-compatible endpoint (llamaswap example below)
OPENAI_BASE_URL=http://swap.obahan.xyz/v1
# Any OpenAI-compatible endpoint. Use the LAN address from the minisforum —
# the public URL hairpins through the router and can stall.
OPENAI_BASE_URL=http://minisforum:65000/v1
OPENAI_MODEL=halogen-qwen3.8-flash-next
OPENAI_API_KEY=***
+3 -1
View File
@@ -124,7 +124,9 @@ def generate_jokes(article_title: str, article_extract: str) -> list[str]:
"max_tokens": 800,
}
resp = httpx.post(url, headers=headers, json=body, timeout=120)
# 300s: the model may be shared (e.g. serving this agent too), so a
# request can legitimately queue behind other traffic.
resp = httpx.post(url, headers=headers, json=body, timeout=300)
resp.raise_for_status()
payload = resp.json()
content = payload["choices"][0]["message"]["content"]