From 7dc32a50e2aeca90a9bffc0e9d7ab38be9abd982 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Wed, 30 Sep 2026 08:06:18 +0000 Subject: [PATCH] Raise LLM read timeout to 300s; prefer LAN endpoint in env example MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Midnight failures were read timeouts against the public URL — hairpin NAT and/or the request queuing behind other traffic on the shared model. LAN endpoint + longer timeout covers both. --- .env.example | 5 +++-- app/llm_client.py | 4 +++- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/.env.example b/.env.example index c81c7b3..16733da 100644 --- a/.env.example +++ b/.env.example @@ -1,7 +1,8 @@ # Copy to .env and adjust. docker compose reads this automatically. -# Any OpenAI-compatible endpoint (llamaswap example below) -OPENAI_BASE_URL=http://swap.obahan.xyz/v1 +# Any OpenAI-compatible endpoint. Use the LAN address from the minisforum — +# the public URL hairpins through the router and can stall. +OPENAI_BASE_URL=http://minisforum:65000/v1 OPENAI_MODEL=halogen-qwen3.8-flash-next OPENAI_API_KEY=*** diff --git a/app/llm_client.py b/app/llm_client.py index da44cb3..13a11f8 100644 --- a/app/llm_client.py +++ b/app/llm_client.py @@ -124,7 +124,9 @@ def generate_jokes(article_title: str, article_extract: str) -> list[str]: "max_tokens": 800, } - resp = httpx.post(url, headers=headers, json=body, timeout=120) + # 300s: the model may be shared (e.g. serving this agent too), so a + # request can legitimately queue behind other traffic. + resp = httpx.post(url, headers=headers, json=body, timeout=300) resp.raise_for_status() payload = resp.json() content = payload["choices"][0]["message"]["content"]