From c6fd9a5c77fd1210ff65678c54ec1ba48e888318 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Tue, 29 Sep 2026 19:54:56 +0000 Subject: [PATCH] Repair curly-quote JSON from LLM before parsing MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit halogen-qwen3.8-flash-next sometimes uses typographic quotes (“ ”) as JSON string delimiters, which breaks strict json.loads. Add a scanner that normalizes curly-delimited strings to straight quotes while preserving curly quotes used as content inside straight-quoted strings. --- app/llm_client.py | 67 ++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 66 insertions(+), 1 deletion(-) diff --git a/app/llm_client.py b/app/llm_client.py index d79dcae..da44cb3 100644 --- a/app/llm_client.py +++ b/app/llm_client.py @@ -16,6 +16,71 @@ SYSTEM_PROMPT = ( ) +def _try_parse(candidate: str): + try: + return json.loads(candidate) + except json.JSONDecodeError: + return None + + +def _normalize_quotes(text: str) -> str: + """Rewrite LLM quote soup into valid JSON. + + Handles models that use curly quotes (“ ”) as string DELIMITERS and also + as content inside straight-quoted strings. A scanner tracks which quote + character opened the current string so content quotes are preserved + (escaped) instead of breaking the structure. + """ + out: list[str] = [] + in_str: str | None = None # '"' or '“' + i = 0 + n = len(text) + while i < n: + c = text[i] + if in_str is None: + if c == '"': + in_str = '"' + out.append(c) + elif c == "\u201c": # “ opens a string + in_str = "\u201c" + out.append('"') + else: + out.append(c) + else: + if c == "\\" and i + 1 < n: # keep escape pairs intact + out.append(c) + out.append(text[i + 1]) + i += 2 + continue + if in_str == '"': + # Straight-delimited: curly quotes are just content. + out.append(c) + if c == '"': + in_str = None + else: # curly-delimited string + if c == "\u201d": # ” closes it + in_str = None + out.append('"') + elif c == '"': # raw straight quote inside -> escape + out.append('\\"') + else: + out.append(c) + i += 1 + return "".join(out) + + +def _repair_and_parse(text: str): + """Parse JSON, repairing common LLM quote mistakes if strict parse fails.""" + try: + return json.loads(text) + except json.JSONDecodeError: + pass + data = _try_parse(_normalize_quotes(text)) + if data is not None: + return data + raise ValueError(f"Could not parse jokes JSON even after repair: {text[:200]!r}") + + def _extract_json_array(text: str) -> list[str]: """Pull a JSON array of strings out of a possibly messy LLM response.""" text = text.strip() @@ -28,7 +93,7 @@ def _extract_json_array(text: str) -> list[str]: end = text.rfind("]") if start == -1 or end == -1 or end <= start: raise ValueError(f"No JSON array found in LLM response: {text[:200]!r}") - data = json.loads(text[start : end + 1]) + data = _repair_and_parse(text[start : end + 1]) if not isinstance(data, list): raise ValueError("Parsed JSON is not a list") jokes = [str(j).strip() for j in data if str(j).strip()]