Repair curly-quote JSON from LLM before parsing
halogen-qwen3.8-flash-next sometimes uses typographic quotes (“ ”) as JSON string delimiters, which breaks strict json.loads. Add a scanner that normalizes curly-delimited strings to straight quotes while preserving curly quotes used as content inside straight-quoted strings.
This commit is contained in:
+66
-1
@@ -16,6 +16,71 @@ SYSTEM_PROMPT = (
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _try_parse(candidate: str):
|
||||||
|
try:
|
||||||
|
return json.loads(candidate)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _normalize_quotes(text: str) -> str:
|
||||||
|
"""Rewrite LLM quote soup into valid JSON.
|
||||||
|
|
||||||
|
Handles models that use curly quotes (“ ”) as string DELIMITERS and also
|
||||||
|
as content inside straight-quoted strings. A scanner tracks which quote
|
||||||
|
character opened the current string so content quotes are preserved
|
||||||
|
(escaped) instead of breaking the structure.
|
||||||
|
"""
|
||||||
|
out: list[str] = []
|
||||||
|
in_str: str | None = None # '"' or '“'
|
||||||
|
i = 0
|
||||||
|
n = len(text)
|
||||||
|
while i < n:
|
||||||
|
c = text[i]
|
||||||
|
if in_str is None:
|
||||||
|
if c == '"':
|
||||||
|
in_str = '"'
|
||||||
|
out.append(c)
|
||||||
|
elif c == "\u201c": # “ opens a string
|
||||||
|
in_str = "\u201c"
|
||||||
|
out.append('"')
|
||||||
|
else:
|
||||||
|
out.append(c)
|
||||||
|
else:
|
||||||
|
if c == "\\" and i + 1 < n: # keep escape pairs intact
|
||||||
|
out.append(c)
|
||||||
|
out.append(text[i + 1])
|
||||||
|
i += 2
|
||||||
|
continue
|
||||||
|
if in_str == '"':
|
||||||
|
# Straight-delimited: curly quotes are just content.
|
||||||
|
out.append(c)
|
||||||
|
if c == '"':
|
||||||
|
in_str = None
|
||||||
|
else: # curly-delimited string
|
||||||
|
if c == "\u201d": # ” closes it
|
||||||
|
in_str = None
|
||||||
|
out.append('"')
|
||||||
|
elif c == '"': # raw straight quote inside -> escape
|
||||||
|
out.append('\\"')
|
||||||
|
else:
|
||||||
|
out.append(c)
|
||||||
|
i += 1
|
||||||
|
return "".join(out)
|
||||||
|
|
||||||
|
|
||||||
|
def _repair_and_parse(text: str):
|
||||||
|
"""Parse JSON, repairing common LLM quote mistakes if strict parse fails."""
|
||||||
|
try:
|
||||||
|
return json.loads(text)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
pass
|
||||||
|
data = _try_parse(_normalize_quotes(text))
|
||||||
|
if data is not None:
|
||||||
|
return data
|
||||||
|
raise ValueError(f"Could not parse jokes JSON even after repair: {text[:200]!r}")
|
||||||
|
|
||||||
|
|
||||||
def _extract_json_array(text: str) -> list[str]:
|
def _extract_json_array(text: str) -> list[str]:
|
||||||
"""Pull a JSON array of strings out of a possibly messy LLM response."""
|
"""Pull a JSON array of strings out of a possibly messy LLM response."""
|
||||||
text = text.strip()
|
text = text.strip()
|
||||||
@@ -28,7 +93,7 @@ def _extract_json_array(text: str) -> list[str]:
|
|||||||
end = text.rfind("]")
|
end = text.rfind("]")
|
||||||
if start == -1 or end == -1 or end <= start:
|
if start == -1 or end == -1 or end <= start:
|
||||||
raise ValueError(f"No JSON array found in LLM response: {text[:200]!r}")
|
raise ValueError(f"No JSON array found in LLM response: {text[:200]!r}")
|
||||||
data = json.loads(text[start : end + 1])
|
data = _repair_and_parse(text[start : end + 1])
|
||||||
if not isinstance(data, list):
|
if not isinstance(data, list):
|
||||||
raise ValueError("Parsed JSON is not a list")
|
raise ValueError("Parsed JSON is not a list")
|
||||||
jokes = [str(j).strip() for j in data if str(j).strip()]
|
jokes = [str(j).strip() for j in data if str(j).strip()]
|
||||||
|
|||||||
Reference in New Issue
Block a user