Repair curly-quote JSON from LLM before parsing
halogen-qwen3.8-flash-next sometimes uses typographic quotes (“ ”) as JSON string delimiters, which breaks strict json.loads. Add a scanner that normalizes curly-delimited strings to straight quotes while preserving curly quotes used as content inside straight-quoted strings.
This commit is contained in:
+66
-1
@@ -16,6 +16,71 @@ SYSTEM_PROMPT = (
|
||||
)
|
||||
|
||||
|
||||
def _try_parse(candidate: str):
|
||||
try:
|
||||
return json.loads(candidate)
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
|
||||
|
||||
def _normalize_quotes(text: str) -> str:
|
||||
"""Rewrite LLM quote soup into valid JSON.
|
||||
|
||||
Handles models that use curly quotes (“ ”) as string DELIMITERS and also
|
||||
as content inside straight-quoted strings. A scanner tracks which quote
|
||||
character opened the current string so content quotes are preserved
|
||||
(escaped) instead of breaking the structure.
|
||||
"""
|
||||
out: list[str] = []
|
||||
in_str: str | None = None # '"' or '“'
|
||||
i = 0
|
||||
n = len(text)
|
||||
while i < n:
|
||||
c = text[i]
|
||||
if in_str is None:
|
||||
if c == '"':
|
||||
in_str = '"'
|
||||
out.append(c)
|
||||
elif c == "\u201c": # “ opens a string
|
||||
in_str = "\u201c"
|
||||
out.append('"')
|
||||
else:
|
||||
out.append(c)
|
||||
else:
|
||||
if c == "\\" and i + 1 < n: # keep escape pairs intact
|
||||
out.append(c)
|
||||
out.append(text[i + 1])
|
||||
i += 2
|
||||
continue
|
||||
if in_str == '"':
|
||||
# Straight-delimited: curly quotes are just content.
|
||||
out.append(c)
|
||||
if c == '"':
|
||||
in_str = None
|
||||
else: # curly-delimited string
|
||||
if c == "\u201d": # ” closes it
|
||||
in_str = None
|
||||
out.append('"')
|
||||
elif c == '"': # raw straight quote inside -> escape
|
||||
out.append('\\"')
|
||||
else:
|
||||
out.append(c)
|
||||
i += 1
|
||||
return "".join(out)
|
||||
|
||||
|
||||
def _repair_and_parse(text: str):
|
||||
"""Parse JSON, repairing common LLM quote mistakes if strict parse fails."""
|
||||
try:
|
||||
return json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
data = _try_parse(_normalize_quotes(text))
|
||||
if data is not None:
|
||||
return data
|
||||
raise ValueError(f"Could not parse jokes JSON even after repair: {text[:200]!r}")
|
||||
|
||||
|
||||
def _extract_json_array(text: str) -> list[str]:
|
||||
"""Pull a JSON array of strings out of a possibly messy LLM response."""
|
||||
text = text.strip()
|
||||
@@ -28,7 +93,7 @@ def _extract_json_array(text: str) -> list[str]:
|
||||
end = text.rfind("]")
|
||||
if start == -1 or end == -1 or end <= start:
|
||||
raise ValueError(f"No JSON array found in LLM response: {text[:200]!r}")
|
||||
data = json.loads(text[start : end + 1])
|
||||
data = _repair_and_parse(text[start : end + 1])
|
||||
if not isinstance(data, list):
|
||||
raise ValueError("Parsed JSON is not a list")
|
||||
jokes = [str(j).strip() for j in data if str(j).strip()]
|
||||
|
||||
Reference in New Issue
Block a user