"""宽松 JSON 提取:模型常把 LaTeX 的 \\ 写进 JSON,需要修复非法转义。""" from __future__ import annotations import json import re VALID_JSON_ESCAPES = set('"\\/bfnrtu') def _repair_json_escapes(content: str) -> str: """把字符串里未被正确转义的单个反斜杠补成双反斜杠。""" result: list[str] = [] index = 0 in_string = False length = len(content) while index < length: char = content[index] if not in_string: result.append(char) if char == '"': in_string = True index += 1 continue if char != "\\": result.append(char) if char == '"': in_string = False index += 1 continue run_end = index while run_end < length and content[run_end] == "\\": run_end += 1 run_length = run_end - index next_char = content[run_end] if run_end < length else "" if ( run_length % 2 == 1 and next_char and next_char not in VALID_JSON_ESCAPES ): result.append("\\" * (run_length + 1)) else: result.append("\\" * run_length) index = run_end return "".join(result) def extract_json_lax(content: str) -> dict: text = content.strip() if text.startswith("```"): text = re.sub(r"^```(?:json)?\s*", "", text) text = re.sub(r"\s*```$", "", text) end = text.rfind("}") if end <= 0: raise ValueError("模型没有返回 JSON,请重试或调整模型") starts = [ match.start() for match in re.finditer(r'\{"(questions|chapters)":', text) ] starts.extend(match.start() for match in re.finditer(r"\{", text)) starts = sorted(set(starts)) # 限制扫描数量,避免模型把大量思考文本误当成 JSON for start in starts: if start >= end: break payload = text[start : end + 1] try: return json.loads(payload, strict=False) except json.JSONDecodeError: repaired = _repair_json_escapes(payload) try: return json.loads(repaired, strict=False) except json.JSONDecodeError: continue # 没有可解析候选时,给出最后一次错误细节 payload = text[starts[0] : end + 1] if starts else text try: json.loads(payload, strict=False) except json.JSONDecodeError as exc: raise ValueError( "模型返回的 JSON 格式不正确" f"({exc.msg},位置 {exc.pos}),请重试或调整模型" ) from exc raise ValueError("模型返回的 JSON 格式不正确,请重试或调整模型")