nex_math/backend/services/catalog_generator.py

270 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

"""通过大模型按教材/课程生成章节目录。"""
from __future__ import annotations
from models import LlmSetting, Textbook
from services import llm_client
SYSTEM_PROMPT = (
"你是一位资深数学教材编审与课程设计专家,熟悉盖尔范德"
"《函数和图像》《代数》《三角函数》等中学生数学思维丛书。"
)
def build_chapters_prompt(
*,
textbook: Textbook,
count: int,
existing_names: list[str],
instructions: str,
fixed_titles: list[str] | None = None,
external_context: dict | None = None,
graph_context: list[str] | None = None,
) -> str:
extra = f"\n额外要求:{instructions}" if instructions else ""
existing = "、".join(existing_names) if existing_names else "(暂无章节)"
isbn = textbook.isbn or ""
external = external_context or {}
external_title = external.get("title") or ""
external_description = external.get("description") or ""
lookup_status = (
"ISBN 公开目录未收录,请结合书名 + ISBN + 作者/出版社"
"确认正式出版物后作答"
if (isbn and not external)
else "已通过 ISBN 查询到公开书目信息"
)
knowledge_hint = (
"、".join(graph_context) if graph_context else "(按该书内容从知识图谱选择)"
)
graph_line = f"\n知识图谱可用节点:{knowledge_hint}"
if fixed_titles:
numbered = "\n".join(
f"{index}. {title}" for index, title in enumerate(fixed_titles, start=1)
)
lookup_note = "已根据 ISBN 查询到该书正式目录"
if external_title:
lookup_note += f"({external_title})"
knowledge_hint = (
"、".join(graph_context) if graph_context else "(按该书内容选择)"
)
return f"""通过 ISBN 查询到《{textbook.name}》的正式章节目录。
{lookup_note}。请为下列章节编写符合数学教材风格的内容摘要。
正式目录章节:
{numbered}
系统知识图谱可用节点:{knowledge_hint}
要求:
1. 章节名称必须与上面给定名称完全一致,不得改名、不得新增或删减;
2. 每章 summary 用 1-2 句中文说明该章核心主题、知识能力与学习任务;
3. 每章从系统知识图谱可用节点中按语义选择 1-5 个作为 knowledge_names;
4. 不要输出多余文字,只输出 JSON:
{{"chapters":[{{"name":"给定章节名","summary":"摘要","knowledge_names":["节点名"]}}]}}"""
return f"""请像检索真实书目一样,根据以下信息还原《{textbook.name}》中文正式版的完整章节目录。
书目信息:
- 书名:{textbook.name}
- ISBN:{isbn or "(未填写)"}
- 作者/翻译/出版:{textbook.author} {textbook.publisher}
- 版本/学段:{textbook.edition} {textbook.grade}
- 简介:{textbook.description or "请结合标题判断"}
- 本书目前已存在章节:{existing or "(暂无)"}
{f"- 外部书目补充:{external_description}" if external_description else ""}
{graph_line}
要求:
1. 只做“目录还原”,不要编撰新章节,不要套用通用课程大纲;
2. 请检索/回忆该 ISBN 对应版本的中文正式目录,原样输出目录条目
(引言、第 X 章、附录/习题解答等都要保留原名,例如“第 8 章 独立求解的问题”);
3. 输出全部正式目录;若确有把握有限,也可只输出能确认的部分,不要补造;
4. 为每个目录条目写 1 句 summary(前言/解答等可写其用途);
5. 每个教学章节从“知识图谱可用节点”中选择 1-5 个作为 knowledge_names;
前言、习题解答等非教学条目 knowledge_names 输出空数组;
6. 不要输出多余文字,只输出 JSON:
{{"chapters":[{{"name":"目录原样条目","summary":"摘要","knowledge_names":["节点名"]}}]}}
"""
def _extract_json(content: str) -> dict:
from services.json_utils import extract_json_lax
return extract_json_lax(content)
def generate_chapters(
*,
setting: LlmSetting,
textbook: Textbook,
count: int,
existing_names: list[str],
instructions: str,
fixed_titles: list[str] | None = None,
external_context: dict | None = None,
graph_context: list[str] | None = None,
) -> list[dict]:
use_titles = fixed_titles or []
if use_titles:
count = len(use_titles)
prompt = build_chapters_prompt(
textbook=textbook,
count=count,
existing_names=existing_names,
instructions=instructions,
fixed_titles=use_titles or None,
external_context=external_context,
graph_context=graph_context,
)
payload = {}
last_error = ""
for attempt in range(3):
try:
content = llm_client.chat_completion(
base_url=setting.base_url,
api_key=setting.api_key,
model=setting.model,
temperature=0.3,
max_tokens=setting.max_tokens or 5000,
timeout=200.0,
messages=[
{"role": "system", "content": SYSTEM_PROMPT},
{"role": "user", "content": prompt},
],
)
except ValueError as exc:
last_error = str(exc)
continue
try:
payload = _extract_json(content)
break
except ValueError as exc:
last_error = str(exc)
else:
raise ValueError(
f"模型连续 3 次未返回合法 JSON,最后一次:{last_error}"
)
raw_chapters = payload.get("chapters")
if not isinstance(raw_chapters, list) or not raw_chapters:
raise ValueError("模型返回内容中没有 chapters 列表")
result: list[dict] = []
fixed_set = set(use_titles)
for raw in raw_chapters[:count]:
if not isinstance(raw, dict):
raise ValueError("模型返回的章节格式不正确")
name = str(raw.get("name", "")).strip()
summary = str(raw.get("summary", "")).strip()
if not name:
raise ValueError("模型返回的章节名称为空")
if fixed_set and name not in fixed_set:
raise ValueError(
f"模型返回的章节“{name}”不在 ISBN 查询到的正式目录中,请重试"
)
raw_knowledge = raw.get("knowledge_names") or []
if not isinstance(raw_knowledge, list):
raw_knowledge = []
result.append(
{
"name": name[:128],
"summary": summary[:1000],
"knowledge_names": [
str(tag).strip()[:64] for tag in raw_knowledge
],
}
)
return result
def organize_reference_chapters(
*,
setting: LlmSetting,
textbook: Textbook,
reference_text: str,
existing_names: list[str],
graph_context: list[str],
) -> list[dict]:
"""根据用户粘贴的目录素材整理出章节,并为每个章节从知识图谱选节点。"""
prompt = f"""你是一名教材目录整理助手。下面是从用户或其他 AI 处得到的
《{textbook.name}》目录原始资料,请整理成结构化章节列表。
教材信息:
- ISBN:{textbook.isbn or "未填写"}
- 作者/出版:{textbook.author} {textbook.publisher}
- 已存在章节:{"、".join(existing_names) if existing_names else "(暂无)"}
知识图谱可用节点(只能从这里选 knowledge_names):
{"、".join(graph_context) if graph_context else "(暂无)"}
用户提供的目录素材:
==========================
{reference_text}
==========================
要求:
1. 只整理素材中真实出现的章节目录,不得新增、不得凭印象补造;
2. 保留正式名称(含“第 X 章”、引言、前言、习题解答等),去除多余格式符号;
3. 无法判断为目录项的文字不要输出;
4. 每章 summary 用 1 句中文说明;
5. 教学章节从“知识图谱可用节点”中按语义选择 1-5 个 knowledge_names,
非教学条目为空数组;不得自造图谱节点;
6. 只输出 JSON:{{"chapters":[{{"name":"目录项","summary":"摘要","knowledge_names":["节点"]}}]}}
"""
payload = {}
last_error = ""
for attempt in range(3):
try:
content = llm_client.chat_completion(
base_url=setting.base_url,
api_key=setting.api_key,
model=setting.model,
temperature=0.1,
max_tokens=setting.max_tokens or 5000,
timeout=200.0,
messages=[
{
"role": "system",
"content": (
"你只负责整理用户提供的真实目录素材,不编造内容。"
"JSON 必须完整闭合。"
),
},
{"role": "user", "content": prompt},
],
)
except ValueError as exc:
last_error = str(exc)
continue
try:
payload = _extract_json(content)
break
except ValueError as exc:
last_error = str(exc)
else:
raise ValueError(
f"模型连续 3 次未返回合法 JSON,最后一次:{last_error}"
)
raw_chapters = payload.get("chapters")
if not isinstance(raw_chapters, list) or not raw_chapters:
raise ValueError("整理结果中没有 chapters 列表")
result: list[dict] = []
for raw in raw_chapters:
if not isinstance(raw, dict):
continue
name = str(raw.get("name", "")).strip()
if not name:
continue
raw_knowledge = raw.get("knowledge_names") or []
if not isinstance(raw_knowledge, list):
raw_knowledge = []
result.append(
{
"name": name[:128],
"summary": str(raw.get("summary", "")).strip()[:1000],
"knowledge_names": [
str(tag).strip()[:64] for tag in raw_knowledge
],
}
)
return result