nex_math/backend/services/isbn_lookup.py

80 lines
2.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

"""通过 ISBN 查询公开书目信息与章节目录。
优先 Open Library,其次 Google Books。网络不可用或 ISBN 无收录时返回空结果,
调用方继续使用书名 + ISBN 让大模型依据正式出版物内容生成。
"""
from __future__ import annotations
import re
import httpx
def _normalize_isbn(isbn: str) -> str:
return re.sub(r"[^0-9Xx]", "", isbn).upper()
def _clean_toc_item(item: dict) -> str | None:
title = str(item.get("title", "")).strip()
label = str(item.get("label", "")).strip()
if not title and not label:
return None
title = title or label
# 去掉 Open Library 常见的前置编号/页眉,避免把“引言”“1”当标题
title = re.sub(r"^\s*(?:page\s*)?\d+\s*[:.]?\s*", "", title, flags=re.I)
return title.strip() or None
def fetch_isbn_context(isbn: str) -> dict:
"""返回 {source, title, description, chapter_titles}。"""
normalized = _normalize_isbn(isbn)
if not normalized:
return {"source": "", "title": "", "description": "", "chapter_titles": []}
# 1) Open Library 按 ISBN 查书目
try:
with httpx.Client(timeout=6.0, follow_redirects=True) as client:
response = client.get(
f"https://openlibrary.org/isbn/{normalized}.json"
)
if response.status_code == 200:
payload = response.json()
titles: list[str] = []
for item in payload.get("table_of_contents") or []:
if not isinstance(item, dict):
continue
title = _clean_toc_item(item)
if title and title not in titles:
titles.append(title)
return {
"source": "openlibrary",
"title": str(payload.get("title", "")),
"description": "",
"chapter_titles": titles,
}
except httpx.HTTPError:
pass
# 2) Google Books 按 ISBN 查书目
try:
with httpx.Client(timeout=6.0) as client:
response = client.get(
"https://www.googleapis.com/books/v1/volumes",
params={"q": f"isbn:{normalized}", "maxResults": 1},
)
if response.status_code == 200:
data = response.json()
items = data.get("items") or []
if items:
volume = items[0].get("volumeInfo", {})
return {
"source": "google_books",
"title": str(volume.get("title", "")),
"description": str(volume.get("description", ""))[:1500],
"chapter_titles": [],
}
except httpx.HTTPError:
pass
return {"source": "", "title": "", "description": "", "chapter_titles": []}