80 lines
2.8 KiB
Python
80 lines
2.8 KiB
Python
"""通过 ISBN 查询公开书目信息与章节目录。
|
||
|
||
优先 Open Library,其次 Google Books。网络不可用或 ISBN 无收录时返回空结果,
|
||
调用方继续使用书名 + ISBN 让大模型依据正式出版物内容生成。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
|
||
import httpx
|
||
|
||
|
||
def _normalize_isbn(isbn: str) -> str:
|
||
return re.sub(r"[^0-9Xx]", "", isbn).upper()
|
||
|
||
|
||
def _clean_toc_item(item: dict) -> str | None:
|
||
title = str(item.get("title", "")).strip()
|
||
label = str(item.get("label", "")).strip()
|
||
if not title and not label:
|
||
return None
|
||
title = title or label
|
||
# 去掉 Open Library 常见的前置编号/页眉,避免把“引言”“1”当标题
|
||
title = re.sub(r"^\s*(?:page\s*)?\d+\s*[:.]?\s*", "", title, flags=re.I)
|
||
return title.strip() or None
|
||
|
||
|
||
def fetch_isbn_context(isbn: str) -> dict:
|
||
"""返回 {source, title, description, chapter_titles}。"""
|
||
normalized = _normalize_isbn(isbn)
|
||
if not normalized:
|
||
return {"source": "", "title": "", "description": "", "chapter_titles": []}
|
||
|
||
# 1) Open Library 按 ISBN 查书目
|
||
try:
|
||
with httpx.Client(timeout=6.0, follow_redirects=True) as client:
|
||
response = client.get(
|
||
f"https://openlibrary.org/isbn/{normalized}.json"
|
||
)
|
||
if response.status_code == 200:
|
||
payload = response.json()
|
||
titles: list[str] = []
|
||
for item in payload.get("table_of_contents") or []:
|
||
if not isinstance(item, dict):
|
||
continue
|
||
title = _clean_toc_item(item)
|
||
if title and title not in titles:
|
||
titles.append(title)
|
||
return {
|
||
"source": "openlibrary",
|
||
"title": str(payload.get("title", "")),
|
||
"description": "",
|
||
"chapter_titles": titles,
|
||
}
|
||
except httpx.HTTPError:
|
||
pass
|
||
|
||
# 2) Google Books 按 ISBN 查书目
|
||
try:
|
||
with httpx.Client(timeout=6.0) as client:
|
||
response = client.get(
|
||
"https://www.googleapis.com/books/v1/volumes",
|
||
params={"q": f"isbn:{normalized}", "maxResults": 1},
|
||
)
|
||
if response.status_code == 200:
|
||
data = response.json()
|
||
items = data.get("items") or []
|
||
if items:
|
||
volume = items[0].get("volumeInfo", {})
|
||
return {
|
||
"source": "google_books",
|
||
"title": str(volume.get("title", "")),
|
||
"description": str(volume.get("description", ""))[:1500],
|
||
"chapter_titles": [],
|
||
}
|
||
except httpx.HTTPError:
|
||
pass
|
||
|
||
return {"source": "", "title": "", "description": "", "chapter_titles": []}
|