"""通过 ISBN 查询公开书目信息与章节目录。 优先 Open Library,其次 Google Books。网络不可用或 ISBN 无收录时返回空结果, 调用方继续使用书名 + ISBN 让大模型依据正式出版物内容生成。 """ from __future__ import annotations import re import httpx def _normalize_isbn(isbn: str) -> str: return re.sub(r"[^0-9Xx]", "", isbn).upper() def _clean_toc_item(item: dict) -> str | None: title = str(item.get("title", "")).strip() label = str(item.get("label", "")).strip() if not title and not label: return None title = title or label # 去掉 Open Library 常见的前置编号/页眉,避免把“引言”“1”当标题 title = re.sub(r"^\s*(?:page\s*)?\d+\s*[:.]?\s*", "", title, flags=re.I) return title.strip() or None def fetch_isbn_context(isbn: str) -> dict: """返回 {source, title, description, chapter_titles}。""" normalized = _normalize_isbn(isbn) if not normalized: return {"source": "", "title": "", "description": "", "chapter_titles": []} # 1) Open Library 按 ISBN 查书目 try: with httpx.Client(timeout=6.0, follow_redirects=True) as client: response = client.get( f"https://openlibrary.org/isbn/{normalized}.json" ) if response.status_code == 200: payload = response.json() titles: list[str] = [] for item in payload.get("table_of_contents") or []: if not isinstance(item, dict): continue title = _clean_toc_item(item) if title and title not in titles: titles.append(title) return { "source": "openlibrary", "title": str(payload.get("title", "")), "description": "", "chapter_titles": titles, } except httpx.HTTPError: pass # 2) Google Books 按 ISBN 查书目 try: with httpx.Client(timeout=6.0) as client: response = client.get( "https://www.googleapis.com/books/v1/volumes", params={"q": f"isbn:{normalized}", "maxResults": 1}, ) if response.status_code == 200: data = response.json() items = data.get("items") or [] if items: volume = items[0].get("volumeInfo", {}) return { "source": "google_books", "title": str(volume.get("title", "")), "description": str(volume.get("description", ""))[:1500], "chapter_titles": [], } except httpx.HTTPError: pass return {"source": "", "title": "", "description": "", "chapter_titles": []}