nex_math/backend/seed/ebooks.py

157 lines
5.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

"""电子书种子:把 res/ 下已有的三本书挂到教材上。
幂等:只补不覆盖,管理员在后台重新上传的文件不会被种子冲掉。
"""
from __future__ import annotations
import shutil
from pathlib import Path
from sqlalchemy.orm import Session
from database import REPO_ROOT
from models import Textbook
from services import ebooks as store
RES_ROOT = REPO_ROOT / "res"
# res/ 里的文件名带书目后缀,展示时用干净的名称
PRINCETON = {
"name": "普林斯顿微积分读本(修订版)",
"author": "阿德里安·班纳 (Adrian Banner)",
"publisher": "人民邮电出版社",
"grade": "高中至大学先修",
"description": "面向微积分入门的系统辅导书,覆盖极限、导数、积分与级数,"
"按考点组织,适合自学与考前查漏补缺。",
"resource": "普林斯顿微积分读本(修订版)",
"display": "普林斯顿微积分读本(修订版).epub",
}
# 已有教材 → res/ 中的 PDF(按教材名前缀匹配)
PDF_LINKS = [
("函数和图像", "函数和图像", "函数和图像(盖尔范德中学生数学思维丛书).pdf"),
("三角函数", "三角函数", "三角函数(盖尔范德中学生数学思维丛书).pdf"),
]
def _find_resource(prefix: str, suffix: str) -> Path | None:
"""res/ 下的文件名带作者与来源后缀,只按标题前缀匹配。"""
if not RES_ROOT.is_dir():
return None
for path in sorted(RES_ROOT.iterdir()):
if (
path.is_file()
and path.suffix.lower() == suffix
and path.name.startswith(prefix)
):
return path
return None
# 旧库里的教材:数据库只是引用磁盘上的文件,库一旦丢失,这些文件就成了孤儿。
# 按“大小 + SHA-256”认领已知电子书,重建教材条目并把文件挪进该书自己的目录,
# 管理员就不必重新上传,种子也不会再往 res/ 里复制第二份。
KNOWN_EBOOKS = [
{
"size": 44110741,
"sha256": "f432fa058420f1032a169a91fe2dcefbc95c61a6793f1b0ec5fd1fcc84ee67da",
"display": "柯西-施瓦茨大师课:不等式的艺术.pdf",
"textbook": {
"name": "柯西-施瓦茨大师课:不等式的艺术",
"author": "J. Michael Steele 著;欧阳顺湘 译",
"publisher": "高等教育出版社 · 世界数学精品译丛 7",
"isbn": "9787040278678",
"grade": "高中至大学",
"description": (
"Cambridge 出版的不等式专题大师课:以 Cauchy-Schwarz 不等式为主线,"
"串起凸性、排序、切比雪夫、Young、Hölder、Minkowski 等常用不等式,"
"每章配大量例题与习题,适合竞赛与数学分析进阶训练。"
),
},
},
]
def _referenced_files(db: Session) -> set[Path]:
used: set[Path] = set()
rows = db.query(Textbook).filter(Textbook.ebook_file != "").all()
for row in rows:
used.add((store.book_dir(row.id) / row.ebook_file).resolve())
return used
def recover_lost_ebooks(db: Session) -> list[str]:
"""认领 data/ebooks 下未被引用的已知电子书,返回恢复的教材名。"""
if not store.EBOOK_ROOT.is_dir():
return []
referenced = _referenced_files(db)
recovered: list[str] = []
for book in KNOWN_EBOOKS:
meta = book["textbook"]
if db.query(Textbook).filter(Textbook.name == meta["name"]).first():
continue
for path in sorted(store.EBOOK_ROOT.glob("*/*")):
if not path.is_file() or path.suffix.lower() not in {".pdf", ".epub"}:
continue
if path.stat().st_size != book["size"] or path.resolve() in referenced:
continue
if store.sha256_file(path) != book["sha256"]:
continue
textbook = Textbook(
**meta, position=db.query(Textbook).count()
)
db.add(textbook)
db.commit()
target_dir = store.book_dir(textbook.id)
target_dir.mkdir(parents=True, exist_ok=True)
shutil.move(str(path), str(target_dir / path.name))
store.adopt_file(
db, textbook, path.name, display_name=book["display"]
)
recovered.append(textbook.name)
break
return recovered
def ensure_ebooks(db: Session) -> None:
created = False
textbook = (
db.query(Textbook).filter(Textbook.name == PRINCETON["name"]).first()
)
if textbook is None:
position = (
db.query(Textbook).count()
)
textbook = Textbook(
name=PRINCETON["name"],
author=PRINCETON["author"],
publisher=PRINCETON["publisher"],
grade=PRINCETON["grade"],
description=PRINCETON["description"],
position=position,
)
db.add(textbook)
db.commit()
created = True
if created:
epub = _find_resource(PRINCETON["resource"], ".epub")
if epub is not None and not textbook.ebook_file:
store.attach_local_file(
db, textbook, epub, display_name=PRINCETON["display"]
)
for name_prefix, file_prefix, display in PDF_LINKS:
textbook = (
db.query(Textbook)
.filter(Textbook.name.startswith(name_prefix))
.first()
)
if textbook is None or textbook.ebook_file:
continue
pdf = _find_resource(file_prefix, ".pdf")
if pdf is None:
continue
store.attach_local_file(db, textbook, pdf, display_name=display)
recover_lost_ebooks(db)