"""电子书种子:把 res/ 下已有的三本书挂到教材上。 幂等:只补不覆盖,管理员在后台重新上传的文件不会被种子冲掉。 """ from __future__ import annotations import shutil from pathlib import Path from sqlalchemy.orm import Session from database import REPO_ROOT from models import Textbook from services import ebooks as store RES_ROOT = REPO_ROOT / "res" # res/ 里的文件名带书目后缀,展示时用干净的名称 PRINCETON = { "name": "普林斯顿微积分读本(修订版)", "author": "阿德里安·班纳 (Adrian Banner)", "publisher": "人民邮电出版社", "grade": "高中至大学先修", "description": "面向微积分入门的系统辅导书,覆盖极限、导数、积分与级数," "按考点组织,适合自学与考前查漏补缺。", "resource": "普林斯顿微积分读本(修订版)", "display": "普林斯顿微积分读本(修订版).epub", } # 已有教材 → res/ 中的 PDF(按教材名前缀匹配) PDF_LINKS = [ ("函数和图像", "函数和图像", "函数和图像(盖尔范德中学生数学思维丛书).pdf"), ("三角函数", "三角函数", "三角函数(盖尔范德中学生数学思维丛书).pdf"), ] def _find_resource(prefix: str, suffix: str) -> Path | None: """res/ 下的文件名带作者与来源后缀,只按标题前缀匹配。""" if not RES_ROOT.is_dir(): return None for path in sorted(RES_ROOT.iterdir()): if ( path.is_file() and path.suffix.lower() == suffix and path.name.startswith(prefix) ): return path return None # 旧库里的教材:数据库只是引用磁盘上的文件,库一旦丢失,这些文件就成了孤儿。 # 按“大小 + SHA-256”认领已知电子书,重建教材条目并把文件挪进该书自己的目录, # 管理员就不必重新上传,种子也不会再往 res/ 里复制第二份。 KNOWN_EBOOKS = [ { "size": 44110741, "sha256": "f432fa058420f1032a169a91fe2dcefbc95c61a6793f1b0ec5fd1fcc84ee67da", "display": "柯西-施瓦茨大师课:不等式的艺术.pdf", "textbook": { "name": "柯西-施瓦茨大师课:不等式的艺术", "author": "J. Michael Steele 著;欧阳顺湘 译", "publisher": "高等教育出版社 · 世界数学精品译丛 7", "isbn": "9787040278678", "grade": "高中至大学", "description": ( "Cambridge 出版的不等式专题大师课:以 Cauchy-Schwarz 不等式为主线," "串起凸性、排序、切比雪夫、Young、Hölder、Minkowski 等常用不等式," "每章配大量例题与习题,适合竞赛与数学分析进阶训练。" ), }, }, ] def _referenced_files(db: Session) -> set[Path]: used: set[Path] = set() rows = db.query(Textbook).filter(Textbook.ebook_file != "").all() for row in rows: used.add((store.book_dir(row.id) / row.ebook_file).resolve()) return used def recover_lost_ebooks(db: Session) -> list[str]: """认领 data/ebooks 下未被引用的已知电子书,返回恢复的教材名。""" if not store.EBOOK_ROOT.is_dir(): return [] referenced = _referenced_files(db) recovered: list[str] = [] for book in KNOWN_EBOOKS: meta = book["textbook"] if db.query(Textbook).filter(Textbook.name == meta["name"]).first(): continue for path in sorted(store.EBOOK_ROOT.glob("*/*")): if not path.is_file() or path.suffix.lower() not in {".pdf", ".epub"}: continue if path.stat().st_size != book["size"] or path.resolve() in referenced: continue if store.sha256_file(path) != book["sha256"]: continue textbook = Textbook( **meta, position=db.query(Textbook).count() ) db.add(textbook) db.commit() target_dir = store.book_dir(textbook.id) target_dir.mkdir(parents=True, exist_ok=True) shutil.move(str(path), str(target_dir / path.name)) store.adopt_file( db, textbook, path.name, display_name=book["display"] ) recovered.append(textbook.name) break return recovered def ensure_ebooks(db: Session) -> None: created = False textbook = ( db.query(Textbook).filter(Textbook.name == PRINCETON["name"]).first() ) if textbook is None: position = ( db.query(Textbook).count() ) textbook = Textbook( name=PRINCETON["name"], author=PRINCETON["author"], publisher=PRINCETON["publisher"], grade=PRINCETON["grade"], description=PRINCETON["description"], position=position, ) db.add(textbook) db.commit() created = True if created: epub = _find_resource(PRINCETON["resource"], ".epub") if epub is not None and not textbook.ebook_file: store.attach_local_file( db, textbook, epub, display_name=PRINCETON["display"] ) for name_prefix, file_prefix, display in PDF_LINKS: textbook = ( db.query(Textbook) .filter(Textbook.name.startswith(name_prefix)) .first() ) if textbook is None or textbook.ebook_file: continue pdf = _find_resource(file_prefix, ".pdf") if pdf is None: continue store.attach_local_file(db, textbook, pdf, display_name=display) recover_lost_ebooks(db)