#!/usr/bin/env python """ 修复周期错位:把「8月第4周」里混入的「9月第1周」内容拆出来,并把现有 「9月第1周」的数据整体后移到「9月第2周」。 背景 ---- 原始 Excel《周工作情况总结》最后一次导入时,表尾只有一个「8月第4周」标签, 其内容里连着贴了两段(8月第4周 + 9月第1周)。导入脚本按 (项目, 子任务, 周期) 去重,同周期内重复出现就 append 到同一条记录,于是两周内容被揉在一起; 9月第1周因此没有独立数据。 现在 Excel 已拆成两个标签块,本脚本按 Excel 现状把数据库对齐: 8月第4周 (id=49) → 只保留 8月第4周内容 9月第1周 (id=50) → 从 8月第4周里拆出的 9月第1周内容 9月第2周 (新建) → 原来误放在 9月第1周 的数据 用法 ---- python scripts/fix_period_split.py --dry-run # 预演 python scripts/fix_period_split.py # 执行 """ from __future__ import annotations import argparse import shutil import sys from datetime import datetime from pathlib import Path ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT)) import openpyxl # noqa: E402 from app.constants import parse_period_label # noqa: E402 from app.db import SessionLocal # noqa: E402 from app.models import ( # noqa: E402 CustomProject, CustomUpdate, Period, Project, Task, Update, ) XLSX = "/Users/jiliu/WorkSpace/定开管理/软件开发部管理工作执行表.xlsx" # --------------------------------------------------------------------------- # # 读取 Excel:定位表尾最后两个周期标签块 # --------------------------------------------------------------------------- # def load_tail_blocks(xlsx: str) -> tuple[dict, dict]: """返回 (block_prev, block_last),每个是 {(项目, 子任务): 内容} 的有序字典。""" wb = openpyxl.load_workbook(xlsx, data_only=True, read_only=True) ws = wb["周工作情况总结"] rows = list(ws.iter_rows(min_row=3, values_only=True)) # 1) 找出所有周期标签所在的行号 marks: list[tuple[int, str]] = [] for i, r in enumerate(rows, start=3): cells = list(r[:6]) + [None] * 6 v = cells[1] if v is None: continue s = str(v).strip() if s and s != "时间": marks.append((i, s)) if len(marks) < 2: raise SystemExit("Excel 里周期标签块不足 2 个") (start_last, label_last), (start_prev, label_prev) = marks[-1], marks[-2] end_prev = start_last - 1 print(f"Excel 表尾两个块:『{label_prev}』行 {start_prev}-{end_prev}" f" / 『{label_last}』行 {start_last}-{len(rows) + 2}") def parse(lo: int, hi: int) -> dict[tuple[str, str], str]: out: dict[tuple[str, str], str] = {} proj: str | None = None for i in range(lo, hi + 1): cells = list(rows[i - 3][:6]) + [None] * 6 work = str(cells[2]).strip() if cells[2] is not None else "" desc = str(cells[3]).strip() if cells[3] is not None else "" done = str(cells[4]).strip() if cells[4] is not None else "" if work: proj = work if not (proj and desc and done): continue out.setdefault((proj, desc), done) return out return parse(start_prev, end_prev), parse(start_last, len(rows) + 2) def guess_progress(text: str): from app.ai_parser import _guess_progress return _guess_progress(text) def guess_status(text: str): from app.ai_parser import _guess_status return _guess_status(text) # --------------------------------------------------------------------------- # def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("--dry-run", action="store_true", help="只打印计划,不写库") ap.add_argument("--no-backup", action="store_true") args = ap.parse_args() db_path = ROOT / "data" / "board.db" if not args.dry_run and not args.no_backup: bak = ROOT / "data" / f"board_before_period_fix_{datetime.now():%Y%m%d_%H%M%S}.bak" shutil.copy2(db_path, bak) print(f"已备份 → {bak.name}\n") prev_block, last_block = load_tail_blocks(XLSX) print(f"\n『上一块』{(len(prev_block))} 条,『最后一块』{len(last_block)} 条") db = SessionLocal() try: p_prev = db.query(Period).filter(Period.sort_key == 20260804).one_or_none() p_last = db.query(Period).filter(Period.sort_key == 20260901).one_or_none() if not p_prev or not p_last: raise SystemExit("找不到 2026年8月第4周 / 9月第1周 周期,请确认先执行过年份平移") print(f"目标周期:8月第4周 id={p_prev.id} / 9月第1周 id={p_last.id}") # ---------- Step 1: 新建 9月第2周 ---------- p_next = db.query(Period).filter(Period.sort_key == 20260902).one_or_none() if not p_next: p_next = Period( label="9月第2周", kind="week", year=2026, month=9, week=2, sort_key=20260902, ) db.add(p_next) db.flush() print(f"\n[1] 新建周期:2026年9月第2周 id={p_next.id}") else: print(f"\n[1] 复用已有周期:2026年9月第2周 id={p_next.id}") # ---------- Step 2: 现有 9月第1周 数据 → 9月第2周 ---------- moved_u = db.query(Update).filter(Update.period_id == p_last.id).all() moved_c = db.query(CustomUpdate).filter(CustomUpdate.period_id == p_last.id).all() print(f"[2] 9月第1周 现有数据 → 9月第2周:常规 {len(moved_u)} 条 / 定制 {len(moved_c)} 条") for u in moved_u: print(f" 常规#{u.id} {u.project.name}" f"{'/' + u.task.name if u.task else ''}") for c in moved_c: print(f" 定制#{c.id} {c.project.name}") # ---------- Step 3: 拆分 8月第4周 ---------- print(f"\n[3] 拆分 8月第4周(把属于 9月第1周 的内容移到 id={p_last.id})") cur_updates = { (u.project.name, u.task.name if u.task else ""): u for u in db.query(Update).filter(Update.period_id == p_prev.id).all() } plan_move, plan_new, plan_trim = [], [], [] for (proj, task_name), content_last in last_block.items(): if proj == "定制开发": continue # 定制项目走 Step 4,不进常规表 key = (proj, task_name) u = cur_updates.get(key) if u is None: plan_new.append((proj, task_name, content_last)) continue content_prev = prev_block.get(key) if content_prev is None: # 该子任务只属于 9月第1周 → 整条搬走 plan_move.append((u, content_last)) else: cur = (u.content or "").strip() if cur == f"{content_prev}\n{content_last}".strip(): plan_trim.append((u, content_prev, content_last)) elif cur == content_last: plan_move.append((u, content_last)) elif cur.endswith(content_last): plan_trim.append((u, cur[: -len(content_last)].rstrip("\n").strip(), content_last)) else: print(f" !! 无法自动拆分 常规#{u.id} {proj}/{task_name}") print(f" 现有内容:{cur[:90]!r}") print(f" 期望尾部:{content_last[:90]!r}") print(f" 整条搬走 {len(plan_move)} 条 / 拆出后新建 {len(plan_trim) + len(plan_new)} 条") for u, _ in plan_move: print(f" 搬走 常规#{u.id} {u.project.name}/{u.task.name}") for u, keep, out in plan_trim: print(f" 拆分 常规#{u.id} {u.project.name}/{u.task.name}" f" (保留{len(keep)}字 / 拆出{len(out)}字)") for proj, task_name, content in plan_new: print(f" 新建 {proj}/{task_name}({len(content)}字)") # ---------- Step 4: 定制记录 ---------- print(f"\n[4] 定制记录归类") cur_custom = db.query(CustomUpdate).filter( CustomUpdate.period_id == p_prev.id).all() last_names = {d for (_p, d) in last_block} custom_move = [] for c in cur_custom: if c.project.name in last_names or any( c.project.name and c.project.name in d for d in last_names ): custom_move.append(c) for c in cur_custom: tag = "→ 9月第1周" if c in custom_move else "留在 8月第4周" print(f" 定制#{c.id} {c.project.name[:26]:28s} {tag}") # ---------- Step 5: 清理空的多余周期 ---------- junk = [ p for p in db.query(Period).all() if p.sort_key not in (p_prev.sort_key, p_last.sort_key, p_next.sort_key) and p.sort_key > 20260804 and db.query(Update).filter(Update.period_id == p.id).count() == 0 and db.query(CustomUpdate).filter(CustomUpdate.period_id == p.id).count() == 0 ] print(f"\n[5] 清理无记录的异常周期:{[f'{p.year}年{p.month}月第{p.week}周' for p in junk] or '无'}") if args.dry_run: print("\n[dry-run] 未写入数据库") db.rollback() return # ================= 执行 ================= for u in moved_u: u.period_id = p_next.id for c in moved_c: c.period_id = p_next.id for u, content in plan_move: u.period_id = p_last.id if (u.content or "").strip() != content.strip(): u.content = content for u, keep, out in plan_trim: u.content = keep u.progress = guess_progress(keep) u.status = guess_status(keep) db.add(Update( project_id=u.project_id, task_id=u.task_id, period_id=p_last.id, content=out, progress=guess_progress(out), status=guess_status(out), source=u.source or "import", )) for proj, task_name, content in plan_new: p = db.query(Project).filter(Project.name == proj).one_or_none() if not p: print(f" !! 项目不存在,跳过:{proj}") continue t = db.query(Task).filter( Task.project_id == p.id, Task.name == task_name[:255]).one_or_none() if not t: t = Task(project_id=p.id, name=task_name[:255], sort_order=999) db.add(t) db.flush() db.add(Update( project_id=p.id, task_id=t.id, period_id=p_last.id, content=content, progress=guess_progress(content), status=guess_status(content), source="import", )) for c in custom_move: c.period_id = p_last.id for p in junk: db.delete(p) db.commit() # ---------- 重算受影响项目的进度 / 状态 ---------- print("\n[6] 重算受影响项目的最新进度与状态") touched = {u.project_id for u in moved_u} | {u.project_id for u, _ in plan_move} for u, _, _ in plan_trim: touched.add(u.project_id) for pid in sorted(touched): p = db.get(Project, pid) ups = (db.query(Update).filter(Update.project_id == pid) .order_by(Update.period_id.desc(), Update.id.desc()).limit(30).all()) prs = [x.progress for x in ups if x.progress is not None] if prs: p.progress = max(prs) if ups: p.status = ups[0].status if p.status == "done" and (p.progress or 0) < 100: p.status = "in_progress" print(f" {p.name[:22]:24s} 进度={p.progress} 状态={p.status}") # ---------- 重算受影响定制项目的入库时间 ---------- print("\n[7] 重算受影响定制项目的入库时间") from app.constants import period_first_day cid_touched = {c.custom_project_id for c in moved_c} | { c.custom_project_id for c in custom_move} for cid in sorted(x for x in cid_touched if x): cp = db.get(CustomProject, cid) ups = (db.query(CustomUpdate) .filter(CustomUpdate.custom_project_id == cid) .join(Period, CustomUpdate.period_id == Period.id) .order_by(Period.sort_key).all()) if not ups: continue per = min(ups, key=lambda x: x.period.sort_key).period d = period_first_day(per.year, per.month, per.week) old = cp.entry_date if cp.entry_date is None or d < cp.entry_date: cp.entry_date = d print(f" {cp.name[:24]:26s} {old} → {cp.entry_date}") db.commit() print("\n✅ 完成") # ---------- 校验 ---------- print("\n[校验] 各周期记录数:") for p in db.query(Period).order_by(Period.sort_key).all()[-5:]: n1 = db.query(Update).filter(Update.period_id == p.id).count() n2 = db.query(CustomUpdate).filter(CustomUpdate.period_id == p.id).count() print(f" {p.year}年{p.month}月第{p.week}周 常规 {n1:3d} / 定制 {n2:3d}") finally: db.close() if __name__ == "__main__": main()