unis_manager/scripts/fix_period_split.py

327 lines
13 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

#!/usr/bin/env python
"""
修复周期错位:把「8月第4周」里混入的「9月第1周」内容拆出来,并把现有
「9月第1周」的数据整体后移到「9月第2周」。
背景
----
原始 Excel《周工作情况总结》最后一次导入时,表尾只有一个「8月第4周」标签,
其内容里连着贴了两段(8月第4周 + 9月第1周)。导入脚本按 (项目, 子任务, 周期)
去重,同周期内重复出现就 append 到同一条记录,于是两周内容被揉在一起;
9月第1周因此没有独立数据。
现在 Excel 已拆成两个标签块,本脚本按 Excel 现状把数据库对齐:
8月第4周 (id=49) → 只保留 8月第4周内容
9月第1周 (id=50) → 从 8月第4周里拆出的 9月第1周内容
9月第2周 (新建) → 原来误放在 9月第1周 的数据
用法
----
python scripts/fix_period_split.py --dry-run # 预演
python scripts/fix_period_split.py # 执行
"""
from __future__ import annotations
import argparse
import shutil
import sys
from datetime import datetime
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(ROOT))
import openpyxl # noqa: E402
from app.constants import parse_period_label # noqa: E402
from app.db import SessionLocal # noqa: E402
from app.models import ( # noqa: E402
CustomProject,
CustomUpdate,
Period,
Project,
Task,
Update,
)
XLSX = "/Users/jiliu/WorkSpace/定开管理/软件开发部管理工作执行表.xlsx"
# --------------------------------------------------------------------------- #
# 读取 Excel:定位表尾最后两个周期标签块
# --------------------------------------------------------------------------- #
def load_tail_blocks(xlsx: str) -> tuple[dict, dict]:
"""返回 (block_prev, block_last),每个是 {(项目, 子任务): 内容} 的有序字典。"""
wb = openpyxl.load_workbook(xlsx, data_only=True, read_only=True)
ws = wb["周工作情况总结"]
rows = list(ws.iter_rows(min_row=3, values_only=True))
# 1) 找出所有周期标签所在的行号
marks: list[tuple[int, str]] = []
for i, r in enumerate(rows, start=3):
cells = list(r[:6]) + [None] * 6
v = cells[1]
if v is None:
continue
s = str(v).strip()
if s and s != "时间":
marks.append((i, s))
if len(marks) < 2:
raise SystemExit("Excel 里周期标签块不足 2 个")
(start_last, label_last), (start_prev, label_prev) = marks[-1], marks[-2]
end_prev = start_last - 1
print(f"Excel 表尾两个块:『{label_prev}』行 {start_prev}-{end_prev}"
f" / 『{label_last}』行 {start_last}-{len(rows) + 2}")
def parse(lo: int, hi: int) -> dict[tuple[str, str], str]:
out: dict[tuple[str, str], str] = {}
proj: str | None = None
for i in range(lo, hi + 1):
cells = list(rows[i - 3][:6]) + [None] * 6
work = str(cells[2]).strip() if cells[2] is not None else ""
desc = str(cells[3]).strip() if cells[3] is not None else ""
done = str(cells[4]).strip() if cells[4] is not None else ""
if work:
proj = work
if not (proj and desc and done):
continue
out.setdefault((proj, desc), done)
return out
return parse(start_prev, end_prev), parse(start_last, len(rows) + 2)
def guess_progress(text: str):
from app.ai_parser import _guess_progress
return _guess_progress(text)
def guess_status(text: str):
from app.ai_parser import _guess_status
return _guess_status(text)
# --------------------------------------------------------------------------- #
def main() -> None:
ap = argparse.ArgumentParser()
ap.add_argument("--dry-run", action="store_true", help="只打印计划,不写库")
ap.add_argument("--no-backup", action="store_true")
args = ap.parse_args()
db_path = ROOT / "data" / "board.db"
if not args.dry_run and not args.no_backup:
bak = ROOT / "data" / f"board_before_period_fix_{datetime.now():%Y%m%d_%H%M%S}.bak"
shutil.copy2(db_path, bak)
print(f"已备份 → {bak.name}\n")
prev_block, last_block = load_tail_blocks(XLSX)
print(f"\n『上一块』{(len(prev_block))} 条,『最后一块』{len(last_block)} 条")
db = SessionLocal()
try:
p_prev = db.query(Period).filter(Period.sort_key == 20260804).one_or_none()
p_last = db.query(Period).filter(Period.sort_key == 20260901).one_or_none()
if not p_prev or not p_last:
raise SystemExit("找不到 2026年8月第4周 / 9月第1周 周期,请确认先执行过年份平移")
print(f"目标周期:8月第4周 id={p_prev.id} / 9月第1周 id={p_last.id}")
# ---------- Step 1: 新建 9月第2周 ----------
p_next = db.query(Period).filter(Period.sort_key == 20260902).one_or_none()
if not p_next:
p_next = Period(
label="9月第2周", kind="week", year=2026, month=9, week=2,
sort_key=20260902,
)
db.add(p_next)
db.flush()
print(f"\n[1] 新建周期:2026年9月第2周 id={p_next.id}")
else:
print(f"\n[1] 复用已有周期:2026年9月第2周 id={p_next.id}")
# ---------- Step 2: 现有 9月第1周 数据 → 9月第2周 ----------
moved_u = db.query(Update).filter(Update.period_id == p_last.id).all()
moved_c = db.query(CustomUpdate).filter(CustomUpdate.period_id == p_last.id).all()
print(f"[2] 9月第1周 现有数据 → 9月第2周:常规 {len(moved_u)} 条 / 定制 {len(moved_c)} 条")
for u in moved_u:
print(f" 常规#{u.id} {u.project.name}"
f"{'/' + u.task.name if u.task else ''}")
for c in moved_c:
print(f" 定制#{c.id} {c.project.name}")
# ---------- Step 3: 拆分 8月第4周 ----------
print(f"\n[3] 拆分 8月第4周(把属于 9月第1周 的内容移到 id={p_last.id})")
cur_updates = {
(u.project.name, u.task.name if u.task else ""): u
for u in db.query(Update).filter(Update.period_id == p_prev.id).all()
}
plan_move, plan_new, plan_trim = [], [], []
for (proj, task_name), content_last in last_block.items():
if proj == "定制开发":
continue # 定制项目走 Step 4,不进常规表
key = (proj, task_name)
u = cur_updates.get(key)
if u is None:
plan_new.append((proj, task_name, content_last))
continue
content_prev = prev_block.get(key)
if content_prev is None:
# 该子任务只属于 9月第1周 → 整条搬走
plan_move.append((u, content_last))
else:
cur = (u.content or "").strip()
if cur == f"{content_prev}\n{content_last}".strip():
plan_trim.append((u, content_prev, content_last))
elif cur == content_last:
plan_move.append((u, content_last))
elif cur.endswith(content_last):
plan_trim.append((u, cur[: -len(content_last)].rstrip("\n").strip(),
content_last))
else:
print(f" !! 无法自动拆分 常规#{u.id} {proj}/{task_name}")
print(f" 现有内容:{cur[:90]!r}")
print(f" 期望尾部:{content_last[:90]!r}")
print(f" 整条搬走 {len(plan_move)} 条 / 拆出后新建 {len(plan_trim) + len(plan_new)} 条")
for u, _ in plan_move:
print(f" 搬走 常规#{u.id} {u.project.name}/{u.task.name}")
for u, keep, out in plan_trim:
print(f" 拆分 常规#{u.id} {u.project.name}/{u.task.name}"
f" (保留{len(keep)}字 / 拆出{len(out)}字)")
for proj, task_name, content in plan_new:
print(f" 新建 {proj}/{task_name}({len(content)}字)")
# ---------- Step 4: 定制记录 ----------
print(f"\n[4] 定制记录归类")
cur_custom = db.query(CustomUpdate).filter(
CustomUpdate.period_id == p_prev.id).all()
last_names = {d for (_p, d) in last_block}
custom_move = []
for c in cur_custom:
if c.project.name in last_names or any(
c.project.name and c.project.name in d for d in last_names
):
custom_move.append(c)
for c in cur_custom:
tag = "→ 9月第1周" if c in custom_move else "留在 8月第4周"
print(f" 定制#{c.id} {c.project.name[:26]:28s} {tag}")
# ---------- Step 5: 清理空的多余周期 ----------
junk = [
p for p in db.query(Period).all()
if p.sort_key not in (p_prev.sort_key, p_last.sort_key, p_next.sort_key)
and p.sort_key > 20260804
and db.query(Update).filter(Update.period_id == p.id).count() == 0
and db.query(CustomUpdate).filter(CustomUpdate.period_id == p.id).count() == 0
]
print(f"\n[5] 清理无记录的异常周期:{[f'{p.year}年{p.month}月第{p.week}周' for p in junk] or '无'}")
if args.dry_run:
print("\n[dry-run] 未写入数据库")
db.rollback()
return
# ================= 执行 =================
for u in moved_u:
u.period_id = p_next.id
for c in moved_c:
c.period_id = p_next.id
for u, content in plan_move:
u.period_id = p_last.id
if (u.content or "").strip() != content.strip():
u.content = content
for u, keep, out in plan_trim:
u.content = keep
u.progress = guess_progress(keep)
u.status = guess_status(keep)
db.add(Update(
project_id=u.project_id, task_id=u.task_id, period_id=p_last.id,
content=out, progress=guess_progress(out),
status=guess_status(out), source=u.source or "import",
))
for proj, task_name, content in plan_new:
p = db.query(Project).filter(Project.name == proj).one_or_none()
if not p:
print(f" !! 项目不存在,跳过:{proj}")
continue
t = db.query(Task).filter(
Task.project_id == p.id, Task.name == task_name[:255]).one_or_none()
if not t:
t = Task(project_id=p.id, name=task_name[:255], sort_order=999)
db.add(t)
db.flush()
db.add(Update(
project_id=p.id, task_id=t.id, period_id=p_last.id,
content=content, progress=guess_progress(content),
status=guess_status(content), source="import",
))
for c in custom_move:
c.period_id = p_last.id
for p in junk:
db.delete(p)
db.commit()
# ---------- 重算受影响项目的进度 / 状态 ----------
print("\n[6] 重算受影响项目的最新进度与状态")
touched = {u.project_id for u in moved_u} | {u.project_id for u, _ in plan_move}
for u, _, _ in plan_trim:
touched.add(u.project_id)
for pid in sorted(touched):
p = db.get(Project, pid)
ups = (db.query(Update).filter(Update.project_id == pid)
.order_by(Update.period_id.desc(), Update.id.desc()).limit(30).all())
prs = [x.progress for x in ups if x.progress is not None]
if prs:
p.progress = max(prs)
if ups:
p.status = ups[0].status
if p.status == "done" and (p.progress or 0) < 100:
p.status = "in_progress"
print(f" {p.name[:22]:24s} 进度={p.progress} 状态={p.status}")
# ---------- 重算受影响定制项目的入库时间 ----------
print("\n[7] 重算受影响定制项目的入库时间")
from app.constants import period_first_day
cid_touched = {c.custom_project_id for c in moved_c} | {
c.custom_project_id for c in custom_move}
for cid in sorted(x for x in cid_touched if x):
cp = db.get(CustomProject, cid)
ups = (db.query(CustomUpdate)
.filter(CustomUpdate.custom_project_id == cid)
.join(Period, CustomUpdate.period_id == Period.id)
.order_by(Period.sort_key).all())
if not ups:
continue
per = min(ups, key=lambda x: x.period.sort_key).period
d = period_first_day(per.year, per.month, per.week)
old = cp.entry_date
if cp.entry_date is None or d < cp.entry_date:
cp.entry_date = d
print(f" {cp.name[:24]:26s} {old} → {cp.entry_date}")
db.commit()
print("\n✅ 完成")
# ---------- 校验 ----------
print("\n[校验] 各周期记录数:")
for p in db.query(Period).order_by(Period.sort_key).all()[-5:]:
n1 = db.query(Update).filter(Update.period_id == p.id).count()
n2 = db.query(CustomUpdate).filter(CustomUpdate.period_id == p.id).count()
print(f" {p.year}年{p.month}月第{p.week}周 常规 {n1:3d} / 定制 {n2:3d}")
finally:
db.close()
if __name__ == "__main__":
main()