"""把《定开项目.xlsx》的市场台账导入定开项目表(custom_projects)。 字段映射(表格列 → 数据库列): | 表格列 | 数据库列 | 说明 | |--------------|-----------------|------| | 项目归属 | origin | 华智下单 / 存量项目 / 汇智直签 / 提前交付(「新签订单」自动映射为「华智下单」) | | 项目名称 | name | 匹配键之一 | | 项目ID | project_code | 匹配键(优先) | | 下单时间 | sign_date | 日期取 YYYY-MM-DD,像「2024年3月前」这种文本原样保留 | | 办事处 | office / region | office 存原样;region 归一化进区域词表,供「区域」筛选 | | 行业 | industry | | | 市场下单人天 | man_days | | | 订单金额 | amount | | | 市场责任人 | market_owner | 销售侧人员,与花名册无关,可「张宁/叶镇」这种多人写法 | | 项目状态 | project_status | 已验收 / 待开发 / 交付完成 / 进行中 / 退单(「交付中」自动映射为「进行中」) | | 验收时间 | accept_date | 存的是季度文本(Q1..Q4),不是日期 | | 推动验收计划 | accept_plan | Q1..Q4 / 27年 | | 风险值 | risk_level | 1.高风险 / 2.低风险 | | 备注 | remark | | | (脚本补) | update_date | 更新时间,跟随「下单时间」(sign_date),便于按周查看(sign_date 缺失时回退到导入日期) | 匹配与更新口径: - 优先按「项目ID」匹配已有行;**有项目ID 时只按 ID 匹配**(表里存在同名不同 ID 的项目, 退回名称匹配会把它们互相覆盖)。只有项目ID 为空才退回按「项目名称」匹配。 - 同一行数据在一次运行内只匹配一个已有项目,避免重名互相吞掉。 - **命中**:只比较上面这些台账字段,有变化才写库,并把 `update_date` 刷成「下单时间」(sign_date); 没变化则整行不动(`update_date` 保持原值)。sign_date 缺失时回退到导入日期。 - **未命中**:新增,`update_date` = 下单时间(sign_date),`entry_date` 兜底取下单时间。 - 不触碰 stage / progress / 计收 / 开票 / 回款等业务字段。 用法: python scripts/import_custom_ledger.py --dry-run # 只预演,不写库 python scripts/import_custom_ledger.py # 实际写入 python scripts/import_custom_ledger.py --date 2026-09-17 # 指定「更新时间」 python scripts/import_custom_ledger.py --xlsx "D:/定开项目.xlsx" """ from __future__ import annotations import argparse import re import sys from datetime import date, datetime from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) import openpyxl # noqa: E402 from app.constants import normalize_region # noqa: E402 from app.db import SessionLocal # noqa: E402 from app.models import CustomProject # noqa: E402 DEFAULT_XLSX = str(Path.home() / "Desktop" / "定开项目.xlsx") # 表格列名 → 数据库字段 COLMAP = { "项目归属": "origin", "项目名称": "name", "项目ID": "project_code", "下单时间": "sign_date", "办事处": "office", "行业": "industry", "市场下单人天": "man_days", "订单金额": "amount", "市场责任人": "market_owner", "项目状态": "project_status", "验收时间": "accept_date", "推动验收计划": "accept_plan", "风险值": "risk_level", "备注": "remark", } NUMERIC = {"man_days", "amount"} # 参与「有没有变化」比较的字段(region 由 office 推导,一并比较) TRACKED = list(COLMAP.values()) + ["region"] def _text(value) -> str | None: if value is None: return None s = str(value).strip() return s or None def _num(value) -> float | None: if value is None: return None s = str(value).replace(",", "").strip() if not s: return None try: return float(s) except ValueError: return None def _date_text(value) -> str | None: """下单时间:真日期 → YYYY-MM-DD;文本(如「2024年3月前」「2024/」)原样保留。""" if isinstance(value, datetime): return value.strftime("%Y-%m-%d") if isinstance(value, date): return value.strftime("%Y-%m-%d") return _text(value) def read_rows(xlsx: str) -> list[dict]: wb = openpyxl.load_workbook(xlsx, data_only=True) ws = wb.worksheets[0] header = [_text(ws.cell(1, c).value) for c in range(1, ws.max_column + 1)] missing = [name for name in COLMAP if name not in header] if missing: raise SystemExit(f"[x] 表格缺少这些列:{missing}\n 实际表头:{header}") col = {field: header.index(name) + 1 for name, field in COLMAP.items()} rows: list[dict] = [] for r in range(2, ws.max_row + 1): raw = {field: ws.cell(r, c).value for field, c in col.items()} if not any(v is not None and str(v).strip() for v in raw.values()): continue row: dict = {} for field, value in raw.items(): if field in NUMERIC: row[field] = _num(value) elif field == "sign_date": row[field] = _date_text(value) else: row[field] = _text(value) if not row.get("name"): print(f"[warn] 第 {r} 行没有项目名称,跳过") continue row["region"] = normalize_region(row.get("office")) # 历史取值改名(2026-09-20):项目归属 新签订单 → 华智下单;项目状态 交付中 → 进行中 if row.get("origin") == "新签订单": row["origin"] = "华智下单" if row.get("project_status") == "交付中": row["project_status"] = "进行中" # 验收日期(真实日期,区别于 accept_date 季度文本):仅对「已验收」且验收时间为季度文本的项目, # 补一个当年季度末占位,使「本年/本月验收」统计有数可算。用户后续在界面上改了真日期不会被覆盖。 row["accept_date_real"] = None if row.get("project_status") == "已验收" and row.get("accept_date"): m = re.match(r"Q([1-4])", str(row["accept_date"])) if m: q = int(m.group(1)) me = [(3, 31), (6, 30), (9, 30), (12, 31)][q - 1] row["accept_date_real"] = f"{date.today().year}-{me[0]:02d}-{me[1]:02d}" rows.append(row) return rows def apply_fields(cp: CustomProject, row: dict, only: list[str] | None = None) -> list[tuple[str, object, object]]: """把 row 的字段写到 cp 上,返回 (字段, 旧值, 新值) 的变更列表。""" changes: list[tuple[str, object, object]] = [] for field in (only or list(row)): if field not in row: continue old = getattr(cp, field) new = row[field] # 数值统一按 float 比,避免 1000000 / 1000000.0 被判成变化 if field in NUMERIC: old_cmp = _num(old) new_cmp = _num(new) else: old_cmp = old new_cmp = new if old_cmp != new_cmp: changes.append((field, old, new)) setattr(cp, field, new) return changes def main() -> None: ap = argparse.ArgumentParser(description="导入定开项目台账") ap.add_argument("--xlsx", default=DEFAULT_XLSX, help="台账 Excel 路径") ap.add_argument("--date", default=date.today().isoformat(), help="写入的「更新时间」YYYY-MM-DD,默认今天") ap.add_argument("--dry-run", action="store_true", help="只预演,不写库") args = ap.parse_args() xlsx = Path(args.xlsx) if not xlsx.exists(): raise SystemExit(f"[x] 找不到文件:{xlsx}") rows = read_rows(str(xlsx)) print(f"==> 读取 {xlsx}") print(f" 有效数据 {len(rows)} 行;「更新时间」口径 = {args.date}") db = SessionLocal() try: by_code: dict[str, CustomProject] = {} by_name: dict[str, CustomProject] = {} for cp in db.query(CustomProject).all(): if cp.project_code: by_code.setdefault(cp.project_code, cp) by_name.setdefault((cp.name or "").strip(), cp) matched_ids: set[int] = set() inserted = updated = unchanged = 0 change_log: list[str] = [] for row in rows: cp = None code = row.get("project_code") if code: # 有项目ID 时只认 ID:表里存在同名不同 ID 的两个项目(如甘肃省水利厅二期), # 退回按名称匹配会把它们互相覆盖。 cand = by_code.get(code) if cand is not None and cand.id not in matched_ids: cp = cand else: cand = by_name.get(row["name"]) if cand is not None and cand.id not in matched_ids: cp = cand if cp is None: inserted += 1 change_log.append(f" [新增] {row['name']}({row.get('office') or '—'} / {row.get('project_status') or '—'})") if args.dry_run: continue cp = CustomProject(name=row["name"], source="import") cp.entry_date = row.get("sign_date") or args.date cp.update_date = row.get("sign_date") or args.date apply_fields(cp, row, [f for f in row if f != "name"]) db.add(cp) db.flush() # 拿到 id,供 matched_ids if cp.project_code: by_code.setdefault(cp.project_code, cp) by_name.setdefault(row["name"], cp) matched_ids.add(cp.id) continue changes = apply_fields(cp, row, TRACKED) matched_ids.add(cp.id) # 验收日期:仅当库里还为空时补占位(用户已在界面填了真日期则不覆盖) if cp.accept_date_real is None and row.get("accept_date_real"): cp.accept_date_real = row["accept_date_real"] changes.append(("accept_date_real", None, cp.accept_date_real)) if not changes: unchanged += 1 continue updated += 1 if not args.dry_run: cp.update_date = row.get("sign_date") or args.date detail = ";".join(f"{f} {old!r}→{new!r}" for f, old, new in changes) change_log.append(f" [更新] {row['name']}:{detail}") if args.dry_run: db.rollback() else: db.commit() print(f"\n{'[dry-run] 预演结果' if args.dry_run else '[ok] 导入完成'}") print(f" 新增 {inserted} 行 / 更新 {updated} 行 / 无变化 {unchanged} 行") if change_log: print("\n明细:") for line in change_log[:40]: print(line) if len(change_log) > 40: print(f" ... 另有 {len(change_log) - 40} 条") if not args.dry_run: total = db.query(CustomProject).count() print(f"\n 当前 custom_projects 共 {total} 行") finally: db.close() if __name__ == "__main__": main()