unis_manager/scripts/import_custom_ledger.py

273 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters!

This file contains ambiguous Unicode characters that may be confused with others in your current locale. If your use case is intentional and legitimate, you can safely ignore this warning. Use the Escape button to highlight these characters.

"""把《定开项目.xlsx》的市场台账导入定开项目表(custom_projects)。
字段映射(表格列 → 数据库列):
| 表格列 | 数据库列 | 说明 |
|--------------|-----------------|------|
| 项目归属 | origin | 新签订单 / 存量项目 / 汇智直签 |
| 项目名称 | name | 匹配键之一 |
| 项目ID | project_code | 匹配键(优先) |
| 下单时间 | sign_date | 日期取 YYYY-MM-DD,像「2024年3月前」这种文本原样保留 |
| 办事处 | office / region | office 存原样;region 归一化进区域词表,供「区域」筛选 |
| 行业 | industry | |
| 市场下单人天 | man_days | |
| 订单金额 | amount | |
| 市场责任人 | market_owner | 销售侧人员,与花名册无关,可「张宁/叶镇」这种多人写法 |
| 项目状态 | project_status | 已验收 / 待开发 / 交付完成 / 交付中 / 退单 |
| 验收时间 | accept_date | 存的是季度文本(Q1..Q4),不是日期 |
| 推动验收计划 | accept_plan | Q1..Q4 / 27年 |
| 风险值 | risk_level | 1.高风险 / 2.低风险 |
| 备注 | remark | |
| (脚本补) | update_date | 更新时间,跟随「下单时间」(sign_date),便于按周查看(sign_date 缺失时回退到导入日期) |
匹配与更新口径:
- 优先按「项目ID」匹配已有行;**有项目ID 时只按 ID 匹配**(表里存在同名不同 ID 的项目,
退回名称匹配会把它们互相覆盖)。只有项目ID 为空才退回按「项目名称」匹配。
- 同一行数据在一次运行内只匹配一个已有项目,避免重名互相吞掉。
- **命中**:只比较上面这些台账字段,有变化才写库,并把 `update_date` 刷成「下单时间」(sign_date);
没变化则整行不动(`update_date` 保持原值)。sign_date 缺失时回退到导入日期。
- **未命中**:新增,`update_date` = 下单时间(sign_date),`entry_date` 兜底取下单时间。
- 不触碰 stage / progress / 计收 / 开票 / 回款等业务字段。
用法:
python scripts/import_custom_ledger.py --dry-run # 只预演,不写库
python scripts/import_custom_ledger.py # 实际写入
python scripts/import_custom_ledger.py --date 2026-09-17 # 指定「更新时间」
python scripts/import_custom_ledger.py --xlsx "D:/定开项目.xlsx"
"""
from __future__ import annotations
import argparse
import re
import sys
from datetime import date, datetime
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
import openpyxl # noqa: E402
from app.constants import normalize_region # noqa: E402
from app.db import SessionLocal # noqa: E402
from app.models import CustomProject # noqa: E402
DEFAULT_XLSX = str(Path.home() / "Desktop" / "定开项目.xlsx")
# 表格列名 → 数据库字段
COLMAP = {
"项目归属": "origin",
"项目名称": "name",
"项目ID": "project_code",
"下单时间": "sign_date",
"办事处": "office",
"行业": "industry",
"市场下单人天": "man_days",
"订单金额": "amount",
"市场责任人": "market_owner",
"项目状态": "project_status",
"验收时间": "accept_date",
"推动验收计划": "accept_plan",
"风险值": "risk_level",
"备注": "remark",
}
NUMERIC = {"man_days", "amount"}
# 参与「有没有变化」比较的字段(region 由 office 推导,一并比较)
TRACKED = list(COLMAP.values()) + ["region"]
def _text(value) -> str | None:
if value is None:
return None
s = str(value).strip()
return s or None
def _num(value) -> float | None:
if value is None:
return None
s = str(value).replace(",", "").strip()
if not s:
return None
try:
return float(s)
except ValueError:
return None
def _date_text(value) -> str | None:
"""下单时间:真日期 → YYYY-MM-DD;文本(如「2024年3月前」「2024/」)原样保留。"""
if isinstance(value, datetime):
return value.strftime("%Y-%m-%d")
if isinstance(value, date):
return value.strftime("%Y-%m-%d")
return _text(value)
def read_rows(xlsx: str) -> list[dict]:
wb = openpyxl.load_workbook(xlsx, data_only=True)
ws = wb.worksheets[0]
header = [_text(ws.cell(1, c).value) for c in range(1, ws.max_column + 1)]
missing = [name for name in COLMAP if name not in header]
if missing:
raise SystemExit(f"[x] 表格缺少这些列:{missing}\n 实际表头:{header}")
col = {field: header.index(name) + 1 for name, field in COLMAP.items()}
rows: list[dict] = []
for r in range(2, ws.max_row + 1):
raw = {field: ws.cell(r, c).value for field, c in col.items()}
if not any(v is not None and str(v).strip() for v in raw.values()):
continue
row: dict = {}
for field, value in raw.items():
if field in NUMERIC:
row[field] = _num(value)
elif field == "sign_date":
row[field] = _date_text(value)
else:
row[field] = _text(value)
if not row.get("name"):
print(f"[warn] 第 {r} 行没有项目名称,跳过")
continue
row["region"] = normalize_region(row.get("office"))
# 验收日期(真实日期,区别于 accept_date 季度文本):仅对「已验收」且验收时间为季度文本的项目,
# 补一个当年季度末占位,使「本年/本月验收」统计有数可算。用户后续在界面上改了真日期不会被覆盖。
row["accept_date_real"] = None
if row.get("project_status") == "已验收" and row.get("accept_date"):
m = re.match(r"Q([1-4])", str(row["accept_date"]))
if m:
q = int(m.group(1))
me = [(3, 31), (6, 30), (9, 30), (12, 31)][q - 1]
row["accept_date_real"] = f"{date.today().year}-{me[0]:02d}-{me[1]:02d}"
rows.append(row)
return rows
def apply_fields(cp: CustomProject, row: dict, only: list[str] | None = None) -> list[tuple[str, object, object]]:
"""把 row 的字段写到 cp 上,返回 (字段, 旧值, 新值) 的变更列表。"""
changes: list[tuple[str, object, object]] = []
for field in (only or list(row)):
if field not in row:
continue
old = getattr(cp, field)
new = row[field]
# 数值统一按 float 比,避免 1000000 / 1000000.0 被判成变化
if field in NUMERIC:
old_cmp = _num(old)
new_cmp = _num(new)
else:
old_cmp = old
new_cmp = new
if old_cmp != new_cmp:
changes.append((field, old, new))
setattr(cp, field, new)
return changes
def main() -> None:
ap = argparse.ArgumentParser(description="导入定开项目台账")
ap.add_argument("--xlsx", default=DEFAULT_XLSX, help="台账 Excel 路径")
ap.add_argument("--date", default=date.today().isoformat(), help="写入的「更新时间」YYYY-MM-DD,默认今天")
ap.add_argument("--dry-run", action="store_true", help="只预演,不写库")
args = ap.parse_args()
xlsx = Path(args.xlsx)
if not xlsx.exists():
raise SystemExit(f"[x] 找不到文件:{xlsx}")
rows = read_rows(str(xlsx))
print(f"==> 读取 {xlsx}")
print(f" 有效数据 {len(rows)} 行;「更新时间」口径 = {args.date}")
db = SessionLocal()
try:
by_code: dict[str, CustomProject] = {}
by_name: dict[str, CustomProject] = {}
for cp in db.query(CustomProject).all():
if cp.project_code:
by_code.setdefault(cp.project_code, cp)
by_name.setdefault((cp.name or "").strip(), cp)
matched_ids: set[int] = set()
inserted = updated = unchanged = 0
change_log: list[str] = []
for row in rows:
cp = None
code = row.get("project_code")
if code:
# 有项目ID 时只认 ID:表里存在同名不同 ID 的两个项目(如甘肃省水利厅二期),
# 退回按名称匹配会把它们互相覆盖。
cand = by_code.get(code)
if cand is not None and cand.id not in matched_ids:
cp = cand
else:
cand = by_name.get(row["name"])
if cand is not None and cand.id not in matched_ids:
cp = cand
if cp is None:
inserted += 1
change_log.append(f" [新增] {row['name']}({row.get('office') or '—'} / {row.get('project_status') or '—'})")
if args.dry_run:
continue
cp = CustomProject(name=row["name"], source="import")
cp.entry_date = row.get("sign_date") or args.date
cp.update_date = row.get("sign_date") or args.date
apply_fields(cp, row, [f for f in row if f != "name"])
db.add(cp)
db.flush() # 拿到 id,供 matched_ids
if cp.project_code:
by_code.setdefault(cp.project_code, cp)
by_name.setdefault(row["name"], cp)
matched_ids.add(cp.id)
continue
changes = apply_fields(cp, row, TRACKED)
matched_ids.add(cp.id)
# 验收日期:仅当库里还为空时补占位(用户已在界面填了真日期则不覆盖)
if cp.accept_date_real is None and row.get("accept_date_real"):
cp.accept_date_real = row["accept_date_real"]
changes.append(("accept_date_real", None, cp.accept_date_real))
if not changes:
unchanged += 1
continue
updated += 1
if not args.dry_run:
cp.update_date = row.get("sign_date") or args.date
detail = ";".join(f"{f} {old!r}→{new!r}" for f, old, new in changes)
change_log.append(f" [更新] {row['name']}:{detail}")
if args.dry_run:
db.rollback()
else:
db.commit()
print(f"\n{'[dry-run] 预演结果' if args.dry_run else '[ok] 导入完成'}")
print(f" 新增 {inserted} 行 / 更新 {updated} 行 / 无变化 {unchanged} 行")
if change_log:
print("\n明细:")
for line in change_log[:40]:
print(line)
if len(change_log) > 40:
print(f" ... 另有 {len(change_log) - 40} 条")
if not args.dry_run:
total = db.query(CustomProject).count()
print(f"\n 当前 custom_projects 共 {total} 行")
finally:
db.close()
if __name__ == "__main__":
main()