273 lines
11 KiB
Python
273 lines
11 KiB
Python
"""把《定开项目.xlsx》的市场台账导入定开项目表(custom_projects)。
|
||
|
||
字段映射(表格列 → 数据库列):
|
||
|
||
| 表格列 | 数据库列 | 说明 |
|
||
|--------------|-----------------|------|
|
||
| 项目归属 | origin | 新签订单 / 存量项目 / 汇智直签 |
|
||
| 项目名称 | name | 匹配键之一 |
|
||
| 项目ID | project_code | 匹配键(优先) |
|
||
| 下单时间 | sign_date | 日期取 YYYY-MM-DD,像「2024年3月前」这种文本原样保留 |
|
||
| 办事处 | office / region | office 存原样;region 归一化进区域词表,供「区域」筛选 |
|
||
| 行业 | industry | |
|
||
| 市场下单人天 | man_days | |
|
||
| 订单金额 | amount | |
|
||
| 市场责任人 | market_owner | 销售侧人员,与花名册无关,可「张宁/叶镇」这种多人写法 |
|
||
| 项目状态 | project_status | 已验收 / 待开发 / 交付完成 / 交付中 / 退单 |
|
||
| 验收时间 | accept_date | 存的是季度文本(Q1..Q4),不是日期 |
|
||
| 推动验收计划 | accept_plan | Q1..Q4 / 27年 |
|
||
| 风险值 | risk_level | 1.高风险 / 2.低风险 |
|
||
| 备注 | remark | |
|
||
| (脚本补) | update_date | 更新时间,跟随「下单时间」(sign_date),便于按周查看(sign_date 缺失时回退到导入日期) |
|
||
|
||
匹配与更新口径:
|
||
|
||
- 优先按「项目ID」匹配已有行;**有项目ID 时只按 ID 匹配**(表里存在同名不同 ID 的项目,
|
||
退回名称匹配会把它们互相覆盖)。只有项目ID 为空才退回按「项目名称」匹配。
|
||
- 同一行数据在一次运行内只匹配一个已有项目,避免重名互相吞掉。
|
||
- **命中**:只比较上面这些台账字段,有变化才写库,并把 `update_date` 刷成「下单时间」(sign_date);
|
||
没变化则整行不动(`update_date` 保持原值)。sign_date 缺失时回退到导入日期。
|
||
- **未命中**:新增,`update_date` = 下单时间(sign_date),`entry_date` 兜底取下单时间。
|
||
- 不触碰 stage / progress / 计收 / 开票 / 回款等业务字段。
|
||
|
||
用法:
|
||
|
||
python scripts/import_custom_ledger.py --dry-run # 只预演,不写库
|
||
python scripts/import_custom_ledger.py # 实际写入
|
||
python scripts/import_custom_ledger.py --date 2026-09-17 # 指定「更新时间」
|
||
python scripts/import_custom_ledger.py --xlsx "D:/定开项目.xlsx"
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import re
|
||
import sys
|
||
from datetime import date, datetime
|
||
from pathlib import Path
|
||
|
||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||
|
||
import openpyxl # noqa: E402
|
||
|
||
from app.constants import normalize_region # noqa: E402
|
||
from app.db import SessionLocal # noqa: E402
|
||
from app.models import CustomProject # noqa: E402
|
||
|
||
DEFAULT_XLSX = str(Path.home() / "Desktop" / "定开项目.xlsx")
|
||
|
||
# 表格列名 → 数据库字段
|
||
COLMAP = {
|
||
"项目归属": "origin",
|
||
"项目名称": "name",
|
||
"项目ID": "project_code",
|
||
"下单时间": "sign_date",
|
||
"办事处": "office",
|
||
"行业": "industry",
|
||
"市场下单人天": "man_days",
|
||
"订单金额": "amount",
|
||
"市场责任人": "market_owner",
|
||
"项目状态": "project_status",
|
||
"验收时间": "accept_date",
|
||
"推动验收计划": "accept_plan",
|
||
"风险值": "risk_level",
|
||
"备注": "remark",
|
||
}
|
||
|
||
NUMERIC = {"man_days", "amount"}
|
||
# 参与「有没有变化」比较的字段(region 由 office 推导,一并比较)
|
||
TRACKED = list(COLMAP.values()) + ["region"]
|
||
|
||
|
||
def _text(value) -> str | None:
|
||
if value is None:
|
||
return None
|
||
s = str(value).strip()
|
||
return s or None
|
||
|
||
|
||
def _num(value) -> float | None:
|
||
if value is None:
|
||
return None
|
||
s = str(value).replace(",", "").strip()
|
||
if not s:
|
||
return None
|
||
try:
|
||
return float(s)
|
||
except ValueError:
|
||
return None
|
||
|
||
|
||
def _date_text(value) -> str | None:
|
||
"""下单时间:真日期 → YYYY-MM-DD;文本(如「2024年3月前」「2024/」)原样保留。"""
|
||
if isinstance(value, datetime):
|
||
return value.strftime("%Y-%m-%d")
|
||
if isinstance(value, date):
|
||
return value.strftime("%Y-%m-%d")
|
||
return _text(value)
|
||
|
||
|
||
def read_rows(xlsx: str) -> list[dict]:
|
||
wb = openpyxl.load_workbook(xlsx, data_only=True)
|
||
ws = wb.worksheets[0]
|
||
header = [_text(ws.cell(1, c).value) for c in range(1, ws.max_column + 1)]
|
||
|
||
missing = [name for name in COLMAP if name not in header]
|
||
if missing:
|
||
raise SystemExit(f"[x] 表格缺少这些列:{missing}\n 实际表头:{header}")
|
||
|
||
col = {field: header.index(name) + 1 for name, field in COLMAP.items()}
|
||
|
||
rows: list[dict] = []
|
||
for r in range(2, ws.max_row + 1):
|
||
raw = {field: ws.cell(r, c).value for field, c in col.items()}
|
||
if not any(v is not None and str(v).strip() for v in raw.values()):
|
||
continue
|
||
|
||
row: dict = {}
|
||
for field, value in raw.items():
|
||
if field in NUMERIC:
|
||
row[field] = _num(value)
|
||
elif field == "sign_date":
|
||
row[field] = _date_text(value)
|
||
else:
|
||
row[field] = _text(value)
|
||
|
||
if not row.get("name"):
|
||
print(f"[warn] 第 {r} 行没有项目名称,跳过")
|
||
continue
|
||
|
||
row["region"] = normalize_region(row.get("office"))
|
||
# 验收日期(真实日期,区别于 accept_date 季度文本):仅对「已验收」且验收时间为季度文本的项目,
|
||
# 补一个当年季度末占位,使「本年/本月验收」统计有数可算。用户后续在界面上改了真日期不会被覆盖。
|
||
row["accept_date_real"] = None
|
||
if row.get("project_status") == "已验收" and row.get("accept_date"):
|
||
m = re.match(r"Q([1-4])", str(row["accept_date"]))
|
||
if m:
|
||
q = int(m.group(1))
|
||
me = [(3, 31), (6, 30), (9, 30), (12, 31)][q - 1]
|
||
row["accept_date_real"] = f"{date.today().year}-{me[0]:02d}-{me[1]:02d}"
|
||
rows.append(row)
|
||
return rows
|
||
|
||
|
||
def apply_fields(cp: CustomProject, row: dict, only: list[str] | None = None) -> list[tuple[str, object, object]]:
|
||
"""把 row 的字段写到 cp 上,返回 (字段, 旧值, 新值) 的变更列表。"""
|
||
changes: list[tuple[str, object, object]] = []
|
||
for field in (only or list(row)):
|
||
if field not in row:
|
||
continue
|
||
old = getattr(cp, field)
|
||
new = row[field]
|
||
# 数值统一按 float 比,避免 1000000 / 1000000.0 被判成变化
|
||
if field in NUMERIC:
|
||
old_cmp = _num(old)
|
||
new_cmp = _num(new)
|
||
else:
|
||
old_cmp = old
|
||
new_cmp = new
|
||
if old_cmp != new_cmp:
|
||
changes.append((field, old, new))
|
||
setattr(cp, field, new)
|
||
return changes
|
||
|
||
|
||
def main() -> None:
|
||
ap = argparse.ArgumentParser(description="导入定开项目台账")
|
||
ap.add_argument("--xlsx", default=DEFAULT_XLSX, help="台账 Excel 路径")
|
||
ap.add_argument("--date", default=date.today().isoformat(), help="写入的「更新时间」YYYY-MM-DD,默认今天")
|
||
ap.add_argument("--dry-run", action="store_true", help="只预演,不写库")
|
||
args = ap.parse_args()
|
||
|
||
xlsx = Path(args.xlsx)
|
||
if not xlsx.exists():
|
||
raise SystemExit(f"[x] 找不到文件:{xlsx}")
|
||
|
||
rows = read_rows(str(xlsx))
|
||
print(f"==> 读取 {xlsx}")
|
||
print(f" 有效数据 {len(rows)} 行;「更新时间」口径 = {args.date}")
|
||
|
||
db = SessionLocal()
|
||
try:
|
||
by_code: dict[str, CustomProject] = {}
|
||
by_name: dict[str, CustomProject] = {}
|
||
for cp in db.query(CustomProject).all():
|
||
if cp.project_code:
|
||
by_code.setdefault(cp.project_code, cp)
|
||
by_name.setdefault((cp.name or "").strip(), cp)
|
||
|
||
matched_ids: set[int] = set()
|
||
inserted = updated = unchanged = 0
|
||
change_log: list[str] = []
|
||
|
||
for row in rows:
|
||
cp = None
|
||
code = row.get("project_code")
|
||
if code:
|
||
# 有项目ID 时只认 ID:表里存在同名不同 ID 的两个项目(如甘肃省水利厅二期),
|
||
# 退回按名称匹配会把它们互相覆盖。
|
||
cand = by_code.get(code)
|
||
if cand is not None and cand.id not in matched_ids:
|
||
cp = cand
|
||
else:
|
||
cand = by_name.get(row["name"])
|
||
if cand is not None and cand.id not in matched_ids:
|
||
cp = cand
|
||
|
||
if cp is None:
|
||
inserted += 1
|
||
change_log.append(f" [新增] {row['name']}({row.get('office') or '—'} / {row.get('project_status') or '—'})")
|
||
if args.dry_run:
|
||
continue
|
||
cp = CustomProject(name=row["name"], source="import")
|
||
cp.entry_date = row.get("sign_date") or args.date
|
||
cp.update_date = row.get("sign_date") or args.date
|
||
apply_fields(cp, row, [f for f in row if f != "name"])
|
||
db.add(cp)
|
||
db.flush() # 拿到 id,供 matched_ids
|
||
if cp.project_code:
|
||
by_code.setdefault(cp.project_code, cp)
|
||
by_name.setdefault(row["name"], cp)
|
||
matched_ids.add(cp.id)
|
||
continue
|
||
|
||
changes = apply_fields(cp, row, TRACKED)
|
||
matched_ids.add(cp.id)
|
||
# 验收日期:仅当库里还为空时补占位(用户已在界面填了真日期则不覆盖)
|
||
if cp.accept_date_real is None and row.get("accept_date_real"):
|
||
cp.accept_date_real = row["accept_date_real"]
|
||
changes.append(("accept_date_real", None, cp.accept_date_real))
|
||
if not changes:
|
||
unchanged += 1
|
||
continue
|
||
|
||
updated += 1
|
||
if not args.dry_run:
|
||
cp.update_date = row.get("sign_date") or args.date
|
||
detail = ";".join(f"{f} {old!r}→{new!r}" for f, old, new in changes)
|
||
change_log.append(f" [更新] {row['name']}:{detail}")
|
||
|
||
if args.dry_run:
|
||
db.rollback()
|
||
else:
|
||
db.commit()
|
||
|
||
print(f"\n{'[dry-run] 预演结果' if args.dry_run else '[ok] 导入完成'}")
|
||
print(f" 新增 {inserted} 行 / 更新 {updated} 行 / 无变化 {unchanged} 行")
|
||
|
||
if change_log:
|
||
print("\n明细:")
|
||
for line in change_log[:40]:
|
||
print(line)
|
||
if len(change_log) > 40:
|
||
print(f" ... 另有 {len(change_log) - 40} 条")
|
||
|
||
if not args.dry_run:
|
||
total = db.query(CustomProject).count()
|
||
print(f"\n 当前 custom_projects 共 {total} 行")
|
||
finally:
|
||
db.close()
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|