Files
inquiry_robot/tools/seed_quote_templates.py
T

285 lines
9.8 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
首批海陆空报价模板导入辅助脚本。
职责:
1) 扫描仓库「海陆空报价单模板」目录(忽略 ~$ 临时文件与嵌套重复)
2) 按文件名判定 bizType / category / mode(auto|manual) / 关键词
3) 用 openpyxl 做标签粗映射,写出 seed 清单 JSON(供人工核对)
4) 若提供 --upload-url 与 token,则逐个 POST 到主账 /inquiry/quoteTemplate/upload
不写 MySQL;不发明价格。生产真相仍是主账版本表。
用法示例:
python tools/seed_quote_templates.py --dry-run
python tools/seed_quote_templates.py --upload-url http://127.0.0.1:8180/jeecgboot --token <X-Access-Token>
"""
from __future__ import annotations
import argparse
import json
import re
import sys
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parents[1]
TEMPLATE_DIR = ROOT / "海陆空报价单模板"
CONTRACT = ROOT / "contracts" / "quote-template-mapping-v1.json"
OUT_JSON = ROOT / "deploy" / "seed" / "quote_templates_seed_manifest_v1.json"
def classify(name: str) -> dict[str, Any]:
"""按文件名归类业务/分类/模式。"""
n = name
mode = "auto"
if any(k in n for k in ("人工报价", "双清", "保税仓", "仓储")):
mode = "manual"
if "海运" in n:
return {
"bizType": "海运",
"category": "海运报价单模板",
"keywords": ["海运", "整柜", "FCL"],
"mode": mode,
}
if "空运到门" in n:
return {
"bizType": "空运",
"category": "空运到门报价单模板",
"keywords": ["空运", "到门"],
"mode": mode,
}
if "空运" in n:
return {
"bizType": "空运",
"category": "空运到港报价单模板",
"keywords": ["空运", "到港"],
"mode": mode,
}
if "中港" in n:
return {
"bizType": "陆运",
"category": "中港",
"keywords": ["中港", "陆运"],
"mode": mode,
}
if "拼车" in n:
return {
"bizType": "陆运",
"category": "国内拼车",
"keywords": ["拼车", "零担"],
"mode": mode,
}
if "国内" in n and "整车" in n:
return {
"bizType": "陆运",
"category": "国内整车",
"keywords": ["整车", "国内"],
"mode": mode,
}
if "跨境整车" in n:
return {
"bizType": "陆运",
"category": "跨境整车",
"keywords": ["跨境", "整车"],
"mode": mode,
}
if "跨境集拼" in n or "中越" in n or "集拼" in n:
cat = "双清或中越" if ("中越" in n or "双清" in n) else "跨境集拼"
return {
"bizType": "陆运",
"category": cat,
"keywords": ["集拼", "跨境"] if "集拼" in n else ["中越"],
"mode": mode,
}
if "凭祥" in n or "仓储" in n:
return {
"bizType": "陆运",
"category": "凭祥保税仓&普仓仓储",
"keywords": ["仓储", "凭祥"],
"mode": "manual",
}
if "双清" in n:
return {
"bizType": "陆运",
"category": "双清或中越",
"keywords": ["双清"],
"mode": "manual",
}
return {
"bizType": "陆运",
"category": "国内拼车",
"keywords": [Path(name).stem[:20]],
"mode": mode,
}
def unique_xlsx_files() -> list[Path]:
files = [p for p in TEMPLATE_DIR.rglob("*.xlsx") if not p.name.startswith("~$")]
by_name: dict[str, Path] = {}
for p in files:
# 优先更短路径(去重嵌套目录)
if p.name not in by_name or len(str(p)) < len(str(by_name[p.name])):
by_name[p.name] = p
return [by_name[k] for k in sorted(by_name)]
def rough_mapping_points(path: Path, label_dict: dict[str, str]) -> dict[str, Any]:
"""粗扫标签,统计可建议映射点数(不依赖黄区精确色)。"""
try:
from openpyxl import load_workbook
except ImportError:
return {"mappingPointCount": 0, "fields": {}, "error": "openpyxl_missing"}
fields: dict[str, str] = {}
wb = load_workbook(path, read_only=True, data_only=True)
try:
ws = wb.worksheets[0]
sheet = ws.title
for row in ws.iter_rows(max_row=60, max_col=20):
for cell in row:
val = cell.value
if val is None:
continue
text = str(val).strip()
if not text:
continue
norm = text.replace(":", ":")
code = None
for lab, c in label_dict.items():
lab_n = lab.replace(":", ":").replace(":", "").strip()
t_n = norm.replace(":", "").strip()
if lab_n and (t_n == lab_n or t_n.startswith(lab_n)):
code = c
break
if code and code not in fields:
# 右侧一格作为建议填点
fill = f"{sheet}!{cell.coordinate}"
col_i = int(getattr(cell, "column", 1) or 1)
if col_i < 40:
from openpyxl.utils import get_column_letter
fill = f"{sheet}!{get_column_letter(col_i + 1)}{cell.row}"
fields[code] = fill
finally:
wb.close()
return {"mappingPointCount": len(fields), "fields": fields}
def build_manifest() -> list[dict[str, Any]]:
contract = json.loads(CONTRACT.read_text(encoding="utf-8"))
label_dict = contract.get("labelDictionary") or {}
items = []
for path in unique_xlsx_files():
meta = classify(path.name)
mapped = rough_mapping_points(path, label_dict)
items.append(
{
"fileName": path.name,
"relativePath": str(path.relative_to(ROOT)).replace("\\", "/"),
"bizType": meta["bizType"],
"category": meta["category"],
"keywords": meta["keywords"],
"mode": meta["mode"],
"suggestedMappingPointCount": mapped.get("mappingPointCount", 0),
"suggestedFields": mapped.get("fields") or {},
"parseNote": mapped.get("error"),
}
)
return items
def upload_one(base_url: str, token: str, item: dict[str, Any]) -> dict[str, Any]:
import urllib.request
boundary = "----YtdQuoteTemplateBoundary"
file_path = ROOT / item["relativePath"]
file_bytes = file_path.read_bytes()
fields = {
"name": Path(item["fileName"]).stem,
"bizType": item["bizType"],
"category": item["category"],
"keywordsJson": json.dumps(item["keywords"], ensure_ascii=False),
"mode": item["mode"],
"isDefault": "false",
}
body = bytearray()
for k, v in fields.items():
body.extend(f"--{boundary}\r\n".encode())
body.extend(f'Content-Disposition: form-data; name="{k}"\r\n\r\n'.encode())
body.extend(f"{v}\r\n".encode("utf-8"))
body.extend(f"--{boundary}\r\n".encode())
body.extend(
f'Content-Disposition: form-data; name="file"; filename="{item["fileName"]}"\r\n'.encode(
"utf-8"
)
)
body.extend(b"Content-Type: application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\r\n\r\n")
body.extend(file_bytes)
body.extend(b"\r\n")
body.extend(f"--{boundary}--\r\n".encode())
url = base_url.rstrip("/") + "/inquiry/quoteTemplate/upload"
req = urllib.request.Request(url, data=bytes(body), method="POST")
req.add_header("Content-Type", f"multipart/form-data; boundary={boundary}")
req.add_header("X-Access-Token", token)
with urllib.request.urlopen(req, timeout=60) as resp:
return json.loads(resp.read().decode("utf-8"))
def main(argv: list[str]) -> int:
parser = argparse.ArgumentParser(description="首批报价模板 seed")
parser.add_argument("--dry-run", action="store_true", help="只写清单,不上传")
parser.add_argument("--upload-url", default="", help="主账 base,如 http://127.0.0.1:8180/jeecgboot")
parser.add_argument("--token", default="", help="X-Access-Token")
args = parser.parse_args(argv)
if not TEMPLATE_DIR.is_dir():
print(f"模板目录不存在: {TEMPLATE_DIR}", file=sys.stderr)
return 1
items = build_manifest()
OUT_JSON.parent.mkdir(parents=True, exist_ok=True)
OUT_JSON.write_text(
json.dumps(
{
"schemaVersion": "quote-template-seed-v1",
"contract": "contracts/quote-template-mapping-v1.json",
"count": len(items),
"items": items,
},
ensure_ascii=False,
indent=2,
),
encoding="utf-8",
)
print(f"wrote {OUT_JSON} count={len(items)}")
for it in items:
print(
f" [{it['mode']}] {it['bizType']}/{it['category']} "
f"map~{it['suggestedMappingPointCount']} {it['fileName']}"
)
if args.dry_run or not args.upload_url:
print("dry-run 完成(未上传)。加 --upload-url 与 --token 可推主账。")
return 0
if not args.token:
print("上传需要 --token", file=sys.stderr)
return 2
for it in items:
try:
resp = upload_one(args.upload_url, args.token, it)
print(f"UPLOAD OK {it['fileName']} -> {resp.get('message') or resp.get('code')}")
except Exception as exc:
print(f"UPLOAD FAIL {it['fileName']}: {exc}", file=sys.stderr)
return 0
if __name__ == "__main__":
raise SystemExit(main(sys.argv[1:]))