553 lines
18 KiB
Python
553 lines
18 KiB
Python
"""
|
||
字段合同校验:读 inquiry-required-fields-v1,只做作用域/非空/默认值。
|
||
|
||
本文件职责:判断首次查价缺项;报价日期未说则填当天,不进入补问。
|
||
口语已写 TMS 单位(件/KGS/CBM)时,空着的件数/毛重/体积可机读补上。
|
||
禁止:正则猜港口/箱型;禁止用封闭枚举拒绝用户原文;禁止改六态。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import logging
|
||
import re
|
||
from typing import Any
|
||
from zoneinfo import ZoneInfo
|
||
|
||
from agent.schema.contracts_loader import require_formal_contract
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
_SHANGHAI = ZoneInfo("Asia/Shanghai")
|
||
|
||
# 空运询价必填 8 项:销售侧名称。内部仍用合同键。
|
||
AIR_REQUIRED_FIELDS = (
|
||
("起运港", "起运地"),
|
||
("目的港", "目的地"),
|
||
("品名", "品名"),
|
||
("件数", "件数"),
|
||
("毛重", "重量"),
|
||
("体积", "体积"),
|
||
("包装方式", "包装类型"),
|
||
("报价日期", "报价日期"),
|
||
)
|
||
AIR_DISPLAY = {key: label for key, label in AIR_REQUIRED_FIELDS}
|
||
|
||
# 海运询价必填 7 项:销售侧名称。内部键「货量」对外叫「货物数量」。
|
||
SEA_REQUIRED_FIELDS = (
|
||
("起运港", "起运港"),
|
||
("目的港", "目的港"),
|
||
("品名", "品名"),
|
||
("货量", "货物数量"),
|
||
("整柜或拼柜", "整柜或拼柜"),
|
||
("箱型箱量", "箱型箱量"),
|
||
("报价日期", "报价日期"),
|
||
)
|
||
SEA_DISPLAY = {key: label for key, label in SEA_REQUIRED_FIELDS}
|
||
|
||
# 陆运销售侧名称:卡片写始发站/货物品名;内部仍用合同键起运港/品名。
|
||
LAND_DISPLAY = {
|
||
"起运港": "始发站",
|
||
"目的港": "目的地",
|
||
"品名": "货物品名",
|
||
"毛重": "重量",
|
||
"体积": "体积",
|
||
"车型/数量": "车型/数量",
|
||
"通关口岸": "通关口岸",
|
||
"报价日期": "报价日期",
|
||
}
|
||
|
||
# 协同补充字段:拉群后可静默写入,不挡首次查价。内部键仍是 HS编码。
|
||
SEA_COLLAB_FIELDS = ("贸易条款", "货好时间", "HS编码", "是否含油", "是否含电", "是否含磁")
|
||
# 空运群只补含电/含磁,摘要不要把海运六项搬过去。
|
||
AIR_COLLAB_FIELDS = ("是否含电", "是否含磁")
|
||
# 群摘要对外展示名。内部继续用 HS编码,避免改合同键。
|
||
SEA_COLLAB_DISPLAY = {
|
||
"HS编码": "商品海关编码",
|
||
}
|
||
# 陆运摘要标题:海关编码,不要写成海运的商品海关编码。
|
||
LAND_COLLAB_DISPLAY = {
|
||
"HS编码": "海关编码",
|
||
}
|
||
# 填写样例允许拉群的类型+线路 → 协同内部键(顺序与规格一致)。
|
||
_LAND_COLLAB_BY_PAIR: dict[tuple[str, str], tuple[str, ...]] = {
|
||
("国内运输拼车", "国内长途/零担"): (
|
||
"客户名称",
|
||
"包装方式",
|
||
"货值",
|
||
"是否为危险品",
|
||
),
|
||
("国内运输拼车", "国内城配/拖车/打包"): (
|
||
"客户名称",
|
||
"包装方式",
|
||
"货值",
|
||
"是否为危险品",
|
||
),
|
||
("国内运输整车", "国内长途/零担"): (
|
||
"客户名称",
|
||
"货值",
|
||
"包装方式",
|
||
"是否为危险品",
|
||
),
|
||
("国内运输整车", "国内城配/拖车/打包"): (
|
||
"客户名称",
|
||
"货值",
|
||
"包装方式",
|
||
"是否为危险品",
|
||
),
|
||
("跨境集拼", "中港/中亚/中欧"): (
|
||
"客户名称",
|
||
"HS编码",
|
||
"贸易条款",
|
||
"包装方式",
|
||
"是否为危险品",
|
||
),
|
||
("跨境集拼", "东南亚"): (
|
||
"客户名称",
|
||
"HS编码",
|
||
"贸易条款",
|
||
"包装方式",
|
||
"是否为危险品",
|
||
),
|
||
("跨境整车", "中港/中亚/中欧"): (
|
||
"客户名称",
|
||
"HS编码",
|
||
"贸易条款",
|
||
"货值",
|
||
"包装方式",
|
||
"是否为危险品",
|
||
),
|
||
("跨境整车", "东南亚"): (
|
||
"客户名称",
|
||
"HS编码",
|
||
"贸易条款",
|
||
"货值",
|
||
"包装方式",
|
||
"是否为危险品",
|
||
),
|
||
("中港整车", "中港/中亚/中欧"): (
|
||
"客户名称",
|
||
"HS编码",
|
||
"车型/数量",
|
||
"贸易条款",
|
||
"货值",
|
||
"包装方式",
|
||
"是否为危险品",
|
||
),
|
||
("中港零担/集拼", "中港/中亚/中欧"): (
|
||
"客户名称",
|
||
"HS编码",
|
||
"贸易条款",
|
||
"包装方式",
|
||
"是否为危险品",
|
||
),
|
||
}
|
||
# 意图分类用:海运六项 ∪ 陆运全部协同键。
|
||
ALL_COLLAB_KEYS = tuple(
|
||
dict.fromkeys(
|
||
list(SEA_COLLAB_FIELDS)
|
||
+ ["客户名称", "包装方式", "货值", "是否为危险品", "车型/数量"]
|
||
)
|
||
)
|
||
|
||
# 口语里已经带上 TMS 文档规定的单位时,补进空字段。
|
||
# 只认「数字+单位」,不猜港口,不把长宽高换算成立方。
|
||
_RE_PIECES = re.compile(r"(?<![\d.])(\d+)\s*(件|pcs)\b", re.I)
|
||
_RE_WEIGHT = re.compile(r"(?<![\d.])(\d+(?:\.\d+)?)\s*(kgs|kg|公斤|千克)\b", re.I)
|
||
_RE_VOLUME = re.compile(
|
||
r"(?<![\d.])(\d+(?:\.\d+)?)\s*(cbm|立方米|立方|m3|m³)\b",
|
||
re.I,
|
||
)
|
||
_RE_VOLUME_FANG = re.compile(r"(?<![\d.])(\d+(?:\.\d+)?)\s*方(?!向)")
|
||
_RE_WEIGHT_WORD = re.compile(r"(?:毛重|重量)\s*[::]?\s*(\d+(?:\.\d+)?)")
|
||
_RE_VOLUME_WORD = re.compile(r"体积\s*[::]?\s*(\d+(?:\.\d+)?)")
|
||
# 销售原话里的包装词:当前句出现则覆盖旧值,散货也要收下
|
||
_PACK_WORDS = ("纸箱", "木箱", "托盘", "卡板", "散货")
|
||
# 报价日期:9月15日 / 2026-09-15;未写年用上海当年
|
||
_RE_DATE_YMD = re.compile(
|
||
r"(20\d{2})\s*[-/.年]\s*(\d{1,2})\s*[-/.月]\s*(\d{1,2})\s*[日号]?"
|
||
)
|
||
_RE_DATE_MD = re.compile(r"(?<!\d)(\d{1,2})\s*月\s*(\d{1,2})\s*[日号]?")
|
||
|
||
# 展示名 / 别名 → 内部键
|
||
_ALIASES = {
|
||
"起运地": "起运港",
|
||
"目的地": "目的港",
|
||
"包装类型": "包装方式",
|
||
"货物数量": "货量",
|
||
"始发站": "起运港",
|
||
"货物品名": "品名",
|
||
"重量(KG)": "毛重",
|
||
"重量": "毛重",
|
||
"体积(CBM)": "体积",
|
||
"车型/数量": "车型/数量",
|
||
"车型数量": "车型/数量",
|
||
"数量": "数量",
|
||
"HS": "HS编码",
|
||
"HS CODE": "HS编码",
|
||
"商品海关编码": "HS编码",
|
||
"海关编码": "HS编码",
|
||
"客户名称": "客户名称",
|
||
"货值": "货值",
|
||
"是否为危险品": "是否为危险品",
|
||
"通关口岸": "通关口岸",
|
||
"通关口岸(非必填)": "通关口岸",
|
||
}
|
||
|
||
|
||
def shanghai_today() -> str:
|
||
"""报价日期系统默认:Asia/Shanghai 当天。"""
|
||
from datetime import datetime
|
||
|
||
return datetime.now(_SHANGHAI).date().isoformat()
|
||
|
||
|
||
def normalize_facts(facts: dict[str, Any] | None) -> dict[str, str]:
|
||
"""
|
||
把展示名/别名折成合同内部键,去掉首尾空白。
|
||
|
||
空值丢弃;不解释语义。
|
||
"""
|
||
out: dict[str, str] = {}
|
||
for raw_key, raw_val in (facts or {}).items():
|
||
key = str(raw_key).strip()
|
||
key = _ALIASES.get(key, key)
|
||
if raw_val is None:
|
||
continue
|
||
val = str(raw_val).strip()
|
||
if not val:
|
||
continue
|
||
out[key] = val
|
||
_fold_land_vehicle_qty(out)
|
||
return out
|
||
|
||
|
||
def _fold_land_vehicle_qty(merged: dict[str, str]) -> None:
|
||
"""
|
||
把拆开的车型、数量收成一个「车型/数量」。
|
||
|
||
销售侧只认这一项必填;模型若仍拆成两项,这里合并,禁止再分别补问。
|
||
不改主账。有「车型/数量」则保留原文,缺的再用车型、数量拼上。
|
||
"""
|
||
combo = (merged.get("车型/数量") or "").strip()
|
||
vehicle = (merged.get("车型") or "").strip()
|
||
qty = (merged.get("数量") or "").strip()
|
||
if not combo:
|
||
if vehicle and qty:
|
||
combo = f"{vehicle}/{qty}"
|
||
elif vehicle:
|
||
combo = vehicle
|
||
elif qty:
|
||
combo = qty
|
||
elif qty and qty not in combo:
|
||
combo = f"{combo}/{qty}"
|
||
if combo:
|
||
merged["车型/数量"] = combo
|
||
|
||
|
||
def _harvest_packaging(raw: str) -> str:
|
||
"""当前原话里最后出现的包装词;没有则空串。"""
|
||
last = ""
|
||
last_pos = -1
|
||
for word in _PACK_WORDS:
|
||
pos = raw.rfind(word)
|
||
if pos > last_pos:
|
||
last_pos = pos
|
||
last = word
|
||
return last
|
||
|
||
|
||
def _harvest_quote_date(raw: str) -> str:
|
||
"""当前原话里的报价日期,规范成 yyyy-MM-dd;没有则空串。"""
|
||
from datetime import datetime
|
||
|
||
hit = _RE_DATE_YMD.search(raw)
|
||
if hit:
|
||
year, month, day = int(hit.group(1)), int(hit.group(2)), int(hit.group(3))
|
||
else:
|
||
hit = _RE_DATE_MD.search(raw)
|
||
if not hit:
|
||
return ""
|
||
year = datetime.now(_SHANGHAI).year
|
||
month, day = int(hit.group(1)), int(hit.group(2))
|
||
try:
|
||
return datetime(year, month, day).date().isoformat()
|
||
except ValueError:
|
||
return ""
|
||
|
||
|
||
def harvest_oral_measures(text: str, facts: dict[str, str] | None = None) -> dict[str, str]:
|
||
"""
|
||
从当前原话机读件数/重量/体积/包装/报价日期。
|
||
|
||
当前句写了的覆盖旧值:散货替换托盘,9月16日写入报价日期。
|
||
副作用:无。不改主账。不猜港口。
|
||
"""
|
||
merged = dict(facts or {})
|
||
raw = text or ""
|
||
hit = _RE_PIECES.search(raw)
|
||
if hit:
|
||
merged["件数"] = f"{hit.group(1)}{hit.group(2)}"
|
||
hit = _RE_WEIGHT.search(raw)
|
||
if hit:
|
||
# 原样留下用户单位:1.5KG 不要改成 1.5KGS
|
||
merged["毛重"] = f"{hit.group(1)}{hit.group(2)}"
|
||
else:
|
||
hit = _RE_WEIGHT_WORD.search(raw)
|
||
if hit:
|
||
merged["毛重"] = hit.group(1)
|
||
hit = _RE_VOLUME.search(raw)
|
||
if hit:
|
||
merged["体积"] = f"{hit.group(1)}{hit.group(2)}"
|
||
else:
|
||
hit = _RE_VOLUME_FANG.search(raw) or _RE_VOLUME_WORD.search(raw)
|
||
if hit:
|
||
merged["体积"] = hit.group(1)
|
||
pack = _harvest_packaging(raw)
|
||
if pack:
|
||
merged["包装方式"] = pack
|
||
quote_date = _harvest_quote_date(raw)
|
||
if quote_date:
|
||
merged["报价日期"] = quote_date
|
||
from agent.schema.tms_air_query import harvest_named_air_route
|
||
|
||
return harvest_named_air_route(raw, merged)
|
||
|
||
|
||
def land_collab_fields(land_type: str, route: str) -> tuple[str, ...]:
|
||
"""
|
||
本单陆运协同内部键,顺序与规格一致。
|
||
|
||
对不上填写样例返回空元组,调用方不得按海运六项去列。
|
||
海关编码内部仍是 HS编码。副作用:无。
|
||
"""
|
||
from agent.schema.land_options import canonicalize_route, canonicalize_transport_type
|
||
|
||
t = canonicalize_transport_type(land_type) or (land_type or "").strip()
|
||
r = canonicalize_route(route) or (route or "").strip()
|
||
return _LAND_COLLAB_BY_PAIR.get((t, r), ())
|
||
|
||
|
||
def collab_fields_for_ticket(facts: dict[str, str] | None, business_line: str) -> tuple[str, ...]:
|
||
"""
|
||
按业务线取出群摘要 / 写入允许的协同键。
|
||
|
||
LAND 看事实里的运输类型+线路类别;SEA/AIR 用固定名单。
|
||
"""
|
||
line = (business_line or "").strip().upper()
|
||
if line == "AIR":
|
||
return AIR_COLLAB_FIELDS
|
||
if line == "LAND":
|
||
src = dict(facts or {})
|
||
return land_collab_fields(
|
||
str(src.get("运输类型") or ""),
|
||
str(src.get("线路类别") or ""),
|
||
)
|
||
return SEA_COLLAB_FIELDS
|
||
|
||
|
||
def collab_display_name(key: str, business_line: str) -> str:
|
||
"""协同项对外标题。陆运 HS编码 → 海关编码;海运 → 商品海关编码。"""
|
||
line = (business_line or "").strip().upper()
|
||
if line == "LAND":
|
||
return LAND_COLLAB_DISPLAY.get(key, key)
|
||
return SEA_COLLAB_DISPLAY.get(key, key)
|
||
|
||
|
||
def pick_collab_from_facts(
|
||
facts: dict[str, str] | None,
|
||
keys: tuple[str, ...],
|
||
) -> dict[str, str]:
|
||
"""
|
||
从私聊事实里取出已有协同键,供进群预填。
|
||
|
||
只取 keys 里非空值。海关编码/商品海关编码经 normalize 折成 HS编码。
|
||
不在 keys 的询价项(如跨境整车车型/数量)不会被取出。
|
||
不覆盖调用方已有协同值;本函数只返回候选。
|
||
"""
|
||
allowed = set(keys or ())
|
||
if not allowed:
|
||
return {}
|
||
src = normalize_facts(facts)
|
||
out: dict[str, str] = {}
|
||
for key in keys:
|
||
val = (src.get(key) or "").strip()
|
||
if val:
|
||
out[key] = val
|
||
return out
|
||
|
||
|
||
def harvest_oral_collab(text: str) -> dict[str, str]:
|
||
"""
|
||
群里口语协同:含油/含电/含磁、货好时间、是否为危险品。
|
||
|
||
只认原话里已经说死的词(不含电、随时可提、普货),不猜贸易条款和海关编码。
|
||
危险品:否词(不是危险品/非危/普货)先于是词,避免「不是危险品」收成是。
|
||
"""
|
||
raw = text or ""
|
||
out: dict[str, str] = {}
|
||
yes_no = (
|
||
("是否含电", ("不含电", "不带电", "无电"), ("含电", "带电")),
|
||
("是否含油", ("不含油", "不带油", "无油"), ("含油", "带油")),
|
||
("是否含磁", ("不含磁", "不带磁", "无磁"), ("含磁", "带磁")),
|
||
(
|
||
"是否为危险品",
|
||
("不是危险品", "非危险品", "非危", "普货"),
|
||
("危险品", "有危", "含危"),
|
||
),
|
||
)
|
||
for key, no_words, yes_words in yes_no:
|
||
if any(w in raw for w in no_words):
|
||
out[key] = "否"
|
||
elif any(w in raw for w in yes_words):
|
||
out[key] = "是"
|
||
ready = ("随时可提", "货好随时", "随时货好", "货已好", "已经货好")
|
||
for word in ready:
|
||
if word in raw:
|
||
out["货好时间"] = word
|
||
break
|
||
return out
|
||
|
||
|
||
def project_sea_cargo_qty(facts: dict[str, str] | None) -> dict[str, str]:
|
||
"""
|
||
海运货物数量投影:原话「15件」常被抽成空运「件数」。
|
||
|
||
货量已有则不覆盖。件数有、货量空时,货量用件数。不猜吨位、不改空运校验。
|
||
"""
|
||
merged = dict(facts or {})
|
||
qty = (merged.get("货量") or merged.get("货物数量") or "").strip()
|
||
pieces = (merged.get("件数") or "").strip()
|
||
if not qty and pieces:
|
||
merged["货量"] = pieces
|
||
return merged
|
||
|
||
|
||
def apply_quote_date_default(facts: dict[str, str]) -> dict[str, str]:
|
||
"""销售未提供报价日期时填当天;已有则不覆盖。"""
|
||
merged = dict(facts)
|
||
if not merged.get("报价日期"):
|
||
merged["报价日期"] = shanghai_today()
|
||
logger.info("报价日期缺省为当天 date=%s", merged["报价日期"])
|
||
return merged
|
||
|
||
|
||
def _pricing_required_keys(
|
||
contract: dict[str, Any],
|
||
business_line: str,
|
||
land_subtype: str,
|
||
facts: dict[str, str] | None = None,
|
||
) -> list[str]:
|
||
"""当前激活的首次查价内部键列表。中港整车/零担是两个运输类型。"""
|
||
_ = facts
|
||
line = (business_line or "").strip().upper()
|
||
pricing = contract.get("pricing_required") or {}
|
||
if line == "SEA":
|
||
return list((pricing.get("SEA") or {}).get("always") or [])
|
||
if line == "AIR":
|
||
return list((pricing.get("AIR") or {}).get("always") or [])
|
||
if line == "LAND":
|
||
land = pricing.get("LAND") or {}
|
||
block = land.get(land_subtype) or {}
|
||
if not block:
|
||
return []
|
||
return list(block.get("always") or [])
|
||
return []
|
||
|
||
|
||
def display_name(internal_key: str, business_line: str = "AIR") -> str:
|
||
"""补问时给销售看的名字。"""
|
||
line = (business_line or "").upper()
|
||
if line == "AIR":
|
||
return AIR_DISPLAY.get(internal_key, internal_key)
|
||
if line == "SEA":
|
||
return SEA_DISPLAY.get(internal_key, internal_key)
|
||
if line == "LAND":
|
||
return LAND_DISPLAY.get(internal_key, internal_key)
|
||
return internal_key
|
||
|
||
|
||
def validate_required_fields(
|
||
*,
|
||
facts: dict[str, Any],
|
||
business_line: str = "",
|
||
land_subtype: str = "",
|
||
require_formal: bool = True,
|
||
) -> dict[str, Any]:
|
||
"""
|
||
首次查价完整性。
|
||
|
||
返回:ok / implemented / missing(内部键)/ missing_display / facts(含默认报价日期)/ errors。
|
||
报价日期不进 missing。未识别业务线时 missing 为空、need_transport_mode=True。
|
||
"""
|
||
_ = require_formal
|
||
contract = require_formal_contract("inquiry-required-fields-v1")
|
||
normalized = apply_quote_date_default(normalize_facts(facts))
|
||
line = (business_line or "").strip().upper()
|
||
if line == "SEA":
|
||
normalized = project_sea_cargo_qty(normalized)
|
||
if line not in {"SEA", "AIR", "LAND"}:
|
||
return {
|
||
"ok": False,
|
||
"implemented": True,
|
||
"need_transport_mode": True,
|
||
"missing": [],
|
||
"missing_display": [],
|
||
"facts": normalized,
|
||
"errors": [],
|
||
"business_line": line,
|
||
}
|
||
if line == "LAND":
|
||
land_subtype = (land_subtype or "").strip() or str(
|
||
normalized.get("运输类型") or ""
|
||
).strip()
|
||
if line == "LAND" and not (land_subtype or "").strip():
|
||
return {
|
||
"ok": False,
|
||
"implemented": True,
|
||
"need_transport_mode": False,
|
||
"need_land_subtype": True,
|
||
"missing": [],
|
||
"missing_display": [],
|
||
"facts": normalized,
|
||
"errors": [],
|
||
"business_line": line,
|
||
}
|
||
|
||
required = _pricing_required_keys(contract, line, land_subtype, normalized)
|
||
missing = [k for k in required if k != "报价日期" and not normalized.get(k)]
|
||
missing_display = [display_name(k, line) for k in missing]
|
||
ok = len(missing) == 0
|
||
logger.info(
|
||
"schema.validate line=%s land=%s ok=%s missing=%s",
|
||
line,
|
||
land_subtype or "-",
|
||
ok,
|
||
missing,
|
||
)
|
||
return {
|
||
"ok": ok,
|
||
"implemented": True,
|
||
"need_transport_mode": False,
|
||
"need_land_subtype": False,
|
||
"missing": missing,
|
||
"missing_display": missing_display,
|
||
"facts": normalized,
|
||
"errors": [],
|
||
"business_line": line,
|
||
"land_subtype": land_subtype,
|
||
}
|
||
|
||
|
||
def validate_required_fields_shell(
|
||
*,
|
||
facts: dict[str, Any],
|
||
business_line: str = "",
|
||
require_formal: bool = True,
|
||
) -> dict[str, Any]:
|
||
"""兼容骨架入口:现已接入正式激活规则。"""
|
||
return validate_required_fields(
|
||
facts=facts,
|
||
business_line=business_line,
|
||
require_formal=require_formal,
|
||
)
|