后台成本价、销售价、利润点按费用逐行展示,费用合计用报价官方总价。选完线路只关页面。报价单文件名带上模板名,空运起运地和转人工跟对应工单走。
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -57,6 +57,7 @@ _OPTIONAL_INQUIRY_LABELS = (
|
||||
"提货地址(非必填)",
|
||||
"收货地址(非必填)",
|
||||
"贸易条款(非必填)",
|
||||
"起运地(非必填)",
|
||||
"起运港(非必填)",
|
||||
"货源地",
|
||||
"货好时间",
|
||||
@@ -129,6 +130,7 @@ _SHEET_CLEAN_KEYS = frozenset(
|
||||
"整柜或拼柜",
|
||||
"箱型箱量",
|
||||
"贸易条款",
|
||||
"货好时间",
|
||||
"运输分类",
|
||||
"报价日期",
|
||||
*QUOTE_DATE_SYNONYMS,
|
||||
@@ -136,7 +138,52 @@ _SHEET_CLEAN_KEYS = frozenset(
|
||||
"包装类型",
|
||||
}
|
||||
)
|
||||
# 报价单从这里往下是费用和条款,不是询价栏。切掉后再抽字段。
|
||||
_QUOTE_BODY_MARKS = (
|
||||
"报价明细",
|
||||
"报价条件",
|
||||
"预估费用",
|
||||
"空舱费",
|
||||
"托运方",
|
||||
"承运方",
|
||||
)
|
||||
# 表头常被拉成「体 积」「计 费 重」。只收这些词中间的空格,不跨行。
|
||||
_TIGHTEN_LABELS = tuple(
|
||||
sorted(
|
||||
{
|
||||
*_QUOTE_BODY_MARKS,
|
||||
"计费重",
|
||||
"时效要求",
|
||||
"派送地址",
|
||||
"提货地址",
|
||||
"收货地址",
|
||||
"货好时间",
|
||||
"贸易条款",
|
||||
"报价日期",
|
||||
"起运地",
|
||||
"起运港",
|
||||
"目的地",
|
||||
"目的港",
|
||||
"包装类型",
|
||||
"包装方式",
|
||||
"体积",
|
||||
"重量",
|
||||
"品名",
|
||||
"件数",
|
||||
"货源地",
|
||||
"运输方式",
|
||||
},
|
||||
key=len,
|
||||
reverse=True,
|
||||
)
|
||||
)
|
||||
_FEE_NAME = re.compile(r"^\*?[\u4e00-\u9fffA-Za-z0-9]{1,12}费$")
|
||||
_SHEET_STOP = (
|
||||
"报价明细",
|
||||
"报价条件",
|
||||
"预估费用",
|
||||
"计费重",
|
||||
"时效要求",
|
||||
"要求运输时效",
|
||||
"Requested Transit",
|
||||
"费用汇总",
|
||||
@@ -291,6 +338,52 @@ def _merge_spoken_collab(text: str, facts: dict[str, str]) -> dict[str, str]:
|
||||
return merged
|
||||
|
||||
|
||||
def prepare_sheet_oral(text: str) -> str:
|
||||
"""
|
||||
报价单只留询价栏。
|
||||
|
||||
费用明细、报价条件、落款整段丢掉。表头里被空格拉开的「体 积」收回成「体积」,
|
||||
否则上一栏会把后面的报价条件整段吞进去。计费重不是毛重,单独拿掉。
|
||||
副作用:无。
|
||||
"""
|
||||
body = _tighten_sheet_labels(text or "")
|
||||
cut = len(body)
|
||||
for mark in _QUOTE_BODY_MARKS:
|
||||
pos = body.find(mark)
|
||||
if 0 <= pos < cut:
|
||||
cut = pos
|
||||
body = body[:cut]
|
||||
body = re.sub(r"计费重\s*[::]?\s*[0-9.]+\s*(?:KG|KGS|kg|公斤)?", " ", body)
|
||||
return body.strip()
|
||||
|
||||
|
||||
def _tighten_sheet_labels(text: str) -> str:
|
||||
"""已知表头中间只允许空格。不跨行,避免把两行并成一个词。"""
|
||||
body = text or ""
|
||||
gap = r"[ \t\u3000]{0,6}"
|
||||
for label in _TIGHTEN_LABELS:
|
||||
if len(label) < 2:
|
||||
continue
|
||||
pat = re.compile(
|
||||
r"(?<![\u4e00-\u9fffA-Za-z0-9])" + gap.join(re.escape(ch) for ch in label)
|
||||
)
|
||||
body = pat.sub(label, body)
|
||||
return body
|
||||
|
||||
|
||||
def _drop_fee_place_values(facts: dict[str, str]) -> dict[str, str]:
|
||||
"""起运地/目的地收成「操作费、文件费」时丢掉。那是费用栏,不是港口。"""
|
||||
out = dict(facts or {})
|
||||
for key in ("起运港", "起运地", "目的港", "目的地"):
|
||||
val = str(out.get(key) or "").strip()
|
||||
if not val:
|
||||
continue
|
||||
bits = [p.strip(" *::\t") for p in re.split(r"[\n/、,,]+", val) if p.strip()]
|
||||
if bits and all(_FEE_NAME.match(p) for p in bits):
|
||||
out.pop(key, None)
|
||||
return out
|
||||
|
||||
|
||||
def extract_inquiry_snapshot(
|
||||
text: str,
|
||||
*,
|
||||
@@ -317,7 +410,7 @@ def extract_inquiry_snapshot(
|
||||
for key in ("货好时间", "提货地址", "收货地址", "货源地", "贸易条款"):
|
||||
if str(facts.get(key) or "").strip():
|
||||
facts[key] = clip_inquiry_value(str(facts[key]))
|
||||
facts = _clip_sheet_inquiry_facts(facts)
|
||||
facts = _drop_fee_place_values(_clip_sheet_inquiry_facts(facts))
|
||||
subtype = str(facts.get("运输类型") or "").strip()
|
||||
return {
|
||||
"business_line": mode,
|
||||
@@ -325,6 +418,8 @@ def extract_inquiry_snapshot(
|
||||
"facts": facts,
|
||||
"source": "injected",
|
||||
}
|
||||
# 附件常是整张报价单。先丢掉费用明细和条款,再抽询价栏。
|
||||
text = prepare_sheet_oral(text)
|
||||
mode = detect_transport_mode(text)
|
||||
facts = parse_labeled_facts(text)
|
||||
source = "labeled_or_mode"
|
||||
@@ -361,7 +456,7 @@ def extract_inquiry_snapshot(
|
||||
for key in ("货好时间", "提货地址", "收货地址", "货源地", "贸易条款"):
|
||||
if str(facts.get(key) or "").strip():
|
||||
facts[key] = clip_inquiry_value(str(facts[key]))
|
||||
facts = _clip_sheet_inquiry_facts(facts)
|
||||
facts = _drop_fee_place_values(_clip_sheet_inquiry_facts(facts))
|
||||
land_subtype = str((b_payload or {}).get("land_subtype") or facts.get("运输类型") or "").strip()
|
||||
if land_subtype and not str(facts.get("运输类型") or "").strip():
|
||||
facts["运输类型"] = land_subtype
|
||||
@@ -453,6 +548,10 @@ def clip_sheet_field_value(raw: str) -> str:
|
||||
parts = [p for p in parts if p and not _SHEET_HEADER_NOISE.search(p)]
|
||||
if not parts:
|
||||
return ""
|
||||
# 格子里只剩下一个字段名(体积、时效要求)时,这一栏是空的,不能把字段名当成值。
|
||||
token = parts[0].strip().strip("::")
|
||||
if token in _SHEET_CLEAN_KEYS or token in _TIGHTEN_LABELS:
|
||||
return ""
|
||||
return parts[0].strip()
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user