这用于审核报关单商品与海关编码是否匹配,并给出修改建议。它通过名称相似度、关键词命中率和申报要素完整度三个维度打分,输出匹配结论和候选编码。
```python
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""
报关单商品 —— 海关编码(HS Code)匹配性审核 & 修改建议
=========================================================
核心思路:
1. 税则库(HSItem)为每个税号维护:编码 / 名称 / 特征关键词 / 申报要素 / 归类提示
2. 对报单上每条商品,从三个维度打分:
- 名称相似度 (字符 bigram Dice 系数)
- 关键词命中率 (按词长加权,长词权重高)
- 申报要素完整度
3. 得分排序,与「申报编码」自身得分对比,判定:
匹配 / 基本匹配 / 疑似归类错误 / 匹配度过低 / 编码无效
4. 输出可执行的修改建议(改编码、补要素、补描述、人工复核)
依赖:仅标准库。生产环境可把关键词匹配升级为
向量检索(embedding)+ 大模型归类复核。
"""
from __future__ import annotations
import csv
from dataclasses import dataclass, field
from typing import Any, Iterable
# ==========================================================
# 一、文本相似度工具
# ==========================================================
def bigrams(text: str) -> set:
"""中文按字符切 bigram,英文/数字同样适用"""
s = "".join(ch for ch in (text or "") if not ch.isspace())
if not s:
return set()
if len(s) == 1:
return {s}
return {s[i:i + 2] for i in range(len(s) - 1)}
def dice(a: set, b: set) -> float:
"""Dice 系数,0~1"""
if not a or not b:
return 0.0
return 2.0 * len(a & b) / (len(a) + len(b))
# ==========================================================
# 二、数据模型
# ==========================================================
@dataclass
class HSItem:
"""税则库中的一条税号"""
code: str # 10 位海关编码
name: str # 税则商品名称
keywords: tuple = () # 特征关键词(同义词都放进来)
elements: tuple = () # 申报要素名,如 ("品名","品牌","型号")
unit: str = "" # 法定计量单位
note: str = "" # 归类提示 / 易错点
@property
def heading(self) -> str: # 品目(前 4 位)
return self.code[:4]
@property
def subheading(self) -> str: # 子目(前 6 位)
return self.code[:6]
@dataclass
class DeclareItem:
"""报关单上的一条商品明细"""
seq: int # 项号
name: str # 商品名称
declared_code: str # 申报的海关编码
spec: str = "" # 规格型号
elements: dict = field(default_factory=dict) # 申报要素 {"品牌": "XXX", ...}
@property
def full_text(self) -> str:
"""用于关键词匹配的全文"""
parts = [self.name or "", self.spec or ""]
parts += [str(v) for v in self.elements.values() if v]
return " ".join(parts)
# ==========================================================
# 三、匹配引擎
# ==========================================================
class HSCodeMatcher:
"""HS 编码匹配审核器"""
HIGH = 0.70 # 高置信阈值
MID = 0.50 # 及格阈值
DELTA = 0.08 # 更优候选需要领先的最小分差
def __init__(self, catalog: Iterable[HSItem],
weights: tuple = (0.35, 0.45, 0.20)):
self.catalog = list(catalog)
self.by_code = {i.code: i for i in self.catalog}
self.w_name, self.w_kw, self.w_elem = weights
# 预计算,避免重复切词
self._name_bg = {i.code: bigrams(i.name) for i in self.catalog}
self._kw_pack = {
i.code: [(k, min(len(k), 4)) for k in i.keywords if k]
for i in self.catalog
}
# ---------- 单条打分 ----------
def _score(self, item: HSItem, dec: DeclareItem) -> tuple:
# 1) 名称相似度
sim_name = dice(self._name_bg[item.code], bigrams(dec.name))
if len(dec.name or "") >= 2 and dec.name and (
dec.name in item.name or item.name in dec.name):
sim_name = max(sim_name, 0.85) # 包含关系给保底分
# 2) 关键词命中(按词长加权)
text = dec.full_text
hits, miss = [], []
hit_w = tot_w = 0.0
for kw, w in self._kw_pack[item.code]:
tot_w += w
if kw in text:
hit_w += w
hits.append(kw)
else:
miss.append(kw)
kw_score = hit_w / tot_w if tot_w else 0.0
# 3) 申报要素完整度
if item.elements:
filled = [e for e in item.elements
if str(dec.elements.get(e, "")).strip()]
missing = [e for e in item.elements if e not in filled]
elem_score = len(filled) / len(item.elements)
else:
missing, elem_score = [], 1.0
score = (self.w_name * sim_name
+ self.w_kw * kw_score
+ self.w_elem * elem_score)
detail = {
"name_sim": round(sim_name, 3),
"kw_score": round(kw_score, 3),
"elem_score": round(elem_score, 3),
"hit_keywords": hits,
"miss_keywords": miss,
"missing_elements": missing,
}
return score, detail
# ---------- 审核主流程 ----------
def audit(self, dec: DeclareItem, topn: int = 3) -> dict:
scored = []
for item in self.catalog:
s, d = self._score(item, dec)
scored.append({"item": item, "score": round(s, 4), "detail": d})
scored.sort(key=lambda x: (-x["score"], x["item"].code))
candidates = scored[:topn]
best = candidates[0] if candidates else None
declared = self.by_code.get(dec.declared_code)
cur_score, cur_detail = (0.0, None)
if declared is not None:
cur_score, cur_detail = self._score(declared, dec)
# ---------- 判定结论 ----------
if declared is None:
status, level = "编码无效", "error"
elif best and best["item"].code == declared.code:
if cur_score >= self.HIGH:
status, level = "匹配", "ok"
else:
status, level = "基本匹配(建议补充申报要素)", "warn"
else:
gap = (best["score"] if best else 0.0) - cur_score
if best and best["score"] >= self.MID and gap >= self.DELTA:
status, level = "疑似归类错误", "error"
elif cur_score < self.MID:
status, level = "匹配度过低,需人工复核", "error"
else:
status, level = "基本匹配(存在更贴合税号,建议复核)", "warn"
# ---------- 生成修改建议 ----------
tips = []
best_item = best["item"] if best else None
if declared is None:
tips.append(f"申报编码「{dec.declared_code}」不在税则库中,"
f"请确认是否为有效的 10 位税号。")
if best_item and best_item.code != dec.declared_code:
tips.append(
f"建议将海关编码由 {dec.declared_code} 修改为 "
f"{best_item.code}({best_item.name}):"
f"该税号匹配度 {best['score']:.0%},"
f"当前编码匹配度仅 {cur_score:.0%}。"
)
if dec.declared_code[:4] != best_item.code[:4]:
tips.append("两者前 4 位(品目)不同,归类差异较大,"
"请重点核对商品属性与归类依据。")
elif dec.declared_code[:6] != best_item.code[:6]:
tips.append("两者前 6 位(国际子目)不一致,"
"建议核对归类依据及海关归类决定。")
if best_item.note:
tips.append(f"归类提示:{best_item.note}")
elif best_item:
tips.append(f"编码与商品基本对应,当前匹配度 {cur_score:.0%}。")
# 要素 & 描述补全
ref_detail = cur_detail if declared is not None else (
best["detail"] if best else None)
if ref_detail:
if ref_detail["missing_elements"]:
tips.append("建议补充申报要素:"
+ "、".join(ref_detail["missing_elements"]))
if ref_detail["miss_keywords"]:
tips.append("建议在品名/规格型号中补充以下特征:"
+ "、".join(ref_detail["miss_keywords"][:6]))
# 候选接近时提示人工
if len(candidates) >= 2 and \
candidates[0]["score"] - candidates[1]["score"] < 0.05:
tips.append("最优与次优税号得分接近,建议人工确认归类。")
if not tips:
tips.append("无需修改。")
return {
"seq": dec.seq,
"name": dec.name,
"declared_code": dec.declared_code,
"status": status,
"level": level,
"current_score": round(cur_score, 4),
"candidates": [
{"code": c["item"].code, "name": c["item"].name,
"score": c["score"], "detail": c["detail"]}
for c in candidates
],
"suggestions": tips,
}
# ==========================================================
# 四、报告输出
# ==========================================================
LEVEL_TAG = {"ok": "[通过]", "warn": "[提醒]", "error": "[异常]"}
def format_report(result: dict) -> str:
lines = []
lines.append("=" * 68)
lines.append(f"第 {result['seq']} 项 {result['name']}")
lines.append(f" 申报编码 : {result['declared_code']} "
f"匹配度: {result['current_score']:.0%}")
lines.append(f" 审核结论 : {LEVEL_TAG[result['level']]} {result['status']}")
lines.append(" --- 候选税号 ---")
for i, c in enumerate(result["candidates"], 1):
mark = " <== 申报" if c["code"] == result["declared_code"] else ""
lines.append(f" {i}. {c['code']} {c['name']} "
f"[{c['score']:.0%}]{mark}")
d = c["detail"]
lines.append(f" 名称 {d['name_sim']:.2f} / "
f"关键词 {d['kw_score']:.2f} / "
f"要素 {d['elem_score']:.2f}"
+ (f" 命中: {'、'.join(d['hit_keywords'][:5])}"
if d["hit_keywords"] else ""))
lines.append(" --- 修改建议 ---")
for t in result["suggestions"]:
lines.append(f" · {t}")
return "\n".join(lines)
def audit_declaration(matcher: HSCodeMatcher,
items: Iterable[DeclareItem]) -> list:
results = [matcher.audit(it) for it in items]
for r in results:
print(format_report(r))
print("=" * 68)
return results
# ==========================================================
# 五、税则库加载(CSV)
# ==========================================================
def load_catalog_from_csv(path: str) -> list:
"""
CSV 表头:code,name,keywords,elements,unit,note
关键词 / 要素 多值用 | 分隔,例如:手机|智能手机|移动电话
"""
catalog = []
with open(path, encoding="utf-8-sig", newline="") as f:
for row in csv.DictReader(f):
catalog.append(HSItem(
code=row["code"].strip(),
name=row["name"].strip(),
keywords=tuple(k.strip() for k in
row.get("keywords", "").split("|") if k.strip()),
elements=tuple(k.strip() for k in
row.get("elements", "").split("|") if k.strip()),
unit=row.get("unit", "").strip(),
note=row.get("note", "").strip(),
))
return catalog
# ==========================================================
# 六、演示
# ==========================================================
if __name__ == "__main__":
# 注意:以下为演示用税则数据,真实业务请导入海关总署《税则》全量数据
CATALOG = [
HSItem(
code="8471300000",
name="便携式自动数据处理设备",
keywords=("笔记本电脑", "平板电脑", "便携式", "自动数据处理"),
elements=("品名", "品牌", "型号", "处理器", "内存", "屏幕尺寸"),
unit="台",
note="笔记本、平板等需注明是否含键盘、屏幕尺寸",
),
HSItem(
code="8517130000",
name="智能手机",
keywords=("手机", "智能手机", "移动电话", "蜂窝网络"),
elements=("品名", "品牌", "型号", "屏幕尺寸", "是否支持蜂窝网络"),
unit="台",
note="仅具通话/上网功能的手机归此税号,带卫星通信需另行确认",
),
HSItem(
code="6109100021",
name="棉制针织男式T恤衫",
keywords=("T恤", "T恤衫", "针织", "棉制", "男式"),
elements=("品名", "品牌", "型号", "成分含量", "织造方法", "性别"),
unit="件",
note="需注明针织/梭织、棉含量、男式/女式",
),
HSItem(
code="3926909090",
name="其他塑料制品",
keywords=("塑料", "塑料制品", "塑料配件", "塑料壳"),
elements=("品名", "品牌", "型号", "材质", "用途"),
unit="千克",
note="手机壳、保护套等塑料制零件归此税号",
),
HSItem(
code="4202210000",
name="以皮革面制的手提包",
keywords=("手提包", "皮包", "皮革", "女包"),
elements=("品名", "品牌", "型号", "材质", "款式"),
unit="个",
),
]
matcher = HSCodeMatcher(CATALOG)
# 模拟一张报关单上的 5 条商品
declaration = [
DeclareItem(
seq=1, name="笔记本电脑", declared_code="8471300000",
spec="14寸 16G+512G",
elements={"品名": "笔记本电脑", "品牌": "LENOVO",
"型号": "ThinkPad X1", "处理器": "i7",
"内存": "16G", "屏幕尺寸": "14英寸"},
),
DeclareItem(
seq=2, name="智能手机", declared_code="8517130000",
spec="6.1英寸",
elements={"品名": "智能手机", "品牌": "HUAWEI"},
),
DeclareItem(
seq=3, name="塑料手机壳", declared_code="8517130000", # 错误归类
spec="PC材质 保护用",
elements={"品名": "塑料手机壳", "品牌": "无",
"型号": "A100", "材质": "PC", "用途": "保护手机"},
),
DeclareItem(
seq=4, name="棉制针织男式T恤", declared_code="6109100021",
spec="100%棉 针织",
elements={"品名": "T恤衫", "品牌": "UNIQLO", "型号": "U-001",
"成分含量": "100%棉", "织造方法": "针织", "性别": "男式"},
),
DeclareItem(
seq=5, name="蓝牙耳机", declared_code="1234567890", # 无效编码
spec="无线 入耳式",
elements={"品名": "蓝牙耳机", "品牌": "小米"},
),
]
audit_declaration(matcher, declaration)
```
审核逻辑与输出解读
代码模拟了一张含5条商品的报关单,逐一演示匹配流程和判定结果。
· 打分机制:以税则库中每条税号为基准,从名称相似度(字符bigram Dice系数)、关键词命中率(按词长加权)和申报要素完整度三个维度综合打分,按权重0.35/0.45/0.20加权求和。
· 判定分级:根据得分与预设阈值比较,划分为“匹配”“基本匹配”“疑似归类错误”“匹配度过低”“编码无效”等状态,并区分通过、提醒、异常三个级别。
· 建议输出:每条商品会展示候选税号列表和具体修改建议,包括编码推荐、申报要素补充、特征词补全等,并提示人工复核的情形。
· 演示数据:内置5条税号,覆盖笔记本、手机、T恤、塑料制品和手提包,并构造了5条商品,包含正确匹配、归类错误、编码无效等典型场景。
---
优化建议: 演示用的税则库和报关单数据均为示例,您可以将CATALOG和declaration替换为真实数据,或通过load_catalog_from_csv函数导入标准税则CSV文件;同时可调整匹配阈值和高低置信度以适应不同商品的分类精度要求。
仅供参考用,可能存在一些问题?