根据货单自动分类海关编码的实现方案
以下提供一套可运行的代码实现,核心思路是:以HS编码描述库为基准,通过文本相似度匹配将货单中的商品描述映射到对应的海关编码,并输出置信度与候选结果供人工复核。
方案架构
```
货单条目(商品名称+规格+用途)
│
▼
┌───────────────────┐
│ 文本预处理 │ 清洗、分词、去停用词
└────────┬──────────┘
▼
┌───────────────────┐
│ 候选编码检索 │ TF-IDF / 模糊搜索 → Top-K候选
└────────┬──────────┘
▼
┌───────────────────┐
│ 排序与置信度评分 │ 相似度排序,阈值判断
└────────┬──────────┘
▼
输出:主编码 + 备选编码 + 置信度
```
代码实现
环境准备
```bash
pip install pyhscodes pandas scikit-learn rapidfuzz
```
核心代码
```python
"""
货单 → 海关编码自动分类
基于 HS 编码描述库的文本相似度匹配
"""
import re
import pandas as pd
from dataclasses import dataclass, field
from typing import Optional
from rapidfuzz import fuzz, process
import pyhscodes
# ==================== 数据结构 ====================
@dataclass
class LineItem:
"""货单中的一个条目"""
raw_name: str # 原始商品名称
specification: str = "" # 规格/型号
usage: str = "" # 用途
material: str = "" # 材质
@property
def search_text(self) -> str:
"""组合所有可用字段作为检索文本"""
parts = [self.raw_name, self.specification, self.usage, self.material]
return " ".join(p for p in parts if p).strip()
@dataclass
class ClassificationResult:
"""分类结果"""
line_item: LineItem
primary_code: Optional[str] = None
primary_desc: Optional[str] = None
confidence: float = 0.0
candidates: list = field(default_factory=list) # [(code, desc, score), ...]
needs_review: bool = True # 默认需要复核
def to_dict(self) -> dict:
return {
"商品名称": self.line_item.raw_name,
"主编码": self.primary_code or "未匹配",
"编码描述": self.primary_desc or "",
"置信度": round(self.confidence, 3),
"需人工复核": self.needs_review,
"备选编码": [
{"编码": c[0], "描述": c[1], "得分": round(c[2], 3)}
for c in self.candidates[:3]
],
}
# ==================== HS 编码库加载 ====================
class HSCodeIndex:
"""HS 编码索引,支持模糊检索"""
def __init__(self, min_level: str = "4"):
"""
min_level: 最低匹配层级
"2" — 章(约100条)
"4" — 品目(约1200条) ← 推荐起点
"6" — 子目(约5400条)
"""
self.codes: list[str] = []
self.descriptions: list[str] = []
self._build_index(min_level)
def _build_index(self, min_level: str):
level_map = {"2": 2, "4": 4, "6": 6}
target_len = level_map.get(min_level, 4)
for hscode in pyhscodes.hscodes:
code_str = hscode.hscode
if len(code_str) >= target_len:
self.codes.append(code_str)
self.descriptions.append(hscode.description or "")
print(f"[HSCodeIndex] 已加载 {len(self.codes)} 条编码 (层级: {min_level}位)")
def search(self, query: str, top_k: int = 5) -> list:
"""模糊搜索,返回 [(code, description, score), ...]"""
if not query.strip():
return []
# 使用 token_set_ratio 对部分匹配更友好
results = process.extract(
query,
self.descriptions,
scorer=fuzz.token_set_ratio,
limit=top_k * 2, # 多取一些再做去重
)
seen = set()
final = []
for desc, score, idx in results:
code = self.codes[idx]
if code in seen:
continue
seen.add(code)
final.append((code, self.descriptions[idx], score / 100.0))
if len(final) >= top_k:
break
return final
# ==================== 分类器 ====================
class HSCodeClassifier:
"""货单条目 → HS 编码分类器"""
def __init__(
self,
index: HSCodeIndex,
confidence_threshold: float = 0.55,
top_k: int = 5,
):
self.index = index
self.confidence_threshold = confidence_threshold
self.top_k = top_k
# 常见干扰词,检索前去除
self._stopwords = {
"的", "和", "与", "及", "或", "等", "型", "号", "款",
"规格", "型号", "用途", "材质", "品名", "商品", "产品",
}
def _clean_text(self, text: str) -> str:
"""清洗文本:去标点、统一分隔符、去停用词"""
text = re.sub(r"[,,。;;、/\\|]+", " ", text)
tokens = text.split()
tokens = [t for t in tokens if t not in self._stopwords and len(t) >= 1]
return " ".join(tokens)
def classify(self, item: LineItem) -> ClassificationResult:
"""对单个货单条目进行分类"""
query = self._clean_text(item.search_text)
candidates = self.index.search(query, top_k=self.top_k)
result = ClassificationResult(line_item=item, candidates=candidates)
if not candidates:
result.needs_review = True
return result
best_code, best_desc, best_score = candidates[0]
result.primary_code = best_code
result.primary_desc = best_desc
result.confidence = best_score
result.needs_review = best_score < self.confidence_threshold
# 若第一名与第二名差距过小,也标记复核
if len(candidates) >= 2:
second_score = candidates[1][2]
if best_score - second_score < 0.08:
result.needs_review = True
return result
def classify_batch(self, items: list[LineItem]) -> pd.DataFrame:
"""批量分类,返回 DataFrame"""
results = [self.classify(item) for item in items]
rows = [r.to_dict() for r in results]
df = pd.DataFrame(rows)
total = len(df)
review_count = df["需人工复核"].sum()
print(f"\n分类完成:共 {total} 条,需复核 {review_count} 条 "
f"({review_count/total*100:.1f}%)")
return df
# ==================== 使用示例 ====================
if __name__ == "__main__":
# 1. 构建索引(品目层级,约1200条,平衡覆盖与速度)
index = HSCodeIndex(min_level="4")
# 2. 初始化分类器
classifier = HSCodeClassifier(
index=index,
confidence_threshold=0.55,
top_k=5,
)
# 3. 模拟货单条目
line_items = [
LineItem(
raw_name="不锈钢保温杯",
specification="500ml 双层真空",
usage="日常饮水",
material="304不锈钢",
),
LineItem(
raw_name="蓝牙无线耳机",
specification="TWS 入耳式",
usage="手机音频",
),
LineItem(
raw_name="塑料收纳箱",
specification="60L 带轮",
material="PP塑料",
),
LineItem(
raw_name="纯棉男士T恤",
specification="短袖 圆领",
material="100%棉",
),
LineItem(
raw_name="锂电池充电宝",
specification="20000mAh",
usage="手机充电",
),
]
# 4. 执行分类
df = classifier.classify_batch(line_items)
# 5. 输出结果
pd.set_option("display.max_colwidth", 40)
pd.set_option("display.width", 200)
for _, row in df.iterrows():
print(f"\n{'='*60}")
print(f"商品: {row['商品名称']}")
print(f"主编码: {row['主编码']} | 描述: {row['编码描述']}")
print(f"置信度: {row['置信度']} | 需复核: {row['需人工复核']}")
if row["备选编码"]:
print("备选:")
for alt in row["备选编码"]:
print(f" → {alt['编码']} {alt['描述']} (得分 {alt['得分']})")
```
运行效果说明
代码使用rapidfuzz的token_set_ratio进行匹配,它对中文短文本的部分匹配有较好的鲁棒性。以“不锈钢保温杯”为例,系统会在品目描述中检索,可能命中7323.93(钢铁制餐桌、厨房或其他家用器具及零件)或9617.00(保温瓶及其他真空容器)等候选,并按相似度排序输出。
HSCodeIndex默认加载4位品目层级(约1200条),这是实践中的推荐起点:章层级(2位)太粗,子目层级(6位)描述文本冗长且区分度低,品目层级在覆盖面和可匹配性之间取得较好平衡。
关键工程考量
置信度阈值与“窄差距复核” 。代码中设置了两个触发人工复核的条件:绝对置信度低于0.55,或第一名与第二名得分差距小于0.08。后者针对的是“两个编码都像,难以区分”的场景,这类情况在海关归类中恰恰是最需要人工判断的(如“钢铁制”与“铝制”的材质区分)。
HS编码版本管理。中国现行版本为《中华人民共和国进出口税则(2026)》,自2026年1月1日起实施。实际生产系统中,编码库应定期同步税则更新。pyhscodes库提供的是WCO国际HS编码(6位),中国10位编码需要额外对接税则数据源。
从规则匹配到语义匹配的进阶路径。当前方案基于文本相似度,对于描述规范、用词标准的货单效果较好。如果货单描述口语化严重或包含大量非标准简称,建议引入语义向量检索(如bge-m3嵌入模型),以“商品特征”而非“字面重合”为匹配依据。更进一步,可以参考ReAct Agent架构,在信息不完备时主动追问缺失的归类要素(如材质、用途),而非强行给出编码。
输出格式建议。实际报关场景中,建议输出“主编码 + 置信度 + 归类依据条文 + 备选编码”四元组,低置信度结果自动转入人工复核队列,而不是直接用于申报。
仅供模型参考用,可能存在问题?