根据货单自动分类海关编码的实现方案以下提供一套可运行的代码实现核心思路是以HS编码描述库为基准通过文本相似度匹配将货单中的商品描述映射到对应的海关编码并输出置信度与候选结果供人工复核。方案架构货单条目商品名称规格用途│▼┌───────────────────┐│ 文本预处理 │ 清洗、分词、去停用词└────────┬──────────┘▼┌───────────────────┐│ 候选编码检索 │ TF-IDF / 模糊搜索 → Top-K候选└────────┬──────────┘▼┌───────────────────┐│ 排序与置信度评分 │ 相似度排序阈值判断└────────┬──────────┘▼输出主编码 备选编码 置信度代码实现环境准备bashpip install pyhscodes pandas scikit-learn rapidfuzz核心代码python货单 → 海关编码自动分类基于 HS 编码描述库的文本相似度匹配import reimport pandas as pdfrom dataclasses import dataclass, fieldfrom typing import Optionalfrom rapidfuzz import fuzz, processimport pyhscodes# 数据结构 dataclassclass LineItem:货单中的一个条目raw_name: str # 原始商品名称specification: str # 规格/型号usage: str # 用途material: str # 材质propertydef search_text(self) - str:组合所有可用字段作为检索文本parts [self.raw_name, self.specification, self.usage, self.material]return .join(p for p in parts if p).strip()dataclassclass ClassificationResult:分类结果line_item: LineItemprimary_code: Optional[str] Noneprimary_desc: Optional[str] Noneconfidence: float 0.0candidates: list field(default_factorylist) # [(code, desc, score), ...]needs_review: bool True # 默认需要复核def to_dict(self) - dict:return {商品名称: self.line_item.raw_name,主编码: self.primary_code or 未匹配,编码描述: self.primary_desc or ,置信度: round(self.confidence, 3),需人工复核: self.needs_review,备选编码: [{编码: c[0], 描述: c[1], 得分: round(c[2], 3)}for c in self.candidates[:3]],}# HS 编码库加载 class HSCodeIndex:HS 编码索引支持模糊检索def __init__(self, min_level: str 4):min_level: 最低匹配层级2 — 章约100条4 — 品目约1200条 ← 推荐起点6 — 子目约5400条self.codes: list[str] []self.descriptions: list[str] []self._build_index(min_level)def _build_index(self, min_level: str):level_map {2: 2, 4: 4, 6: 6}target_len level_map.get(min_level, 4)for hscode in pyhscodes.hscodes:code_str hscode.hscodeif len(code_str) target_len:self.codes.append(code_str)self.descriptions.append(hscode.description or )print(f[HSCodeIndex] 已加载 {len(self.codes)} 条编码 (层级: {min_level}位))def search(self, query: str, top_k: int 5) - list:模糊搜索返回 [(code, description, score), ...]if not query.strip():return []# 使用 token_set_ratio 对部分匹配更友好results process.extract(query,self.descriptions,scorerfuzz.token_set_ratio,limittop_k * 2, # 多取一些再做去重)seen set()final []for desc, score, idx in results:code self.codes[idx]if code in seen:continueseen.add(code)final.append((code, self.descriptions[idx], score / 100.0))if len(final) top_k:breakreturn final# 分类器 class HSCodeClassifier:货单条目 → HS 编码分类器def __init__(self,index: HSCodeIndex,confidence_threshold: float 0.55,top_k: int 5,):self.index indexself.confidence_threshold confidence_thresholdself.top_k top_k# 常见干扰词检索前去除self._stopwords {的, 和, 与, 及, 或, 等, 型, 号, 款,规格, 型号, 用途, 材质, 品名, 商品, 产品,}def _clean_text(self, text: str) - str:清洗文本去标点、统一分隔符、去停用词text re.sub(r[,。;、/\\|], , text)tokens text.split()tokens [t for t in tokens if t not in self._stopwords and len(t) 1]return .join(tokens)def classify(self, item: LineItem) - ClassificationResult:对单个货单条目进行分类query self._clean_text(item.search_text)candidates self.index.search(query, top_kself.top_k)result ClassificationResult(line_itemitem, candidatescandidates)if not candidates:result.needs_review Truereturn resultbest_code, best_desc, best_score candidates[0]result.primary_code best_coderesult.primary_desc best_descresult.confidence best_scoreresult.needs_review best_score self.confidence_threshold# 若第一名与第二名差距过小也标记复核if len(candidates) 2:second_score candidates[1][2]if best_score - second_score 0.08:result.needs_review Truereturn resultdef classify_batch(self, items: list[LineItem]) - pd.DataFrame:批量分类返回 DataFrameresults [self.classify(item) for item in items]rows [r.to_dict() for r in results]df pd.DataFrame(rows)total len(df)review_count df[需人工复核].sum()print(f\n分类完成共 {total} 条需复核 {review_count} 条 f({review_count/total*100:.1f}%))return df# 使用示例 if __name__ __main__:# 1. 构建索引品目层级约1200条平衡覆盖与速度index HSCodeIndex(min_level4)# 2. 初始化分类器classifier HSCodeClassifier(indexindex,confidence_threshold0.55,top_k5,)# 3. 模拟货单条目line_items [LineItem(raw_name不锈钢保温杯,specification500ml 双层真空,usage日常饮水,material304不锈钢,),LineItem(raw_name蓝牙无线耳机,specificationTWS 入耳式,usage手机音频,),LineItem(raw_name塑料收纳箱,specification60L 带轮,materialPP塑料,),LineItem(raw_name纯棉男士T恤,specification短袖 圆领,material100%棉,),LineItem(raw_name锂电池充电宝,specification20000mAh,usage手机充电,),]# 4. 执行分类df classifier.classify_batch(line_items)# 5. 输出结果pd.set_option(display.max_colwidth, 40)pd.set_option(display.width, 200)for _, row in df.iterrows():print(f\n{*60})print(f商品: {row[商品名称]})print(f主编码: {row[主编码]} | 描述: {row[编码描述]})print(f置信度: {row[置信度]} | 需复核: {row[需人工复核]})if row[备选编码]:print(备选:)for alt in row[备选编码]:print(f → {alt[编码]} {alt[描述]} (得分 {alt[得分]}))运行效果说明代码使用rapidfuzz的token_set_ratio进行匹配它对中文短文本的部分匹配有较好的鲁棒性。以“不锈钢保温杯”为例系统会在品目描述中检索可能命中7323.93钢铁制餐桌、厨房或其他家用器具及零件或9617.00保温瓶及其他真空容器等候选并按相似度排序输出。HSCodeIndex默认加载4位品目层级约1200条这是实践中的推荐起点章层级2位太粗子目层级6位描述文本冗长且区分度低品目层级在覆盖面和可匹配性之间取得较好平衡。关键工程考量置信度阈值与“窄差距复核” 。代码中设置了两个触发人工复核的条件绝对置信度低于0.55或第一名与第二名得分差距小于0.08。后者针对的是“两个编码都像难以区分”的场景这类情况在海关归类中恰恰是最需要人工判断的如“钢铁制”与“铝制”的材质区分。HS编码版本管理。中国现行版本为《中华人民共和国进出口税则2026》自2026年1月1日起实施。实际生产系统中编码库应定期同步税则更新。pyhscodes库提供的是WCO国际HS编码6位中国10位编码需要额外对接税则数据源。从规则匹配到语义匹配的进阶路径。当前方案基于文本相似度对于描述规范、用词标准的货单效果较好。如果货单描述口语化严重或包含大量非标准简称建议引入语义向量检索如bge-m3嵌入模型以“商品特征”而非“字面重合”为匹配依据。更进一步可以参考ReAct Agent架构在信息不完备时主动追问缺失的归类要素如材质、用途而非强行给出编码。输出格式建议。实际报关场景中建议输出“主编码 置信度 归类依据条文 备选编码”四元组低置信度结果自动转入人工复核队列而不是直接用于申报。仅供模型参考用可能存在问题
阅读完成 · 觉得有帮助?