Files
attachment_classifier/parser.py
Misaka_Company 2110e6f38c feat: add order-attachment LLM classifier
Co-Authored-By: Claude <noreply@anthropic.com>
2026-07-24 13:35:28 +08:00

187 lines
8.1 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""解析与清洗层。
职责边界很明确:
- LLM 只负责"识别"输出约定好的纯文本coarse模式2行fine模式3行
- 本模块只负责"验证格式是否正确 + 提取字段 + 组装成程序自己定义的结构"
不信任 LLM 输出的具体体现:
1. 逐行校验,任何一行不匹配预期正则就判定为格式错误,绝不"猜测式"兜底解析。
2. 枚举值必须完全匹配 prompts.py 里定义的枚举,出现模型编造的第三种值也判格式错误。
3. fine模式下细分类目与大类的归属关系必须自洽"针型阀"必须伴随"配件")。
4. 格式错误交给上层去做重试,而不是在这里勉强凑出一个结果。
enable_other 参数贯穿本模块的校验函数:是否把"其他"当作合法值,取决于这次
分类调用传入的 enable_other必须和生成该输出时 prompts.get_prompt 用的
取值保持一致——由 classifier.py 负责把同一个值同时传给两边。
types 字段说明(对外唯一的类型字段,不再单独暴露 fine_types
- coarse 模式:大类列表,如 ["资料", "配件"]。
- fine 模式:每一项都是"大类:细分类目",如 ["资料:说明书", "配件:针型阀"]
由第二行(大类)和第三行(细分类目)校验通过后拼接而成;启用"其他"时,
兜底项会体现为 "其他:其他",格式与其余项保持一致。
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from prompts import all_fine_types, fine_type_to_category, valid_categories
_LINE1_RE = re.compile(r"^是否携带附件[:]\s*(是|否)\s*$")
_LINE2_RE = re.compile(r"^附件类型[:]\s*(.*)$")
_LINE3_RE = re.compile(r"^具体分类[:]\s*(.*)$")
class FormatError(Exception):
"""LLM 原始输出不符合约定格式,调用方应据此触发重试。"""
@dataclass
class ParsedResult:
"""分类结果。
types 是对外唯一的类型字段coarse 模式下是大类列表fine 模式下每一项
都已经是"大类:细分类目"的组合字符串。不再有单独的 fine_types 字段——
fine 模式的"大类"信息已经内嵌在每个元素里,无需再查第二份数据。
"""
has_attachment: bool
types: list[str] = field(default_factory=list)
def _strip_markdown_fence(raw_text: str) -> str:
cleaned = raw_text.strip()
return re.sub(r"^```[a-zA-Z]*\n?|```$", "", cleaned).strip()
def _parse_line1(line: str) -> bool:
m = _LINE1_RE.match(line)
if not m:
raise FormatError(f"第一行格式不匹配 '是否携带附件: 是/否': {line!r}")
return m.group(1) == ""
def _parse_line2(line: str, has_attachment: bool, categories: set[str]) -> list[str]:
m = _LINE2_RE.match(line)
if not m:
raise FormatError(f"第二行格式不匹配 '附件类型: ...': {line!r}")
types_raw = m.group(1).strip()
if not has_attachment:
if types_raw:
raise FormatError(
f"第一行为'',但第二行附件类型非空: {types_raw!r}(模型输出自相矛盾)"
)
return []
if not types_raw:
raise FormatError("第一行为'',但第二行附件类型为空(模型输出自相矛盾)")
types = [t.strip() for t in types_raw.split(",") if t.strip()]
invalid = [t for t in types if t not in categories]
if invalid:
raise FormatError(f"附件类型出现非法值 {invalid},合法值仅为 {categories}")
if len(types) != len(set(types)):
raise FormatError(f"附件类型出现重复: {types}")
return types
def parse_llm_output_coarse(raw_text: str, enable_other: bool = False) -> ParsedResult:
"""校验并解析 coarse 模式的两行输出。
enable_other 必须和生成该输出时 prompts.get_prompt 使用的取值一致,
否则"其他"要么被误判为非法值拒绝,要么被误放行。
"""
if raw_text is None:
raise FormatError("LLM 返回内容为 None")
categories = valid_categories(enable_other)
cleaned = _strip_markdown_fence(raw_text)
lines = [ln.strip() for ln in cleaned.splitlines() if ln.strip()]
if len(lines) != 2:
raise FormatError(f"coarse模式预期恰好2行有效内容实际得到{len(lines)}行: {lines!r}")
has_attachment = _parse_line1(lines[0])
types = _parse_line2(lines[1], has_attachment, categories)
return ParsedResult(has_attachment=has_attachment, types=types)
def parse_llm_output_fine(raw_text: str, enable_other: bool = False) -> ParsedResult:
"""校验并解析 fine 模式的三行输出,并把"大类+细分"合并进最终的 types。
在 coarse 校验规则的基础上,额外校验:
- 第三行每个细分类目都必须在生效的细分类目全集内(是否含"其他"取决于
enable_other
- 若第一行为"",第三行必须为空;若为"",第三行至少要有一项。
- 第三行不能有重复项。
- 交叉自洽:第三行每个细分类目所属的大类,必须出现在第二行的大类列表里
(例如选了"针型阀"但第二行没有"配件",视为模型输出自相矛盾)。
全部校验通过后,把第三行的每个细分类目和它所属的大类拼成
"大类:细分类目",作为最终 types 返回——不再单独暴露"大类列表"
"细分类目列表"两份数据。
"""
if raw_text is None:
raise FormatError("LLM 返回内容为 None")
categories = valid_categories(enable_other)
fine_types_enum = all_fine_types(enable_other)
fine_to_category = fine_type_to_category(enable_other)
cleaned = _strip_markdown_fence(raw_text)
lines = [ln.strip() for ln in cleaned.splitlines() if ln.strip()]
if len(lines) != 3:
raise FormatError(f"fine模式预期恰好3行有效内容实际得到{len(lines)}行: {lines!r}")
has_attachment = _parse_line1(lines[0])
line2_types = _parse_line2(lines[1], has_attachment, categories)
m3 = _LINE3_RE.match(lines[2])
if not m3:
raise FormatError(f"第三行格式不匹配 '具体分类: ...': {lines[2]!r}")
fine_raw = m3.group(1).strip()
if not has_attachment:
if fine_raw:
raise FormatError(
f"第一行为'',但第三行具体分类非空: {fine_raw!r}(模型输出自相矛盾)"
)
return ParsedResult(has_attachment=False, types=[])
if not fine_raw:
raise FormatError("第一行为'',但第三行具体分类为空(模型输出自相矛盾)")
fine_types = [t.strip() for t in fine_raw.split(",") if t.strip()]
invalid_fine = [t for t in fine_types if t not in fine_types_enum]
if invalid_fine:
raise FormatError(
f"具体分类出现非法值 {invalid_fine}不在枚举内enable_other={enable_other}"
)
if len(fine_types) != len(set(fine_types)):
raise FormatError(f"具体分类出现重复: {fine_types}")
# 交叉自洽校验:每个细分类目对应的大类,必须出现在第二行里
implied_categories = {fine_to_category[t] for t in fine_types}
missing_categories = implied_categories - set(line2_types)
if missing_categories:
raise FormatError(
f"具体分类 {fine_types} 隐含大类 {implied_categories}"
f"但第二行附件类型只有 {line2_types},缺少 {missing_categories}(模型输出自相矛盾)"
)
# 合并为最终的 "大类:细分类目" 列表,替代原来分开暴露的 types/fine_types
combined = [f"{fine_to_category[t]}:{t}" for t in fine_types]
return ParsedResult(has_attachment=True, types=combined)
def parse_llm_output(raw_text: str, mode: str = "coarse", enable_other: bool = False) -> ParsedResult:
"""按模式分派到对应的解析函数。mode 只接受 'coarse''fine'"""
if mode == "coarse":
return parse_llm_output_coarse(raw_text, enable_other=enable_other)
if mode == "fine":
return parse_llm_output_fine(raw_text, enable_other=enable_other)
raise ValueError(f"不支持的分类模式: {mode!r},只能是 'coarse''fine'")