feat: add order-attachment LLM classifier

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
Misaka_Company
2026-07-24 13:35:28 +08:00
commit 2110e6f38c
12 changed files with 1745 additions and 0 deletions

186
parser.py Normal file
View File

@@ -0,0 +1,186 @@
# -*- coding: utf-8 -*-
"""解析与清洗层。
职责边界很明确:
- LLM 只负责"识别"输出约定好的纯文本coarse模式2行fine模式3行
- 本模块只负责"验证格式是否正确 + 提取字段 + 组装成程序自己定义的结构"
不信任 LLM 输出的具体体现:
1. 逐行校验,任何一行不匹配预期正则就判定为格式错误,绝不"猜测式"兜底解析。
2. 枚举值必须完全匹配 prompts.py 里定义的枚举,出现模型编造的第三种值也判格式错误。
3. fine模式下细分类目与大类的归属关系必须自洽"针型阀"必须伴随"配件")。
4. 格式错误交给上层去做重试,而不是在这里勉强凑出一个结果。
enable_other 参数贯穿本模块的校验函数:是否把"其他"当作合法值,取决于这次
分类调用传入的 enable_other必须和生成该输出时 prompts.get_prompt 用的
取值保持一致——由 classifier.py 负责把同一个值同时传给两边。
types 字段说明(对外唯一的类型字段,不再单独暴露 fine_types
- coarse 模式:大类列表,如 ["资料", "配件"]。
- fine 模式:每一项都是"大类:细分类目",如 ["资料:说明书", "配件:针型阀"]
由第二行(大类)和第三行(细分类目)校验通过后拼接而成;启用"其他"时,
兜底项会体现为 "其他:其他",格式与其余项保持一致。
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from prompts import all_fine_types, fine_type_to_category, valid_categories
_LINE1_RE = re.compile(r"^是否携带附件[:]\s*(是|否)\s*$")
_LINE2_RE = re.compile(r"^附件类型[:]\s*(.*)$")
_LINE3_RE = re.compile(r"^具体分类[:]\s*(.*)$")
class FormatError(Exception):
"""LLM 原始输出不符合约定格式,调用方应据此触发重试。"""
@dataclass
class ParsedResult:
"""分类结果。
types 是对外唯一的类型字段coarse 模式下是大类列表fine 模式下每一项
都已经是"大类:细分类目"的组合字符串。不再有单独的 fine_types 字段——
fine 模式的"大类"信息已经内嵌在每个元素里,无需再查第二份数据。
"""
has_attachment: bool
types: list[str] = field(default_factory=list)
def _strip_markdown_fence(raw_text: str) -> str:
cleaned = raw_text.strip()
return re.sub(r"^```[a-zA-Z]*\n?|```$", "", cleaned).strip()
def _parse_line1(line: str) -> bool:
m = _LINE1_RE.match(line)
if not m:
raise FormatError(f"第一行格式不匹配 '是否携带附件: 是/否': {line!r}")
return m.group(1) == ""
def _parse_line2(line: str, has_attachment: bool, categories: set[str]) -> list[str]:
m = _LINE2_RE.match(line)
if not m:
raise FormatError(f"第二行格式不匹配 '附件类型: ...': {line!r}")
types_raw = m.group(1).strip()
if not has_attachment:
if types_raw:
raise FormatError(
f"第一行为'',但第二行附件类型非空: {types_raw!r}(模型输出自相矛盾)"
)
return []
if not types_raw:
raise FormatError("第一行为'',但第二行附件类型为空(模型输出自相矛盾)")
types = [t.strip() for t in types_raw.split(",") if t.strip()]
invalid = [t for t in types if t not in categories]
if invalid:
raise FormatError(f"附件类型出现非法值 {invalid},合法值仅为 {categories}")
if len(types) != len(set(types)):
raise FormatError(f"附件类型出现重复: {types}")
return types
def parse_llm_output_coarse(raw_text: str, enable_other: bool = False) -> ParsedResult:
"""校验并解析 coarse 模式的两行输出。
enable_other 必须和生成该输出时 prompts.get_prompt 使用的取值一致,
否则"其他"要么被误判为非法值拒绝,要么被误放行。
"""
if raw_text is None:
raise FormatError("LLM 返回内容为 None")
categories = valid_categories(enable_other)
cleaned = _strip_markdown_fence(raw_text)
lines = [ln.strip() for ln in cleaned.splitlines() if ln.strip()]
if len(lines) != 2:
raise FormatError(f"coarse模式预期恰好2行有效内容实际得到{len(lines)}行: {lines!r}")
has_attachment = _parse_line1(lines[0])
types = _parse_line2(lines[1], has_attachment, categories)
return ParsedResult(has_attachment=has_attachment, types=types)
def parse_llm_output_fine(raw_text: str, enable_other: bool = False) -> ParsedResult:
"""校验并解析 fine 模式的三行输出,并把"大类+细分"合并进最终的 types。
在 coarse 校验规则的基础上,额外校验:
- 第三行每个细分类目都必须在生效的细分类目全集内(是否含"其他"取决于
enable_other
- 若第一行为"",第三行必须为空;若为"",第三行至少要有一项。
- 第三行不能有重复项。
- 交叉自洽:第三行每个细分类目所属的大类,必须出现在第二行的大类列表里
(例如选了"针型阀"但第二行没有"配件",视为模型输出自相矛盾)。
全部校验通过后,把第三行的每个细分类目和它所属的大类拼成
"大类:细分类目",作为最终 types 返回——不再单独暴露"大类列表"
"细分类目列表"两份数据。
"""
if raw_text is None:
raise FormatError("LLM 返回内容为 None")
categories = valid_categories(enable_other)
fine_types_enum = all_fine_types(enable_other)
fine_to_category = fine_type_to_category(enable_other)
cleaned = _strip_markdown_fence(raw_text)
lines = [ln.strip() for ln in cleaned.splitlines() if ln.strip()]
if len(lines) != 3:
raise FormatError(f"fine模式预期恰好3行有效内容实际得到{len(lines)}行: {lines!r}")
has_attachment = _parse_line1(lines[0])
line2_types = _parse_line2(lines[1], has_attachment, categories)
m3 = _LINE3_RE.match(lines[2])
if not m3:
raise FormatError(f"第三行格式不匹配 '具体分类: ...': {lines[2]!r}")
fine_raw = m3.group(1).strip()
if not has_attachment:
if fine_raw:
raise FormatError(
f"第一行为'',但第三行具体分类非空: {fine_raw!r}(模型输出自相矛盾)"
)
return ParsedResult(has_attachment=False, types=[])
if not fine_raw:
raise FormatError("第一行为'',但第三行具体分类为空(模型输出自相矛盾)")
fine_types = [t.strip() for t in fine_raw.split(",") if t.strip()]
invalid_fine = [t for t in fine_types if t not in fine_types_enum]
if invalid_fine:
raise FormatError(
f"具体分类出现非法值 {invalid_fine}不在枚举内enable_other={enable_other}"
)
if len(fine_types) != len(set(fine_types)):
raise FormatError(f"具体分类出现重复: {fine_types}")
# 交叉自洽校验:每个细分类目对应的大类,必须出现在第二行里
implied_categories = {fine_to_category[t] for t in fine_types}
missing_categories = implied_categories - set(line2_types)
if missing_categories:
raise FormatError(
f"具体分类 {fine_types} 隐含大类 {implied_categories}"
f"但第二行附件类型只有 {line2_types},缺少 {missing_categories}(模型输出自相矛盾)"
)
# 合并为最终的 "大类:细分类目" 列表,替代原来分开暴露的 types/fine_types
combined = [f"{fine_to_category[t]}:{t}" for t in fine_types]
return ParsedResult(has_attachment=True, types=combined)
def parse_llm_output(raw_text: str, mode: str = "coarse", enable_other: bool = False) -> ParsedResult:
"""按模式分派到对应的解析函数。mode 只接受 'coarse''fine'"""
if mode == "coarse":
return parse_llm_output_coarse(raw_text, enable_other=enable_other)
if mode == "fine":
return parse_llm_output_fine(raw_text, enable_other=enable_other)
raise ValueError(f"不支持的分类模式: {mode!r},只能是 'coarse''fine'")