feat(extractor): Add Excel conversion and post-processing capabilities

Add comprehensive post-processing features to convert downloaded Excel files
into structured data and merge them into a single output file.

New modules:
- extractor_core.py: Stateless pure functions for web operations
- excel_converter.py: Excel to DataFrame conversion utility
- tests/test_extractor_real.py: Real data extraction test suite

Enhanced features:
- post_process_downloads(): Convert and merge multiple Excel files
- extract_and_process(): Complete workflow in single call
- cleanup_temp_files(): Optional cleanup of temporary downloaded files
- Field name mapping for standardized output columns

Dependencies:
- pandas>=2.0.0 for data manipulation
- openpyxl>=3.1.0 for Excel file handling

Documentation:
- Updated CLAUDE.md with new module references
- Added API documentation for extractor components

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Misaka_Company
2026-03-27 14:34:56 +08:00
parent 1b984f5cfd
commit c3bbc919a5
9 changed files with 1734 additions and 3 deletions

View File

@@ -0,0 +1,280 @@
"""
Excel 报表数据转换工具组件
将 Excel 报表数据转换为数据库记录形式
"""
import pandas as pd
import openpyxl
from typing import List, Dict, Optional
import os
class ExcelConverter:
"""Excel 报表数据转换器"""
# 字段名称映射(解决字段名冲突)
FIELD_NAME_MAPPING = {"计划数量": "产品计划数量", "单位": "产品单位"}
def __init__(self, verbose: bool = True):
"""
初始化转换器
Args:
verbose: 是否打印详细日志
"""
self.verbose = verbose
def _print(self, *args, **kwargs):
"""打印日志(如果 verbose=True"""
if self.verbose:
print(*args, **kwargs)
def convert(self, input_file: str, output_file: str = None) -> pd.DataFrame:
"""
转换 Excel 文件
Args:
input_file: 输入文件路径
output_file: 输出文件路径(可选,不指定则不保存)
Returns:
转换后的 DataFrame
"""
# 处理输出文件名
if output_file:
output_file = self._handle_output_file(output_file)
# 读取工作表
wb = openpyxl.load_workbook(input_file)
ws = wb.active
# 解析订单数据
orders = self._parse_sheet(ws)
# 转换为 DataFrame
df = self._convert_to_dataframe(orders)
if not df.empty:
# 保存文件
if output_file:
df.to_excel(output_file, index=False)
# 打印汇总报告
self._print("=" * 60)
self._print("转换完成")
self._print("=" * 60)
self._print(f"订单数: {len(orders)}")
self._print(f"数据行数: {len(df)}")
self._print(f"输出文件: {output_file if output_file else 'N/A'}")
self._print("=" * 60)
return df
def _handle_output_file(self, output_file: str) -> str:
"""
处理输出文件,如果文件存在则尝试删除
Args:
output_file: 输出文件路径
Returns:
实际使用的输出文件路径
"""
if os.path.exists(output_file):
try:
os.remove(output_file)
except PermissionError:
self._print(f"警告: 无法删除 {output_file},可能文件被其他程序打开")
# 修改文件名
base, ext = os.path.splitext(output_file)
output_file = f"{base}_new{ext}"
return output_file
def _parse_sheet(self, ws) -> List[Dict]:
"""
解析一个工作表,返回所有订单的数据
每个订单包含:
- order_info: 订单头信息(包括页脚)
- materials: 物料数据列表
Args:
ws: openpyxl 工作表对象
Returns:
订单列表
"""
orders = []
all_rows = list(ws.iter_rows(values_only=True))
# 逐行扫描,按订单结构解析
i = 0
while i < len(all_rows):
row = all_rows[i]
# 检查是否是订单标题行
if row and "离散备料计划" in str(row[0]):
# 解析订单头信息接下来的4行
order_info = {}
for j in range(1, 5):
if i + j < len(all_rows) and all_rows[i + j]:
self._parse_header_row(all_rows[i + j], order_info)
# 跳过空行,找到表格标题行
table_row = i + 5
while table_row < len(all_rows) and (
not all_rows[table_row] or not all_rows[table_row][0]
):
table_row += 1
# 检查是否是表格标题行
if (
table_row < len(all_rows)
and all_rows[table_row]
and all_rows[table_row][0] == "序号"
):
# 检查表头下一行是否为空,判断是否存在数据
next_row = table_row + 1
is_empty_row = (
next_row < len(all_rows)
and all_rows[next_row]
and all(
cell is None or str(cell).strip() == ""
for cell in all_rows[next_row]
)
)
if is_empty_row:
# 没有数据,查找页脚信息
materials = []
footer_info = {}
data_row = next_row + 1
while data_row < len(all_rows) and all_rows[data_row]:
if all_rows[data_row][0] and (
"制单人" in str(all_rows[data_row][0])
or "打印人" in str(all_rows[data_row][0])
):
self._parse_header_row(all_rows[data_row], footer_info)
if (
data_row + 1 < len(all_rows)
and all_rows[data_row + 1]
):
self._parse_header_row(
all_rows[data_row + 1], footer_info
)
break
data_row += 1
orders.append(
{
"order_info": {**order_info, **footer_info},
"materials": materials,
}
)
else:
# 有数据,开始提取物料
materials = []
footer_info = {} # 页脚信息
data_row = table_row + 1
while data_row < len(all_rows) and all_rows[data_row]:
# 检查是否是页脚信息(制单人、打印人)
if all_rows[data_row + 1][0] and "制单人" in str(
all_rows[data_row + 1][0]
):
# 解析页脚信息
self._parse_header_row(all_rows[data_row], footer_info)
# 检查下一行是否也是页脚信息
if (
data_row + 1 < len(all_rows)
and all_rows[data_row + 1]
):
self._parse_header_row(
all_rows[data_row + 1], footer_info
)
break
# 提取物料数据
material_row = all_rows[data_row]
material = {
"序号": material_row[0],
"材料编码": material_row[1],
"材料名称": material_row[2],
"规格": material_row[3],
"型号": material_row[4],
"图号": material_row[5],
"物料材质": material_row[6],
"计划数量": material_row[7],
"单位": material_row[8],
"需用日期": material_row[9],
"发料仓库": material_row[10],
"单位用量": material_row[11],
"累计出库数量": material_row[12],
}
materials.append(material)
data_row += 1
orders.append(
{
"order_info": {**order_info, **footer_info},
"materials": materials,
}
)
i += 1
return orders
def _parse_header_row(self, row: tuple, info: Dict):
"""
解析订单头信息的一行(字段名和值交错排列)
Args:
row: 行数据
info: 存储解析结果的字典
"""
i = 0
while i < len(row):
cell = row[i]
if cell and str(cell).strip() and "" in str(cell):
# 找到字段名
field_name = str(cell).replace("", "").strip()
# 应用字段名映射
if field_name in self.FIELD_NAME_MAPPING:
field_name = self.FIELD_NAME_MAPPING[field_name]
# 跳过空单元格,找到第一个非字段名的值
j = i + 1
while j < len(row) and (
not row[j] or not str(row[j]).strip() or "" in str(row[j])
):
j += 1
if j < len(row) and row[j] and not "" in str(row[j]):
info[field_name] = str(row[j]).strip()
# 跳过已处理的值,继续找下一个字段名
i = j + 1
else:
i += 1
def _convert_to_dataframe(self, orders: List[Dict]) -> pd.DataFrame:
"""
将订单数据转换为扁平化的 DataFrame
Args:
orders: 订单列表
Returns:
扁平化的 DataFrame
"""
all_records = []
for order in orders:
order_info = order["order_info"]
materials = order["materials"]
for material in materials:
record = {**order_info, **material}
all_records.append(record)
return pd.DataFrame(all_records)