""" Excel 报表数据转换工具组件 将 Excel 报表数据转换为数据库记录形式 """ import pandas as pd import openpyxl from typing import List, Dict, Optional import os class ExcelConverter: """Excel 报表数据转换器""" # 字段名称映射(解决字段名冲突) FIELD_NAME_MAPPING = {"计划数量": "产品计划数量", "单位": "产品单位"} def __init__(self, verbose: bool = True): """ 初始化转换器 Args: verbose: 是否打印详细日志 """ self.verbose = verbose def _print(self, *args, **kwargs): """打印日志(如果 verbose=True)""" if self.verbose: print(*args, **kwargs) def convert(self, input_file: str, output_file: str = None) -> pd.DataFrame: """ 转换 Excel 文件 Args: input_file: 输入文件路径 output_file: 输出文件路径(可选,不指定则不保存) Returns: 转换后的 DataFrame """ # 处理输出文件名 if output_file: output_file = self._handle_output_file(output_file) # 读取工作表 wb = openpyxl.load_workbook(input_file) ws = wb.active # 解析订单数据 orders = self._parse_sheet(ws) # 转换为 DataFrame df = self._convert_to_dataframe(orders) if not df.empty: # 保存文件 if output_file: df.to_excel(output_file, index=False) # 打印汇总报告 self._print("=" * 60) self._print("转换完成") self._print("=" * 60) self._print(f"订单数: {len(orders)}") self._print(f"数据行数: {len(df)}") self._print(f"输出文件: {output_file if output_file else 'N/A'}") self._print("=" * 60) return df def _handle_output_file(self, output_file: str) -> str: """ 处理输出文件,如果文件存在则尝试删除 Args: output_file: 输出文件路径 Returns: 实际使用的输出文件路径 """ if os.path.exists(output_file): try: os.remove(output_file) except PermissionError: self._print(f"警告: 无法删除 {output_file},可能文件被其他程序打开") # 修改文件名 base, ext = os.path.splitext(output_file) output_file = f"{base}_new{ext}" return output_file def _parse_sheet(self, ws) -> List[Dict]: """ 解析一个工作表,返回所有订单的数据 每个订单包含: - order_info: 订单头信息(包括页脚) - materials: 物料数据列表 Args: ws: openpyxl 工作表对象 Returns: 订单列表 """ orders = [] all_rows = list(ws.iter_rows(values_only=True)) # 逐行扫描,按订单结构解析 i = 0 while i < len(all_rows): row = all_rows[i] # 检查是否是订单标题行 if row and "离散备料计划" in str(row[0]): # 解析订单头信息(接下来的4行) order_info = {} for j in range(1, 5): if i + j < len(all_rows) and all_rows[i + j]: self._parse_header_row(all_rows[i + j], order_info) # 跳过空行,找到表格标题行 table_row = i + 5 while table_row < len(all_rows) and ( not all_rows[table_row] or not all_rows[table_row][0] ): table_row += 1 # 检查是否是表格标题行 if ( table_row < len(all_rows) and all_rows[table_row] and all_rows[table_row][0] == "序号" ): # 检查表头下一行是否为空,判断是否存在数据 next_row = table_row + 1 is_empty_row = ( next_row < len(all_rows) and all_rows[next_row] and all( cell is None or str(cell).strip() == "" for cell in all_rows[next_row] ) ) if is_empty_row: # 没有数据,查找页脚信息 materials = [] footer_info = {} data_row = next_row + 1 while data_row < len(all_rows) and all_rows[data_row]: if all_rows[data_row][0] and ( "制单人" in str(all_rows[data_row][0]) or "打印人" in str(all_rows[data_row][0]) ): self._parse_header_row(all_rows[data_row], footer_info) if ( data_row + 1 < len(all_rows) and all_rows[data_row + 1] ): self._parse_header_row( all_rows[data_row + 1], footer_info ) break data_row += 1 orders.append( { "order_info": {**order_info, **footer_info}, "materials": materials, } ) else: # 有数据,开始提取物料 materials = [] footer_info = {} # 页脚信息 data_row = table_row + 1 while data_row < len(all_rows) and all_rows[data_row]: # 检查是否是页脚信息(制单人、打印人) if all_rows[data_row + 1][0] and "制单人" in str( all_rows[data_row + 1][0] ): # 解析页脚信息 self._parse_header_row(all_rows[data_row], footer_info) # 检查下一行是否也是页脚信息 if ( data_row + 1 < len(all_rows) and all_rows[data_row + 1] ): self._parse_header_row( all_rows[data_row + 1], footer_info ) break # 提取物料数据 material_row = all_rows[data_row] material = { "序号": material_row[0], "材料编码": material_row[1], "材料名称": material_row[2], "规格": material_row[3], "型号": material_row[4], "图号": material_row[5], "物料材质": material_row[6], "计划数量": material_row[7], "单位": material_row[8], "需用日期": material_row[9], "发料仓库": material_row[10], "单位用量": material_row[11], "累计出库数量": material_row[12], } materials.append(material) data_row += 1 orders.append( { "order_info": {**order_info, **footer_info}, "materials": materials, } ) i += 1 return orders def _parse_header_row(self, row: tuple, info: Dict): """ 解析订单头信息的一行(字段名和值交错排列) Args: row: 行数据 info: 存储解析结果的字典 """ i = 0 while i < len(row): cell = row[i] if cell and str(cell).strip() and ":" in str(cell): # 找到字段名 field_name = str(cell).replace(":", "").strip() # 应用字段名映射 if field_name in self.FIELD_NAME_MAPPING: field_name = self.FIELD_NAME_MAPPING[field_name] # 跳过空单元格,找到第一个非字段名的值 j = i + 1 while j < len(row) and ( not row[j] or not str(row[j]).strip() or ":" in str(row[j]) ): j += 1 if j < len(row) and row[j] and not ":" in str(row[j]): info[field_name] = str(row[j]).strip() # 跳过已处理的值,继续找下一个字段名 i = j + 1 else: i += 1 def _convert_to_dataframe(self, orders: List[Dict]) -> pd.DataFrame: """ 将订单数据转换为扁平化的 DataFrame Args: orders: 订单列表 Returns: 扁平化的 DataFrame """ all_records = [] for order in orders: order_info = order["order_info"] materials = order["materials"] for material in materials: record = {**order_info, **material} all_records.append(record) return pd.DataFrame(all_records)