302 lines
12 KiB
Python
302 lines
12 KiB
Python
"""
|
||
Excel 报表数据转换工具组件
|
||
将 Excel 报表数据转换为数据库记录形式
|
||
"""
|
||
import pandas as pd
|
||
import openpyxl
|
||
from typing import List, Dict, Optional
|
||
import os
|
||
|
||
|
||
class ExcelConverter:
|
||
"""Excel 报表数据转换器"""
|
||
|
||
# 字段名称映射(解决字段名冲突)
|
||
FIELD_NAME_MAPPING = {
|
||
'计划数量': '产品计划数量',
|
||
'单位': '产品单位'
|
||
}
|
||
|
||
def __init__(self, verbose: bool = True):
|
||
"""
|
||
初始化转换器
|
||
|
||
Args:
|
||
verbose: 是否打印详细日志
|
||
"""
|
||
self.verbose = verbose
|
||
|
||
def _print(self, *args, **kwargs):
|
||
"""打印日志(如果 verbose=True)"""
|
||
if self.verbose:
|
||
print(*args, **kwargs)
|
||
|
||
def convert(self, input_file: str, output_file: str = None) -> pd.DataFrame:
|
||
"""
|
||
转换 Excel 文件
|
||
|
||
Args:
|
||
input_file: 输入文件路径
|
||
output_file: 输出文件路径(可选,不指定则不保存)
|
||
|
||
Returns:
|
||
转换后的 DataFrame
|
||
"""
|
||
# 处理输出文件名
|
||
if output_file:
|
||
output_file = self._handle_output_file(output_file)
|
||
|
||
self._print("=" * 80)
|
||
self._print("开始转换 Excel 数据")
|
||
self._print("=" * 80)
|
||
|
||
# 读取工作表
|
||
wb = openpyxl.load_workbook(input_file)
|
||
ws = wb.active
|
||
|
||
# 解析订单数据
|
||
orders = self._parse_sheet(ws)
|
||
|
||
self._print(f"\n\n共解析到 {len(orders)} 个订单")
|
||
|
||
# 打印每个订单的摘要
|
||
for i, order in enumerate(orders, 1):
|
||
order_info = order['order_info']
|
||
materials = order['materials']
|
||
self._print(f"\n订单 {i}:")
|
||
self._print(f" 备料计划单号: {order_info.get('备料计划单号', 'N/A')}")
|
||
self._print(f" 来源单号: {order_info.get('来源单号', 'N/A')}")
|
||
self._print(f" 产品编码: {order_info.get('产品编码', 'N/A')}")
|
||
self._print(f" 产品名称: {order_info.get('产品名称', 'N/A')}")
|
||
self._print(f" 产品计划数量: {order_info.get('产品计划数量', 'N/A')}")
|
||
self._print(f" 物料数量: {len(materials)}")
|
||
|
||
# 转换为 DataFrame
|
||
df = self._convert_to_dataframe(orders)
|
||
|
||
self._print(f"\n转换后的数据形状: {df.shape}")
|
||
|
||
if not df.empty:
|
||
self._print(f"列名: {list(df.columns)}")
|
||
|
||
# 保存文件
|
||
if output_file:
|
||
df.to_excel(output_file, index=False)
|
||
self._print(f"\n数据已保存到: {output_file}")
|
||
|
||
# 显示前几行数据
|
||
self._print("\n数据预览:")
|
||
pd.set_option('display.max_columns', None)
|
||
pd.set_option('display.width', 200)
|
||
pd.set_option('display.max_colwidth', 30)
|
||
self._print(df.head(20))
|
||
pd.reset_option('display.max_columns')
|
||
pd.reset_option('display.width')
|
||
pd.reset_option('display.max_colwidth')
|
||
else:
|
||
self._print("警告: 没有数据可保存")
|
||
|
||
return df
|
||
|
||
def _handle_output_file(self, output_file: str) -> str:
|
||
"""
|
||
处理输出文件,如果文件存在则尝试删除
|
||
|
||
Args:
|
||
output_file: 输出文件路径
|
||
|
||
Returns:
|
||
实际使用的输出文件路径
|
||
"""
|
||
if os.path.exists(output_file):
|
||
try:
|
||
os.remove(output_file)
|
||
except PermissionError:
|
||
self._print(f"警告: 无法删除 {output_file},可能文件被其他程序打开")
|
||
# 修改文件名
|
||
base, ext = os.path.splitext(output_file)
|
||
output_file = f"{base}_new{ext}"
|
||
return output_file
|
||
|
||
def _parse_sheet(self, ws) -> List[Dict]:
|
||
"""
|
||
解析一个工作表,返回所有订单的数据
|
||
|
||
每个订单包含:
|
||
- order_info: 订单头信息(包括页脚)
|
||
- materials: 物料数据列表
|
||
|
||
Args:
|
||
ws: openpyxl 工作表对象
|
||
|
||
Returns:
|
||
订单列表
|
||
"""
|
||
orders = []
|
||
all_rows = list(ws.iter_rows(values_only=True))
|
||
|
||
# 查找所有空行,用于分割订单
|
||
empty_rows = [i for i, row in enumerate(all_rows)
|
||
if all(cell is None or str(cell).strip() == "" for cell in row)]
|
||
|
||
self._print(f"检测到空行索引: {empty_rows}")
|
||
self._print(f"总行数: {len(all_rows)}")
|
||
|
||
# 逐行扫描,按订单结构解析
|
||
i = 0
|
||
while i < len(all_rows):
|
||
row = all_rows[i]
|
||
|
||
# 检查是否是订单标题行
|
||
if row and '离散备料计划' in str(row[0]):
|
||
self._print(f"\n在行 {i + 1} 发现订单标题")
|
||
|
||
# 解析订单头信息(接下来的4行)
|
||
order_info = {}
|
||
for j in range(1, 5):
|
||
if i + j < len(all_rows) and all_rows[i + j]:
|
||
self._parse_header_row(all_rows[i + j], order_info)
|
||
|
||
self._print(f"订单头信息: {order_info}")
|
||
|
||
# 跳过空行,找到表格标题行
|
||
table_row = i + 5
|
||
while table_row < len(all_rows) and (not all_rows[table_row] or not all_rows[table_row][0]):
|
||
table_row += 1
|
||
|
||
# 检查是否是表格标题行
|
||
if table_row < len(all_rows) and all_rows[table_row] and all_rows[table_row][0] == '序号':
|
||
self._print(f"在行 {table_row + 1} 发现表格标题")
|
||
|
||
# 检查表头下一行是否为空,判断是否存在数据
|
||
next_row = table_row + 1
|
||
is_empty_row = (next_row < len(all_rows) and
|
||
all_rows[next_row] and
|
||
all(cell is None or str(cell).strip() == "" for cell in all_rows[next_row]))
|
||
|
||
if is_empty_row:
|
||
self._print(f"表头下没有数据")
|
||
# 没有数据,查找页脚信息
|
||
materials = []
|
||
footer_info = {}
|
||
data_row = next_row + 1
|
||
while data_row < len(all_rows) and all_rows[data_row]:
|
||
if all_rows[data_row][0] and ('制单人' in str(all_rows[data_row][0]) or '打印人' in str(all_rows[data_row][0])):
|
||
self._print(f"在行 {data_row + 1} 发现页脚信息")
|
||
self._parse_header_row(all_rows[data_row], footer_info)
|
||
if data_row + 1 < len(all_rows) and all_rows[data_row + 1]:
|
||
self._parse_header_row(all_rows[data_row + 1], footer_info)
|
||
self._print(f"页脚信息: {footer_info}")
|
||
break
|
||
data_row += 1
|
||
|
||
orders.append({
|
||
'order_info': {**order_info, **footer_info},
|
||
'materials': materials
|
||
})
|
||
else:
|
||
# 有数据,开始提取物料
|
||
self._print(f"表头下有数据")
|
||
materials = []
|
||
footer_info = {} # 页脚信息
|
||
data_row = table_row + 1
|
||
while data_row < len(all_rows) and all_rows[data_row]:
|
||
# 检查是否是页脚信息(制单人、打印人)
|
||
if all_rows[data_row][0] and ('制单人' in str(all_rows[data_row][0]) or '打印人' in str(all_rows[data_row][0])):
|
||
self._print(f"在行 {data_row + 1} 发现页脚信息")
|
||
# 解析页脚信息
|
||
self._parse_header_row(all_rows[data_row], footer_info)
|
||
# 检查下一行是否也是页脚信息
|
||
if data_row + 1 < len(all_rows) and all_rows[data_row + 1]:
|
||
self._parse_header_row(all_rows[data_row + 1], footer_info)
|
||
self._print(f"页脚信息: {footer_info}")
|
||
break
|
||
|
||
# 提取物料数据
|
||
material_row = all_rows[data_row]
|
||
material = {
|
||
'序号': material_row[0],
|
||
'材料编码': material_row[1],
|
||
'材料名称': material_row[2],
|
||
'规格': material_row[3],
|
||
'型号': material_row[4],
|
||
'图号': material_row[5],
|
||
'物料材质': material_row[6],
|
||
'计划数量': material_row[7],
|
||
'单位': material_row[8],
|
||
'需用日期': material_row[9],
|
||
'发料仓库': material_row[10],
|
||
'单位用量': material_row[11],
|
||
'累计出库数量': material_row[12],
|
||
}
|
||
materials.append(material)
|
||
self._print(f" 添加物料: {material['材料编码']} - {material['材料名称']}")
|
||
|
||
data_row += 1
|
||
|
||
self._print(f"共解析到 {len(materials)} 条物料数据")
|
||
|
||
orders.append({
|
||
'order_info': {**order_info, **footer_info},
|
||
'materials': materials
|
||
})
|
||
|
||
i += 1
|
||
|
||
return orders
|
||
|
||
def _parse_header_row(self, row: tuple, info: Dict):
|
||
"""
|
||
解析订单头信息的一行(字段名和值交错排列)
|
||
|
||
Args:
|
||
row: 行数据
|
||
info: 存储解析结果的字典
|
||
"""
|
||
i = 0
|
||
while i < len(row):
|
||
cell = row[i]
|
||
if cell and str(cell).strip() and ':' in str(cell):
|
||
# 找到字段名
|
||
field_name = str(cell).replace(':', '').strip()
|
||
|
||
# 应用字段名映射
|
||
if field_name in self.FIELD_NAME_MAPPING:
|
||
field_name = self.FIELD_NAME_MAPPING[field_name]
|
||
|
||
# 跳过空单元格,找到第一个非字段名的值
|
||
j = i + 1
|
||
while j < len(row) and (not row[j] or not str(row[j]).strip() or ':' in str(row[j])):
|
||
j += 1
|
||
if j < len(row) and row[j] and not ':' in str(row[j]):
|
||
info[field_name] = str(row[j]).strip()
|
||
# 跳过已处理的值,继续找下一个字段名
|
||
i = j + 1
|
||
else:
|
||
i += 1
|
||
|
||
def _convert_to_dataframe(self, orders: List[Dict]) -> pd.DataFrame:
|
||
"""
|
||
将订单数据转换为扁平化的 DataFrame
|
||
|
||
Args:
|
||
orders: 订单列表
|
||
|
||
Returns:
|
||
扁平化的 DataFrame
|
||
"""
|
||
all_records = []
|
||
|
||
for order in orders:
|
||
order_info = order['order_info']
|
||
materials = order['materials']
|
||
|
||
for material in materials:
|
||
record = {
|
||
**order_info,
|
||
**material
|
||
}
|
||
all_records.append(record)
|
||
|
||
return pd.DataFrame(all_records)
|