Files
playwrite/utils/excel_converter.py

302 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
Excel 报表数据转换工具组件
将 Excel 报表数据转换为数据库记录形式
"""
import pandas as pd
import openpyxl
from typing import List, Dict, Optional
import os
class ExcelConverter:
"""Excel 报表数据转换器"""
# 字段名称映射(解决字段名冲突)
FIELD_NAME_MAPPING = {
'计划数量': '产品计划数量',
'单位': '产品单位'
}
def __init__(self, verbose: bool = True):
"""
初始化转换器
Args:
verbose: 是否打印详细日志
"""
self.verbose = verbose
def _print(self, *args, **kwargs):
"""打印日志(如果 verbose=True"""
if self.verbose:
print(*args, **kwargs)
def convert(self, input_file: str, output_file: str = None) -> pd.DataFrame:
"""
转换 Excel 文件
Args:
input_file: 输入文件路径
output_file: 输出文件路径(可选,不指定则不保存)
Returns:
转换后的 DataFrame
"""
# 处理输出文件名
if output_file:
output_file = self._handle_output_file(output_file)
self._print("=" * 80)
self._print("开始转换 Excel 数据")
self._print("=" * 80)
# 读取工作表
wb = openpyxl.load_workbook(input_file)
ws = wb.active
# 解析订单数据
orders = self._parse_sheet(ws)
self._print(f"\n\n共解析到 {len(orders)} 个订单")
# 打印每个订单的摘要
for i, order in enumerate(orders, 1):
order_info = order['order_info']
materials = order['materials']
self._print(f"\n订单 {i}:")
self._print(f" 备料计划单号: {order_info.get('备料计划单号', 'N/A')}")
self._print(f" 来源单号: {order_info.get('来源单号', 'N/A')}")
self._print(f" 产品编码: {order_info.get('产品编码', 'N/A')}")
self._print(f" 产品名称: {order_info.get('产品名称', 'N/A')}")
self._print(f" 产品计划数量: {order_info.get('产品计划数量', 'N/A')}")
self._print(f" 物料数量: {len(materials)}")
# 转换为 DataFrame
df = self._convert_to_dataframe(orders)
self._print(f"\n转换后的数据形状: {df.shape}")
if not df.empty:
self._print(f"列名: {list(df.columns)}")
# 保存文件
if output_file:
df.to_excel(output_file, index=False)
self._print(f"\n数据已保存到: {output_file}")
# 显示前几行数据
self._print("\n数据预览:")
pd.set_option('display.max_columns', None)
pd.set_option('display.width', 200)
pd.set_option('display.max_colwidth', 30)
self._print(df.head(20))
pd.reset_option('display.max_columns')
pd.reset_option('display.width')
pd.reset_option('display.max_colwidth')
else:
self._print("警告: 没有数据可保存")
return df
def _handle_output_file(self, output_file: str) -> str:
"""
处理输出文件,如果文件存在则尝试删除
Args:
output_file: 输出文件路径
Returns:
实际使用的输出文件路径
"""
if os.path.exists(output_file):
try:
os.remove(output_file)
except PermissionError:
self._print(f"警告: 无法删除 {output_file},可能文件被其他程序打开")
# 修改文件名
base, ext = os.path.splitext(output_file)
output_file = f"{base}_new{ext}"
return output_file
def _parse_sheet(self, ws) -> List[Dict]:
"""
解析一个工作表,返回所有订单的数据
每个订单包含:
- order_info: 订单头信息(包括页脚)
- materials: 物料数据列表
Args:
ws: openpyxl 工作表对象
Returns:
订单列表
"""
orders = []
all_rows = list(ws.iter_rows(values_only=True))
# 查找所有空行,用于分割订单
empty_rows = [i for i, row in enumerate(all_rows)
if all(cell is None or str(cell).strip() == "" for cell in row)]
self._print(f"检测到空行索引: {empty_rows}")
self._print(f"总行数: {len(all_rows)}")
# 逐行扫描,按订单结构解析
i = 0
while i < len(all_rows):
row = all_rows[i]
# 检查是否是订单标题行
if row and '离散备料计划' in str(row[0]):
self._print(f"\n在行 {i + 1} 发现订单标题")
# 解析订单头信息接下来的4行
order_info = {}
for j in range(1, 5):
if i + j < len(all_rows) and all_rows[i + j]:
self._parse_header_row(all_rows[i + j], order_info)
self._print(f"订单头信息: {order_info}")
# 跳过空行,找到表格标题行
table_row = i + 5
while table_row < len(all_rows) and (not all_rows[table_row] or not all_rows[table_row][0]):
table_row += 1
# 检查是否是表格标题行
if table_row < len(all_rows) and all_rows[table_row] and all_rows[table_row][0] == '序号':
self._print(f"在行 {table_row + 1} 发现表格标题")
# 检查表头下一行是否为空,判断是否存在数据
next_row = table_row + 1
is_empty_row = (next_row < len(all_rows) and
all_rows[next_row] and
all(cell is None or str(cell).strip() == "" for cell in all_rows[next_row]))
if is_empty_row:
self._print(f"表头下没有数据")
# 没有数据,查找页脚信息
materials = []
footer_info = {}
data_row = next_row + 1
while data_row < len(all_rows) and all_rows[data_row]:
if all_rows[data_row][0] and ('制单人' in str(all_rows[data_row][0]) or '打印人' in str(all_rows[data_row][0])):
self._print(f"在行 {data_row + 1} 发现页脚信息")
self._parse_header_row(all_rows[data_row], footer_info)
if data_row + 1 < len(all_rows) and all_rows[data_row + 1]:
self._parse_header_row(all_rows[data_row + 1], footer_info)
self._print(f"页脚信息: {footer_info}")
break
data_row += 1
orders.append({
'order_info': {**order_info, **footer_info},
'materials': materials
})
else:
# 有数据,开始提取物料
self._print(f"表头下有数据")
materials = []
footer_info = {} # 页脚信息
data_row = table_row + 1
while data_row < len(all_rows) and all_rows[data_row]:
# 检查是否是页脚信息(制单人、打印人)
if all_rows[data_row][0] and ('制单人' in str(all_rows[data_row][0]) or '打印人' in str(all_rows[data_row][0])):
self._print(f"在行 {data_row + 1} 发现页脚信息")
# 解析页脚信息
self._parse_header_row(all_rows[data_row], footer_info)
# 检查下一行是否也是页脚信息
if data_row + 1 < len(all_rows) and all_rows[data_row + 1]:
self._parse_header_row(all_rows[data_row + 1], footer_info)
self._print(f"页脚信息: {footer_info}")
break
# 提取物料数据
material_row = all_rows[data_row]
material = {
'序号': material_row[0],
'材料编码': material_row[1],
'材料名称': material_row[2],
'规格': material_row[3],
'型号': material_row[4],
'图号': material_row[5],
'物料材质': material_row[6],
'计划数量': material_row[7],
'单位': material_row[8],
'需用日期': material_row[9],
'发料仓库': material_row[10],
'单位用量': material_row[11],
'累计出库数量': material_row[12],
}
materials.append(material)
self._print(f" 添加物料: {material['材料编码']} - {material['材料名称']}")
data_row += 1
self._print(f"共解析到 {len(materials)} 条物料数据")
orders.append({
'order_info': {**order_info, **footer_info},
'materials': materials
})
i += 1
return orders
def _parse_header_row(self, row: tuple, info: Dict):
"""
解析订单头信息的一行(字段名和值交错排列)
Args:
row: 行数据
info: 存储解析结果的字典
"""
i = 0
while i < len(row):
cell = row[i]
if cell and str(cell).strip() and '' in str(cell):
# 找到字段名
field_name = str(cell).replace('', '').strip()
# 应用字段名映射
if field_name in self.FIELD_NAME_MAPPING:
field_name = self.FIELD_NAME_MAPPING[field_name]
# 跳过空单元格,找到第一个非字段名的值
j = i + 1
while j < len(row) and (not row[j] or not str(row[j]).strip() or '' in str(row[j])):
j += 1
if j < len(row) and row[j] and not '' in str(row[j]):
info[field_name] = str(row[j]).strip()
# 跳过已处理的值,继续找下一个字段名
i = j + 1
else:
i += 1
def _convert_to_dataframe(self, orders: List[Dict]) -> pd.DataFrame:
"""
将订单数据转换为扁平化的 DataFrame
Args:
orders: 订单列表
Returns:
扁平化的 DataFrame
"""
all_records = []
for order in orders:
order_info = order['order_info']
materials = order['materials']
for material in materials:
record = {
**order_info,
**material
}
all_records.append(record)
return pd.DataFrame(all_records)