Add Excel analysis and conversion scripts for order data processing

This commit is contained in:
Misaka_Company
2026-01-16 11:15:19 +08:00
parent 537cb58f2e
commit f9651ccf07
2 changed files with 276 additions and 0 deletions

63
analyze_excel.py Normal file
View File

@@ -0,0 +1,63 @@
"""
分析 Excel 文件的数据结构
"""
import pandas as pd
import openpyxl
import sys
# 设置输出编码
if sys.platform == 'win32':
import io
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8')
# 读取 Excel 文件
file_path = "data/导出文件.xlsx"
print("=" * 80)
print("1. 读取所有工作表名称")
print("=" * 80)
wb = openpyxl.load_workbook(file_path)
sheet_names = wb.sheetnames
print(f"工作表数量: {len(sheet_names)}")
print(f"工作表名称: {sheet_names}")
print("\n" + "=" * 80)
print("2. 逐行读取原始数据(使用 openpyxl")
print("=" * 80)
for sheet_name in sheet_names:
print(f"\n--- 工作表: {sheet_name} ---")
ws = wb[sheet_name]
for idx, row in enumerate(ws.iter_rows(values_only=True), 1):
# 检查是否为空行
is_empty = all(cell is None or str(cell).strip() == "" for cell in row)
if is_empty:
print(f"{idx}: [空行]")
else:
# 过滤掉 None 和空字符串,只显示有效数据
valid_data = [str(cell) if cell is not None else "" for cell in row]
print(f"{idx}: {valid_data}")
print("\n" + "=" * 80)
print("3. 使用 pandas 读取数据(观察列结构)")
print("=" * 80)
for sheet_name in sheet_names:
print(f"\n--- 工作表: {sheet_name} ---")
df = pd.read_excel(file_path, sheet_name=sheet_name)
print(f"DataFrame 形状: {df.shape}")
print(f"\n列名:")
for i, col in enumerate(df.columns):
print(f"{i}: {col}")
print(f"\n数据预览:")
pd.set_option('display.max_rows', 25)
pd.set_option('display.max_columns', 20)
pd.set_option('display.width', 200)
pd.set_option('display.max_colwidth', 30)
print(df)
pd.reset_option('display.max_rows')
pd.reset_option('display.max_columns')
pd.reset_option('display.width')
pd.reset_option('display.max_colwidth')