- Standardize quote style (single to double quotes) - Improve code formatting consistency - Apply formatting to utilities, GUI components, and tools - Update imports and docstrings for consistency Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
439 lines
15 KiB
Python
439 lines
15 KiB
Python
"""
|
||
离散备料计划维护数据提取工具
|
||
负责登录、批量下载、转换数据
|
||
"""
|
||
|
||
import os
|
||
import pandas as pd
|
||
from playwright.sync_api import sync_playwright
|
||
from utils.excel_converter import ExcelConverter
|
||
from utils.auth import login, logout
|
||
from db.production_order_query import (
|
||
read_production_ids,
|
||
query_production_order_numbers,
|
||
)
|
||
from typing import Callable, Optional
|
||
|
||
|
||
class DiscreteMaterialPlanExtractor:
|
||
"""离散备料计划维护数据提取器"""
|
||
|
||
def __init__(
|
||
self, username, password, headless=False, verbose=True, batch_size=100
|
||
):
|
||
"""
|
||
初始化提取器
|
||
|
||
Args:
|
||
username: 登录用户名
|
||
password: 登录密码
|
||
headless: 是否无头模式运行
|
||
verbose: 是否打印详细日志
|
||
batch_size: 批次大小
|
||
"""
|
||
self.username = username
|
||
self.password = password
|
||
self.headless = headless
|
||
self.verbose = verbose
|
||
self.batch_size = batch_size
|
||
self.progress_callback = None
|
||
self.converter = ExcelConverter(verbose=verbose)
|
||
|
||
def _print(self, *args, **kwargs):
|
||
"""打印日志(如果 verbose=True)"""
|
||
if self.verbose:
|
||
print(*args, **kwargs)
|
||
|
||
def _report_progress(
|
||
self, stage: str, current: int, total: int, message: str, **detail
|
||
):
|
||
"""
|
||
报告进度
|
||
|
||
Args:
|
||
stage: 阶段标识
|
||
current: 当前进度值
|
||
total: 总量
|
||
message: 显示消息
|
||
**detail: 额外详细信息
|
||
"""
|
||
if self.progress_callback:
|
||
try:
|
||
from gui.progress import ProgressInfo
|
||
|
||
progress_info = ProgressInfo(
|
||
stage=stage,
|
||
current=current,
|
||
total=total,
|
||
message=message,
|
||
detail=detail,
|
||
)
|
||
self.progress_callback(progress_info)
|
||
except Exception:
|
||
# 如果进度回调失败,忽略错误,不影响主流程
|
||
pass
|
||
|
||
def get_production_order_numbers(self, production_id_file, report_progress=False):
|
||
"""
|
||
读取总排号文件并查询数据库获取生产订单号
|
||
|
||
Args:
|
||
production_id_file: ProductionID.txt 文件路径
|
||
report_progress: 是否报告进度
|
||
|
||
Returns:
|
||
生产订单号列表
|
||
"""
|
||
# 读取总排号
|
||
production_ids = read_production_ids(production_id_file)
|
||
self._print(f"从文件读取到 {len(production_ids)} 个总排号")
|
||
|
||
# 查询数据库获取生产订单号
|
||
order_ids = query_production_order_numbers(production_ids)
|
||
self._print(f"查询到 {len(order_ids)} 个生产订单号")
|
||
|
||
if report_progress:
|
||
self._report_progress(
|
||
"query",
|
||
1,
|
||
1,
|
||
f"查询到 {len(order_ids)} 个生产订单号",
|
||
count=len(order_ids),
|
||
)
|
||
|
||
return order_ids
|
||
|
||
def group_order_ids(self, order_ids, group_size=100):
|
||
"""将订单号分组"""
|
||
for i in range(0, len(order_ids), group_size):
|
||
yield order_ids[i : i + group_size]
|
||
|
||
def download_batch(
|
||
self,
|
||
inner_frame,
|
||
order_ids,
|
||
batch_index,
|
||
total_batches,
|
||
page1,
|
||
debug_mode=False,
|
||
debug_batch=None,
|
||
):
|
||
"""下载一批订单号的数据"""
|
||
from playwright.sync_api import TimeoutError
|
||
import re
|
||
import time
|
||
|
||
# 清空文本框
|
||
textbox = inner_frame.get_by_role("textbox", name="来源生产订单号")
|
||
textbox.fill("")
|
||
|
||
# 填充订单号
|
||
textbox.fill(",".join(order_ids))
|
||
|
||
# 点击查询
|
||
inner_frame.locator(".search-component-searchBtn").click()
|
||
self._print(f"第 {batch_index + 1} 批查询完成,等待加载结果...")
|
||
|
||
# 等待加载完成
|
||
loading_locator = inner_frame.locator("div").filter(has_text="加载中").nth(1)
|
||
try:
|
||
loading_locator.wait_for(state="visible", timeout=3000)
|
||
loading_locator.wait_for(state="hidden", timeout=0) # 无限等待,直到消失
|
||
except TimeoutError:
|
||
# 加载很快完成,或者没有出现加载提示
|
||
pass
|
||
self._print(f"第 {batch_index + 1} 批加载完成,开始选择数据...")
|
||
|
||
# 调试模式:只在指定批次暂停
|
||
if debug_mode and (debug_batch is None or batch_index == debug_batch):
|
||
self._print(f"=== 调试暂停:第 {batch_index + 1} 批 ===")
|
||
page1.pause()
|
||
|
||
# 选择所有数据
|
||
inner_frame.get_by_role("row", name="序号").get_by_label("").click()
|
||
|
||
# 点击输出
|
||
inner_frame.get_by_role("button", name="更多").hover()
|
||
inner_frame.get_by_text("输出", exact=True).click()
|
||
|
||
# 设置行数阈值
|
||
input_box = (
|
||
inner_frame.locator("div")
|
||
.filter(has_text=re.compile(r"^行数阈值$"))
|
||
.locator("input[type='text']")
|
||
)
|
||
input_box.fill("300000")
|
||
|
||
# 下载文件
|
||
download_path = f"D:/python/playwrite/data/temp_batch_{batch_index + 1}.xlsx"
|
||
with page1.expect_download() as download_info:
|
||
inner_frame.get_by_role("button", name="确定(Y)").click()
|
||
|
||
download = download_info.value
|
||
download.save_as(download_path)
|
||
self._print(f"第 {batch_index + 1} 批下载完成: {download_path}")
|
||
|
||
# 报告进度
|
||
self._report_progress(
|
||
"download",
|
||
batch_index + 1,
|
||
total_batches,
|
||
f"第 {batch_index + 1}/{total_batches} 批下载完成",
|
||
batch_index=batch_index + 1,
|
||
)
|
||
|
||
# 关闭输出对话框(如果有的话)
|
||
# try:
|
||
# inner_frame.get_by_role("button", name="取消").click()
|
||
# except:
|
||
# pass
|
||
|
||
# 等待页面恢复,准备下一次查询
|
||
time.sleep(1)
|
||
|
||
return download_path
|
||
|
||
def convert_and_merge_files(self, file_paths, output_path):
|
||
"""使用 ExcelConverter 转换并合并所有文件"""
|
||
# 确保输出文件路径是正确的格式
|
||
output_path = os.path.normpath(output_path)
|
||
output_dir = os.path.dirname(output_path)
|
||
output_filename = os.path.basename(output_path)
|
||
|
||
self._print(f"输出文件路径: {output_path}")
|
||
self._print(f"输出目录: {output_dir}")
|
||
self._print(f"输出文件名: {output_filename}")
|
||
|
||
# 确保输出文件的父目录存在
|
||
if output_dir and not os.path.exists(output_dir):
|
||
self._print(f"创建输出目录: {output_dir}")
|
||
os.makedirs(output_dir)
|
||
|
||
all_dataframes = []
|
||
|
||
for i, file_path in enumerate(file_paths, 1):
|
||
self._print(f"转换第 {i} 个文件: {file_path}")
|
||
|
||
# 报告进度
|
||
self._report_progress(
|
||
"convert",
|
||
i,
|
||
len(file_paths),
|
||
f"转换第 {i}/{len(file_paths)} 个文件",
|
||
file_index=i,
|
||
file_path=file_path,
|
||
)
|
||
|
||
df = self.converter.convert(file_path, output_file=None) # 只转换,不保存
|
||
all_dataframes.append(df)
|
||
self._print(f" 提取到 {len(df)} 条记录")
|
||
|
||
if all_dataframes:
|
||
self._print(f"\n合并 {len(all_dataframes)} 个文件的数据...")
|
||
merged_df = pd.concat(all_dataframes, ignore_index=True)
|
||
merged_df.to_excel(output_path, index=False)
|
||
self._print(f"合并完成: {output_path}, 总共 {len(merged_df)} 条记录")
|
||
|
||
# 删除临时文件
|
||
for file_path in file_paths:
|
||
os.remove(file_path)
|
||
self._print(f"已删除临时文件: {file_path}")
|
||
|
||
return output_path
|
||
return None
|
||
|
||
def setup_query_interface(self, inner_frame):
|
||
"""设置查询界面"""
|
||
import re
|
||
|
||
# 点击图标按钮打开查询界面
|
||
inner_frame.locator(".search-name-wrapper > .iconfont").click()
|
||
inner_frame.get_by_text("订单号查询").click()
|
||
inner_frame.get_by_role("tab", name="全部").click()
|
||
|
||
# 填充并验证,如果失败则重试
|
||
max_retries = 3
|
||
expected_value = "5000"
|
||
for attempt in range(max_retries):
|
||
inner_frame.locator("#rc_select_0").fill(expected_value)
|
||
inner_frame.locator("#rc_select_0").press("Enter")
|
||
# 检查填充是否成功
|
||
actual_value = inner_frame.locator("#rc_select_0").input_value()
|
||
if actual_value == expected_value:
|
||
self._print(f"文本框填充成功: {expected_value}")
|
||
break
|
||
else:
|
||
self._print(
|
||
f"第 {attempt + 1} 次填充失败,实际值: {actual_value},重试..."
|
||
)
|
||
if attempt == max_retries - 1:
|
||
self._print(
|
||
f"警告: {max_retries} 次尝试后仍未成功填充,继续执行..."
|
||
)
|
||
|
||
def extract(
|
||
self,
|
||
production_id_file,
|
||
data_dir="D:/python/playwrite/data",
|
||
output_file="D:/python/playwrite/data/离散备料计划维护_合并.xlsx",
|
||
debug_mode=False,
|
||
debug_batch=None,
|
||
progress_callback=None,
|
||
):
|
||
"""
|
||
执行完整的数据提取流程
|
||
|
||
Args:
|
||
production_id_file: ProductionID.txt 文件路径
|
||
data_dir: 数据保存目录
|
||
output_file: 最终输出文件路径
|
||
debug_mode: 是否启用调试模式
|
||
debug_batch: 调试批次号
|
||
progress_callback: 进度回调函数(覆盖初始化时的回调)
|
||
|
||
Returns:
|
||
输出文件路径
|
||
"""
|
||
# 保存原始回调
|
||
original_callback = self.progress_callback
|
||
# 使用传入的回调或初始化时的回调
|
||
self.progress_callback = progress_callback or self.progress_callback
|
||
|
||
try:
|
||
with sync_playwright() as playwright:
|
||
# 调用登录模块
|
||
browser, context, page, main_frame = login(
|
||
playwright=playwright,
|
||
username=self.username,
|
||
password=self.password,
|
||
headless=self.headless,
|
||
ignore_https_errors=True,
|
||
)
|
||
|
||
# 登录完成
|
||
self._report_progress("login", 1, 1, "登录成功")
|
||
|
||
self._print("=" * 80)
|
||
self._print("开始执行离散备料计划维护数据提取")
|
||
self._print("=" * 80)
|
||
|
||
# 登录成功后可以进行后续操作
|
||
# 点击打开"功能菜单"
|
||
main_frame.locator("i").first.click()
|
||
|
||
# 点击打开"离散备料计划维护"
|
||
with page.expect_popup() as page1_info:
|
||
main_frame.get_by_title(
|
||
"离散备料计划维护", exact=True
|
||
).first.click()
|
||
page1 = page1_info.value
|
||
|
||
# 获取 nested iframe
|
||
main_frame = page1.locator("#forwardFrame").content_frame
|
||
inner_frame_locator = main_frame.locator("#mainiframe")
|
||
inner_frame_locator.wait_for(state="visible", timeout=15000)
|
||
inner_frame = inner_frame_locator.content_frame
|
||
|
||
# 设置查询界面
|
||
self.setup_query_interface(inner_frame)
|
||
|
||
# 读取总排号并查询生产订单号
|
||
order_ids = self.get_production_order_numbers(
|
||
production_id_file, report_progress=True
|
||
)
|
||
|
||
# 按批次下载
|
||
downloaded_files = []
|
||
# 计算总批次数
|
||
total_batches = sum(
|
||
1 for _ in self.group_order_ids(order_ids, self.batch_size)
|
||
)
|
||
|
||
for batch_index, order_ids_batch in enumerate(
|
||
self.group_order_ids(order_ids, self.batch_size)
|
||
):
|
||
self._print(
|
||
f"\n=== 开始处理第 {batch_index + 1} 批,共 {len(order_ids_batch)} 个订单号 ==="
|
||
)
|
||
|
||
# 报告开始下载批次
|
||
self._report_progress(
|
||
"download",
|
||
batch_index,
|
||
total_batches,
|
||
f"正在下载第 {batch_index + 1}/{total_batches} 批...",
|
||
batch_index=batch_index + 1,
|
||
batch_size=len(order_ids_batch),
|
||
)
|
||
|
||
downloaded_file = self.download_batch(
|
||
inner_frame,
|
||
order_ids_batch,
|
||
batch_index,
|
||
total_batches,
|
||
page1,
|
||
debug_mode=debug_mode,
|
||
debug_batch=debug_batch,
|
||
)
|
||
downloaded_files.append(downloaded_file)
|
||
|
||
# 执行账号注销
|
||
self._print("\n开始执行账号注销...")
|
||
self._report_progress("logout", 1, 1, "正在注销账号...")
|
||
logout(main_frame, verbose=self.verbose)
|
||
|
||
# 转换并合并文件
|
||
if downloaded_files:
|
||
self._report_progress(
|
||
"convert",
|
||
0,
|
||
len(downloaded_files),
|
||
f"开始转换并合并 {len(downloaded_files)} 个文件",
|
||
file_count=len(downloaded_files),
|
||
)
|
||
self._print(
|
||
f"\n=== 开始转换并合并 {len(downloaded_files)} 个文件 ==="
|
||
)
|
||
self.convert_and_merge_files(downloaded_files, output_file)
|
||
else:
|
||
self._print("\n没有下载到任何文件")
|
||
|
||
self._print(f"\n=== 全部完成 ===")
|
||
self._print(f"最终文件: {output_file}")
|
||
|
||
# 报告完成
|
||
self._report_progress(
|
||
"complete", 1, 1, "提取完成", output_file=output_file
|
||
)
|
||
|
||
# 关闭浏览器
|
||
context.close()
|
||
browser.close()
|
||
|
||
return output_file
|
||
|
||
finally:
|
||
# 恢复原始回调
|
||
self.progress_callback = original_callback
|
||
|
||
|
||
def main():
|
||
"""测试函数"""
|
||
extractor = DiscreteMaterialPlanExtractor(
|
||
username="BLDpengqiangqiang",
|
||
password="Cqbld123456.",
|
||
headless=False,
|
||
verbose=True,
|
||
)
|
||
|
||
production_id_file = os.path.join(os.path.dirname(__file__), "productionID.txt")
|
||
output_file = "D:/python/playwrite/data/离散备料计划维护_合并.xlsx"
|
||
|
||
extractor.extract(production_id_file, output_file)
|
||
|
||
input("按回车退出...")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|