Files
playwrite/utils/离散备料计划维护数据提取.py
Misaka_Company 2ca53f6a55 style: normalize code formatting across the codebase
- Standardize quote style (single to double quotes)
- Improve code formatting consistency
- Apply formatting to utilities, GUI components, and tools
- Update imports and docstrings for consistency

Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
2026-02-05 14:56:35 +08:00

439 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
离散备料计划维护数据提取工具
负责登录、批量下载、转换数据
"""
import os
import pandas as pd
from playwright.sync_api import sync_playwright
from utils.excel_converter import ExcelConverter
from utils.auth import login, logout
from db.production_order_query import (
read_production_ids,
query_production_order_numbers,
)
from typing import Callable, Optional
class DiscreteMaterialPlanExtractor:
"""离散备料计划维护数据提取器"""
def __init__(
self, username, password, headless=False, verbose=True, batch_size=100
):
"""
初始化提取器
Args:
username: 登录用户名
password: 登录密码
headless: 是否无头模式运行
verbose: 是否打印详细日志
batch_size: 批次大小
"""
self.username = username
self.password = password
self.headless = headless
self.verbose = verbose
self.batch_size = batch_size
self.progress_callback = None
self.converter = ExcelConverter(verbose=verbose)
def _print(self, *args, **kwargs):
"""打印日志(如果 verbose=True"""
if self.verbose:
print(*args, **kwargs)
def _report_progress(
self, stage: str, current: int, total: int, message: str, **detail
):
"""
报告进度
Args:
stage: 阶段标识
current: 当前进度值
total: 总量
message: 显示消息
**detail: 额外详细信息
"""
if self.progress_callback:
try:
from gui.progress import ProgressInfo
progress_info = ProgressInfo(
stage=stage,
current=current,
total=total,
message=message,
detail=detail,
)
self.progress_callback(progress_info)
except Exception:
# 如果进度回调失败,忽略错误,不影响主流程
pass
def get_production_order_numbers(self, production_id_file, report_progress=False):
"""
读取总排号文件并查询数据库获取生产订单号
Args:
production_id_file: ProductionID.txt 文件路径
report_progress: 是否报告进度
Returns:
生产订单号列表
"""
# 读取总排号
production_ids = read_production_ids(production_id_file)
self._print(f"从文件读取到 {len(production_ids)} 个总排号")
# 查询数据库获取生产订单号
order_ids = query_production_order_numbers(production_ids)
self._print(f"查询到 {len(order_ids)} 个生产订单号")
if report_progress:
self._report_progress(
"query",
1,
1,
f"查询到 {len(order_ids)} 个生产订单号",
count=len(order_ids),
)
return order_ids
def group_order_ids(self, order_ids, group_size=100):
"""将订单号分组"""
for i in range(0, len(order_ids), group_size):
yield order_ids[i : i + group_size]
def download_batch(
self,
inner_frame,
order_ids,
batch_index,
total_batches,
page1,
debug_mode=False,
debug_batch=None,
):
"""下载一批订单号的数据"""
from playwright.sync_api import TimeoutError
import re
import time
# 清空文本框
textbox = inner_frame.get_by_role("textbox", name="来源生产订单号")
textbox.fill("")
# 填充订单号
textbox.fill(",".join(order_ids))
# 点击查询
inner_frame.locator(".search-component-searchBtn").click()
self._print(f"{batch_index + 1} 批查询完成,等待加载结果...")
# 等待加载完成
loading_locator = inner_frame.locator("div").filter(has_text="加载中").nth(1)
try:
loading_locator.wait_for(state="visible", timeout=3000)
loading_locator.wait_for(state="hidden", timeout=0) # 无限等待,直到消失
except TimeoutError:
# 加载很快完成,或者没有出现加载提示
pass
self._print(f"{batch_index + 1} 批加载完成,开始选择数据...")
# 调试模式:只在指定批次暂停
if debug_mode and (debug_batch is None or batch_index == debug_batch):
self._print(f"=== 调试暂停:第 {batch_index + 1} 批 ===")
page1.pause()
# 选择所有数据
inner_frame.get_by_role("row", name="序号").get_by_label("").click()
# 点击输出
inner_frame.get_by_role("button", name="更多").hover()
inner_frame.get_by_text("输出", exact=True).click()
# 设置行数阈值
input_box = (
inner_frame.locator("div")
.filter(has_text=re.compile(r"^行数阈值$"))
.locator("input[type='text']")
)
input_box.fill("300000")
# 下载文件
download_path = f"D:/python/playwrite/data/temp_batch_{batch_index + 1}.xlsx"
with page1.expect_download() as download_info:
inner_frame.get_by_role("button", name="确定(Y)").click()
download = download_info.value
download.save_as(download_path)
self._print(f"{batch_index + 1} 批下载完成: {download_path}")
# 报告进度
self._report_progress(
"download",
batch_index + 1,
total_batches,
f"{batch_index + 1}/{total_batches} 批下载完成",
batch_index=batch_index + 1,
)
# 关闭输出对话框(如果有的话)
# try:
# inner_frame.get_by_role("button", name="取消").click()
# except:
# pass
# 等待页面恢复,准备下一次查询
time.sleep(1)
return download_path
def convert_and_merge_files(self, file_paths, output_path):
"""使用 ExcelConverter 转换并合并所有文件"""
# 确保输出文件路径是正确的格式
output_path = os.path.normpath(output_path)
output_dir = os.path.dirname(output_path)
output_filename = os.path.basename(output_path)
self._print(f"输出文件路径: {output_path}")
self._print(f"输出目录: {output_dir}")
self._print(f"输出文件名: {output_filename}")
# 确保输出文件的父目录存在
if output_dir and not os.path.exists(output_dir):
self._print(f"创建输出目录: {output_dir}")
os.makedirs(output_dir)
all_dataframes = []
for i, file_path in enumerate(file_paths, 1):
self._print(f"转换第 {i} 个文件: {file_path}")
# 报告进度
self._report_progress(
"convert",
i,
len(file_paths),
f"转换第 {i}/{len(file_paths)} 个文件",
file_index=i,
file_path=file_path,
)
df = self.converter.convert(file_path, output_file=None) # 只转换,不保存
all_dataframes.append(df)
self._print(f" 提取到 {len(df)} 条记录")
if all_dataframes:
self._print(f"\n合并 {len(all_dataframes)} 个文件的数据...")
merged_df = pd.concat(all_dataframes, ignore_index=True)
merged_df.to_excel(output_path, index=False)
self._print(f"合并完成: {output_path}, 总共 {len(merged_df)} 条记录")
# 删除临时文件
for file_path in file_paths:
os.remove(file_path)
self._print(f"已删除临时文件: {file_path}")
return output_path
return None
def setup_query_interface(self, inner_frame):
"""设置查询界面"""
import re
# 点击图标按钮打开查询界面
inner_frame.locator(".search-name-wrapper > .iconfont").click()
inner_frame.get_by_text("订单号查询").click()
inner_frame.get_by_role("tab", name="全部").click()
# 填充并验证,如果失败则重试
max_retries = 3
expected_value = "5000"
for attempt in range(max_retries):
inner_frame.locator("#rc_select_0").fill(expected_value)
inner_frame.locator("#rc_select_0").press("Enter")
# 检查填充是否成功
actual_value = inner_frame.locator("#rc_select_0").input_value()
if actual_value == expected_value:
self._print(f"文本框填充成功: {expected_value}")
break
else:
self._print(
f"{attempt + 1} 次填充失败,实际值: {actual_value},重试..."
)
if attempt == max_retries - 1:
self._print(
f"警告: {max_retries} 次尝试后仍未成功填充,继续执行..."
)
def extract(
self,
production_id_file,
data_dir="D:/python/playwrite/data",
output_file="D:/python/playwrite/data/离散备料计划维护_合并.xlsx",
debug_mode=False,
debug_batch=None,
progress_callback=None,
):
"""
执行完整的数据提取流程
Args:
production_id_file: ProductionID.txt 文件路径
data_dir: 数据保存目录
output_file: 最终输出文件路径
debug_mode: 是否启用调试模式
debug_batch: 调试批次号
progress_callback: 进度回调函数(覆盖初始化时的回调)
Returns:
输出文件路径
"""
# 保存原始回调
original_callback = self.progress_callback
# 使用传入的回调或初始化时的回调
self.progress_callback = progress_callback or self.progress_callback
try:
with sync_playwright() as playwright:
# 调用登录模块
browser, context, page, main_frame = login(
playwright=playwright,
username=self.username,
password=self.password,
headless=self.headless,
ignore_https_errors=True,
)
# 登录完成
self._report_progress("login", 1, 1, "登录成功")
self._print("=" * 80)
self._print("开始执行离散备料计划维护数据提取")
self._print("=" * 80)
# 登录成功后可以进行后续操作
# 点击打开"功能菜单"
main_frame.locator("i").first.click()
# 点击打开"离散备料计划维护"
with page.expect_popup() as page1_info:
main_frame.get_by_title(
"离散备料计划维护", exact=True
).first.click()
page1 = page1_info.value
# 获取 nested iframe
main_frame = page1.locator("#forwardFrame").content_frame
inner_frame_locator = main_frame.locator("#mainiframe")
inner_frame_locator.wait_for(state="visible", timeout=15000)
inner_frame = inner_frame_locator.content_frame
# 设置查询界面
self.setup_query_interface(inner_frame)
# 读取总排号并查询生产订单号
order_ids = self.get_production_order_numbers(
production_id_file, report_progress=True
)
# 按批次下载
downloaded_files = []
# 计算总批次数
total_batches = sum(
1 for _ in self.group_order_ids(order_ids, self.batch_size)
)
for batch_index, order_ids_batch in enumerate(
self.group_order_ids(order_ids, self.batch_size)
):
self._print(
f"\n=== 开始处理第 {batch_index + 1} 批,共 {len(order_ids_batch)} 个订单号 ==="
)
# 报告开始下载批次
self._report_progress(
"download",
batch_index,
total_batches,
f"正在下载第 {batch_index + 1}/{total_batches} 批...",
batch_index=batch_index + 1,
batch_size=len(order_ids_batch),
)
downloaded_file = self.download_batch(
inner_frame,
order_ids_batch,
batch_index,
total_batches,
page1,
debug_mode=debug_mode,
debug_batch=debug_batch,
)
downloaded_files.append(downloaded_file)
# 执行账号注销
self._print("\n开始执行账号注销...")
self._report_progress("logout", 1, 1, "正在注销账号...")
logout(main_frame, verbose=self.verbose)
# 转换并合并文件
if downloaded_files:
self._report_progress(
"convert",
0,
len(downloaded_files),
f"开始转换并合并 {len(downloaded_files)} 个文件",
file_count=len(downloaded_files),
)
self._print(
f"\n=== 开始转换并合并 {len(downloaded_files)} 个文件 ==="
)
self.convert_and_merge_files(downloaded_files, output_file)
else:
self._print("\n没有下载到任何文件")
self._print(f"\n=== 全部完成 ===")
self._print(f"最终文件: {output_file}")
# 报告完成
self._report_progress(
"complete", 1, 1, "提取完成", output_file=output_file
)
# 关闭浏览器
context.close()
browser.close()
return output_file
finally:
# 恢复原始回调
self.progress_callback = original_callback
def main():
"""测试函数"""
extractor = DiscreteMaterialPlanExtractor(
username="BLDpengqiangqiang",
password="Cqbld123456.",
headless=False,
verbose=True,
)
production_id_file = os.path.join(os.path.dirname(__file__), "productionID.txt")
output_file = "D:/python/playwrite/data/离散备料计划维护_合并.xlsx"
extractor.extract(production_id_file, output_file)
input("按回车退出...")
if __name__ == "__main__":
main()