""" 离散备料计划维护数据提取工具 负责登录、批量下载、转换数据 """ import os import pandas as pd from playwright.sync_api import sync_playwright from utils.excel_converter import ExcelConverter from utils.auth import login, logout from db.production_order_query import read_production_ids, query_production_order_numbers from typing import Callable, Optional class DiscreteMaterialPlanExtractor: """离散备料计划维护数据提取器""" def __init__(self, username, password, headless=False, verbose=True, batch_size=100): """ 初始化提取器 Args: username: 登录用户名 password: 登录密码 headless: 是否无头模式运行 verbose: 是否打印详细日志 batch_size: 批次大小 """ self.username = username self.password = password self.headless = headless self.verbose = verbose self.batch_size = batch_size self.progress_callback = None self.converter = ExcelConverter(verbose=verbose) def _print(self, *args, **kwargs): """打印日志(如果 verbose=True)""" if self.verbose: print(*args, **kwargs) def _report_progress(self, stage: str, current: int, total: int, message: str, **detail): """ 报告进度 Args: stage: 阶段标识 current: 当前进度值 total: 总量 message: 显示消息 **detail: 额外详细信息 """ if self.progress_callback: try: from gui.progress import ProgressInfo progress_info = ProgressInfo( stage=stage, current=current, total=total, message=message, detail=detail ) self.progress_callback(progress_info) except Exception: # 如果进度回调失败,忽略错误,不影响主流程 pass def get_production_order_numbers(self, production_id_file, report_progress=False): """ 读取总排号文件并查询数据库获取生产订单号 Args: production_id_file: ProductionID.txt 文件路径 report_progress: 是否报告进度 Returns: 生产订单号列表 """ # 读取总排号 production_ids = read_production_ids(production_id_file) self._print(f"从文件读取到 {len(production_ids)} 个总排号") # 查询数据库获取生产订单号 order_ids = query_production_order_numbers(production_ids) self._print(f"查询到 {len(order_ids)} 个生产订单号") if report_progress: self._report_progress('query', 1, 1, f"查询到 {len(order_ids)} 个生产订单号", count=len(order_ids)) return order_ids def group_order_ids(self, order_ids, group_size=100): """将订单号分组""" for i in range(0, len(order_ids), group_size): yield order_ids[i:i + group_size] def download_batch(self, inner_frame, order_ids, batch_index, total_batches, page1, debug_mode=False, debug_batch=None): """下载一批订单号的数据""" from playwright.sync_api import TimeoutError import re import time # 清空文本框 textbox = inner_frame.get_by_role("textbox", name="来源生产订单号") textbox.fill("") # 填充订单号 textbox.fill(",".join(order_ids)) # 点击查询 inner_frame.locator(".search-component-searchBtn").click() self._print(f"第 {batch_index + 1} 批查询完成,等待加载结果...") # 等待加载完成 loading_locator = inner_frame.locator("div").filter(has_text="加载中").nth(1) try: loading_locator.wait_for(state="visible", timeout=3000) loading_locator.wait_for(state="hidden", timeout=0) # 无限等待,直到消失 except TimeoutError: # 加载很快完成,或者没有出现加载提示 pass self._print(f"第 {batch_index + 1} 批加载完成,开始选择数据...") # 调试模式:只在指定批次暂停 if debug_mode and (debug_batch is None or batch_index == debug_batch): self._print(f"=== 调试暂停:第 {batch_index + 1} 批 ===") page1.pause() # 选择所有数据 inner_frame.get_by_role("row", name="序号").get_by_label("").click() # 点击输出 inner_frame.get_by_role("button", name="更多").hover() inner_frame.get_by_text("输出", exact=True).click() # 设置行数阈值 input_box = inner_frame.locator("div").filter(has_text=re.compile(r"^行数阈值$")).locator("input[type='text']") input_box.fill("300000") # 下载文件 download_path = f"D:/python/playwrite/data/temp_batch_{batch_index + 1}.xlsx" with page1.expect_download() as download_info: inner_frame.get_by_role("button", name="确定(Y)").click() download = download_info.value download.save_as(download_path) self._print(f"第 {batch_index + 1} 批下载完成: {download_path}") # 报告进度 self._report_progress( 'download', batch_index + 1, total_batches, f"第 {batch_index + 1}/{total_batches} 批下载完成", batch_index=batch_index + 1 ) # 关闭输出对话框(如果有的话) # try: # inner_frame.get_by_role("button", name="取消").click() # except: # pass # 等待页面恢复,准备下一次查询 time.sleep(1) return download_path def convert_and_merge_files(self, file_paths, output_path): """使用 ExcelConverter 转换并合并所有文件""" # 确保输出文件路径是正确的格式 output_path = os.path.normpath(output_path) output_dir = os.path.dirname(output_path) output_filename = os.path.basename(output_path) self._print(f"输出文件路径: {output_path}") self._print(f"输出目录: {output_dir}") self._print(f"输出文件名: {output_filename}") # 确保输出文件的父目录存在 if output_dir and not os.path.exists(output_dir): self._print(f"创建输出目录: {output_dir}") os.makedirs(output_dir) all_dataframes = [] for i, file_path in enumerate(file_paths, 1): self._print(f"转换第 {i} 个文件: {file_path}") # 报告进度 self._report_progress( 'convert', i, len(file_paths), f"转换第 {i}/{len(file_paths)} 个文件", file_index=i, file_path=file_path ) df = self.converter.convert(file_path, output_file=None) # 只转换,不保存 all_dataframes.append(df) self._print(f" 提取到 {len(df)} 条记录") if all_dataframes: self._print(f"\n合并 {len(all_dataframes)} 个文件的数据...") merged_df = pd.concat(all_dataframes, ignore_index=True) merged_df.to_excel(output_path, index=False) self._print(f"合并完成: {output_path}, 总共 {len(merged_df)} 条记录") # 删除临时文件 for file_path in file_paths: os.remove(file_path) self._print(f"已删除临时文件: {file_path}") return output_path return None def setup_query_interface(self, inner_frame): """设置查询界面""" import re # 点击图标按钮打开查询界面 inner_frame.locator(".search-name-wrapper > .iconfont").click() inner_frame.get_by_text("订单号查询").click() inner_frame.get_by_role("tab", name="全部").click() # 填充并验证,如果失败则重试 max_retries = 3 expected_value = "5000" for attempt in range(max_retries): inner_frame.locator("#rc_select_0").fill(expected_value) inner_frame.locator("#rc_select_0").press("Enter") # 检查填充是否成功 actual_value = inner_frame.locator("#rc_select_0").input_value() if actual_value == expected_value: self._print(f"文本框填充成功: {expected_value}") break else: self._print(f"第 {attempt + 1} 次填充失败,实际值: {actual_value},重试...") if attempt == max_retries - 1: self._print(f"警告: {max_retries} 次尝试后仍未成功填充,继续执行...") def extract(self, production_id_file, data_dir="D:/python/playwrite/data", output_file="D:/python/playwrite/data/离散备料计划维护_合并.xlsx", debug_mode=False, debug_batch=None, progress_callback=None): """ 执行完整的数据提取流程 Args: production_id_file: ProductionID.txt 文件路径 data_dir: 数据保存目录 output_file: 最终输出文件路径 debug_mode: 是否启用调试模式 debug_batch: 调试批次号 progress_callback: 进度回调函数(覆盖初始化时的回调) Returns: 输出文件路径 """ # 保存原始回调 original_callback = self.progress_callback # 使用传入的回调或初始化时的回调 self.progress_callback = progress_callback or self.progress_callback try: with sync_playwright() as playwright: # 调用登录模块 browser, context, page, main_frame = login( playwright=playwright, username=self.username, password=self.password, headless=self.headless, ignore_https_errors=True ) # 登录完成 self._report_progress('login', 1, 1, "登录成功") self._print("=" * 80) self._print("开始执行离散备料计划维护数据提取") self._print("=" * 80) # 登录成功后可以进行后续操作 # 点击打开"功能菜单" main_frame.locator("i").first.click() # 点击打开"离散备料计划维护" with page.expect_popup() as page1_info: main_frame.get_by_title("离散备料计划维护", exact=True).first.click() page1 = page1_info.value # 获取 nested iframe main_frame = page1.locator("#forwardFrame").content_frame inner_frame_locator = main_frame.locator("#mainiframe") inner_frame_locator.wait_for(state="visible", timeout=15000) inner_frame = inner_frame_locator.content_frame # 设置查询界面 self.setup_query_interface(inner_frame) # 读取总排号并查询生产订单号 order_ids = self.get_production_order_numbers(production_id_file, report_progress=True) # 按批次下载 downloaded_files = [] # 计算总批次数 total_batches = sum(1 for _ in self.group_order_ids(order_ids, self.batch_size)) for batch_index, order_ids_batch in enumerate(self.group_order_ids(order_ids, self.batch_size)): self._print(f"\n=== 开始处理第 {batch_index + 1} 批,共 {len(order_ids_batch)} 个订单号 ===") # 报告开始下载批次 self._report_progress( 'download', batch_index, total_batches, f"正在下载第 {batch_index + 1}/{total_batches} 批...", batch_index=batch_index + 1, batch_size=len(order_ids_batch) ) downloaded_file = self.download_batch( inner_frame, order_ids_batch, batch_index, total_batches, page1, debug_mode=debug_mode, debug_batch=debug_batch ) downloaded_files.append(downloaded_file) # 执行账号注销 self._print("\n开始执行账号注销...") self._report_progress('logout', 1, 1, "正在注销账号...") logout(main_frame, verbose=self.verbose) # 转换并合并文件 if downloaded_files: self._report_progress( 'convert', 0, len(downloaded_files), f"开始转换并合并 {len(downloaded_files)} 个文件", file_count=len(downloaded_files) ) self._print(f"\n=== 开始转换并合并 {len(downloaded_files)} 个文件 ===") self.convert_and_merge_files(downloaded_files, output_file) else: self._print("\n没有下载到任何文件") self._print(f"\n=== 全部完成 ===") self._print(f"最终文件: {output_file}") # 报告完成 self._report_progress('complete', 1, 1, "提取完成", output_file=output_file) # 关闭浏览器 context.close() browser.close() return output_file finally: # 恢复原始回调 self.progress_callback = original_callback def main(): """测试函数""" extractor = DiscreteMaterialPlanExtractor( username="BLDpengqiangqiang", password="Cqbld123456.", headless=False, verbose=True ) production_id_file = os.path.join(os.path.dirname(__file__), "productionID.txt") output_file = "D:/python/playwrite/data/离散备料计划维护_合并.xlsx" extractor.extract(production_id_file, output_file) input("按回车退出...") if __name__ == "__main__": main()