""" 离散备料计划维护数据提取工具 负责登录、批量下载、转换数据 """ import os import pandas as pd from playwright.sync_api import sync_playwright from utils.excel_converter import ExcelConverter from utils.auth import login, logout from db.production_order_query import ( read_production_ids, query_production_order_numbers, ) from typing import Callable, Optional class DiscreteMaterialPlanExtractor: """离散备料计划维护数据提取器""" def __init__( self, username, password, headless=False, verbose=True, batch_size=100, enable_db_persistence=False ): """ 初始化提取器 Args: username: 登录用户名 password: 登录密码 headless: 是否无头模式运行 verbose: 是否打印详细日志 batch_size: 批次大小 enable_db_persistence: 是否启用数据库持久化 """ self.username = username self.password = password self.headless = headless self.verbose = verbose self.batch_size = batch_size self.progress_callback = None self.converter = ExcelConverter(verbose=verbose) self.enable_db_persistence = enable_db_persistence self.dao = None if self.enable_db_persistence: from db.discrete_material_plan_dao import DiscreteMaterialPlanDAO self.dao = DiscreteMaterialPlanDAO() self.dao.__enter__() # Enter context manager def _print(self, *args, **kwargs): """打印日志(如果 verbose=True)""" if self.verbose: print(*args, **kwargs) def _report_progress( self, stage: str, current: int, total: int, message: str, **detail ): """ 报告进度 Args: stage: 阶段标识 current: 当前进度值 total: 总量 message: 显示消息 **detail: 额外详细信息 """ if self.progress_callback: try: from gui.progress import ProgressInfo progress_info = ProgressInfo( stage=stage, current=current, total=total, message=message, detail=detail, ) self.progress_callback(progress_info) except Exception: # 如果进度回调失败,忽略错误,不影响主流程 pass def get_production_order_numbers(self, production_id_file, report_progress=False): """ 读取总排号文件并查询数据库获取生产订单号 Args: production_id_file: ProductionID.txt 文件路径 report_progress: 是否报告进度 Returns: 生产订单号列表 """ if report_progress: self._report_progress( "query", 1, 3, "正在读取总排号文件...", action="read_file", ) # 读取总排号 production_ids = read_production_ids(production_id_file) self._print(f"从文件读取到 {len(production_ids)} 个总排号") if report_progress: self._report_progress( "query", 2, 3, f"正在查询数据库({len(production_ids)} 个总排号)...", action="query_database", production_id_count=len(production_ids), ) # 查询数据库获取生产订单号 order_ids = query_production_order_numbers(production_ids) self._print(f"查询到 {len(order_ids)} 个生产订单号") if report_progress: self._report_progress( "query", 3, 3, f"查询完成:获取到 {len(order_ids)} 个生产订单号", action="query_complete", order_id_count=len(order_ids), ) return order_ids def group_order_ids(self, order_ids, group_size=100): """将订单号分组""" for i in range(0, len(order_ids), group_size): yield order_ids[i : i + group_size] def download_batch( self, inner_frame, order_ids, batch_index, total_batches, page1, debug_mode=False, debug_batch=None, ): """下载一批订单号的数据""" from playwright.sync_api import TimeoutError import re import time # 步骤1:清空文本框 self._report_progress( "download", batch_index * 7 + 1, total_batches * 7, f"第 {batch_index + 1}/{total_batches} 批 - 准备输入订单号", batch_index=batch_index + 1, action="clear_textbox", ) textbox = inner_frame.get_by_role("textbox", name="来源生产订单号") textbox.fill("") # 步骤2:填充订单号 self._report_progress( "download", batch_index * 7 + 2, total_batches * 7, f"第 {batch_index + 1}/{total_batches} 批 - 输入 {len(order_ids)} 个订单号", batch_index=batch_index + 1, action="fill_order_ids", order_count=len(order_ids), ) textbox.fill(",".join(order_ids)) # 步骤3:点击查询 self._report_progress( "download", batch_index * 7 + 3, total_batches * 7, f"第 {batch_index + 1}/{total_batches} 批 - 提交查询请求", batch_index=batch_index + 1, action="click_search", ) inner_frame.locator(".search-component-searchBtn").click() self._print(f"第 {batch_index + 1} 批查询完成,等待加载结果...") # 步骤4:等待加载完成 self._report_progress( "download", batch_index * 7 + 4, total_batches * 7, f"第 {batch_index + 1}/{total_batches} 批 - 等待数据加载...", batch_index=batch_index + 1, action="wait_loading", ) loading_locator = inner_frame.locator("div").filter(has_text="加载中").nth(1) try: loading_locator.wait_for(state="visible", timeout=3000) loading_locator.wait_for(state="hidden", timeout=0) # 无限等待,直到消失 except TimeoutError: # 加载很快完成,或者没有出现加载提示 pass self._print(f"第 {batch_index + 1} 批加载完成,开始选择数据...") # 调试模式:只在指定批次暂停 if debug_mode and (debug_batch is None or batch_index == debug_batch): self._print(f"=== 调试暂停:第 {batch_index + 1} 批 ===") page1.pause() # 步骤5:选择所有数据 self._report_progress( "download", batch_index * 7 + 5, total_batches * 7, f"第 {batch_index + 1}/{total_batches} 批 - 选择所有数据行", batch_index=batch_index + 1, action="select_all_rows", ) inner_frame.get_by_role("row", name="序号").get_by_label("").click() # 步骤6:配置并触发导出 self._report_progress( "download", batch_index * 7 + 6, total_batches * 7, f"第 {batch_index + 1}/{total_batches} 批 - 配置导出参数", batch_index=batch_index + 1, action="configure_export", ) # 点击输出 inner_frame.get_by_role("button", name="更多").hover() inner_frame.get_by_text("输出", exact=True).click() # 设置行数阈值 input_box = ( inner_frame.locator("div") .filter(has_text=re.compile(r"^行数阈值$")) .locator("input[type='text']") ) input_box.fill("300000") # 步骤7:下载文件 self._report_progress( "download", batch_index * 7 + 7, total_batches * 7, f"第 {batch_index + 1}/{total_batches} 批 - 正在下载文件...", batch_index=batch_index + 1, action="downloading_file", ) download_path = f"D:/python/playwrite/data/temp_batch_{batch_index + 1}.xlsx" with page1.expect_download() as download_info: inner_frame.get_by_role("button", name="确定(Y)").click() download = download_info.value download.save_as(download_path) self._print(f"第 {batch_index + 1} 批下载完成: {download_path}") # 报告批次完成 self._report_progress( "download", (batch_index + 1) * 7, total_batches * 7, f"第 {batch_index + 1}/{total_batches} 批下载完成 ✓", batch_index=batch_index + 1, action="batch_complete", file_path=download_path, ) # 等待页面恢复,准备下一次查询 time.sleep(1) return download_path def convert_and_merge_files(self, file_paths, output_path): """使用 ExcelConverter 转换并合并所有文件,返回合并后的 DataFrame""" # 确保输出文件路径是正确的格式 output_path = os.path.normpath(output_path) output_dir = os.path.dirname(output_path) output_filename = os.path.basename(output_path) self._print(f"输出文件路径: {output_path}") self._print(f"输出目录: {output_dir}") self._print(f"输出文件名: {output_filename}") # 步骤1:检查并创建输出目录 self._report_progress( "convert", 1, len(file_paths) * 2 + 3, "准备转换:检查输出目录", action="check_directory", ) if output_dir and not os.path.exists(output_dir): self._print(f"创建输出目录: {output_dir}") os.makedirs(output_dir) all_dataframes = [] # 步骤2-N:转换每个文件 for i, file_path in enumerate(file_paths, 1): self._print(f"转换第 {i} 个文件: {file_path}") # 报告开始转换 self._report_progress( "convert", 1 + (i - 1) * 2 + 1, len(file_paths) * 2 + 3, f"正在转换文件 {i}/{len(file_paths)}", file_index=i, file_path=file_path, action="converting_file", ) df = self.converter.convert(file_path, output_file=None) # 只转换,不保存 all_dataframes.append(df) self._print(f" 提取到 {len(df)} 条记录") # 报告转换完成 self._report_progress( "convert", 1 + (i - 1) * 2 + 2, len(file_paths) * 2 + 3, f"文件 {i}/{len(file_paths)} 转换完成({len(df)} 条记录)", file_index=i, record_count=len(df), action="file_converted", ) merged_df = None if all_dataframes: # 步骤N+1:合并数据 self._report_progress( "convert", len(file_paths) * 2 + 2, len(file_paths) * 2 + 3, f"正在合并 {len(all_dataframes)} 个文件的数据...", action="merging_data", file_count=len(all_dataframes), ) self._print(f"\n合并 {len(all_dataframes)} 个文件的数据...") merged_df = pd.concat(all_dataframes, ignore_index=True) merged_df.to_excel(output_path, index=False) self._print(f"合并完成: {output_path}, 总共 {len(merged_df)} 条记录") # 步骤N+2:删除临时文件 self._report_progress( "convert", len(file_paths) * 2 + 3, len(file_paths) * 2 + 3, f"清理临时文件...", action="cleanup", total_records=len(merged_df), ) for file_path in file_paths: os.remove(file_path) self._print(f"已删除临时文件: {file_path}") return output_path, merged_df return None, None def _save_to_database(self, df: pd.DataFrame): """Save DataFrame to database with progress reporting""" try: self._report_progress( "database", 0, 3, "准备保存到数据库...", action="db_start" ) stats = self.dao.save_dataframe_with_replace(df) self._report_progress( "database", 3, 3, f"数据库保存完成: 删除 {stats['deleted']} 条, 新增 {stats['inserted']} 条", action="db_complete", stats=stats ) self._print(f"\n数据库保存成功:") self._print(f" 删除旧记录: {stats['deleted']} 条") self._print(f" 新增记录: {stats['inserted']} 条") except Exception as e: self._print(f"\n警告: 数据库保存失败: {e}") self._report_progress( "database", 3, 3, f"数据库保存失败: {str(e)}", action="db_error", error=str(e) ) def setup_query_interface(self, inner_frame): """设置查询界面(不报告进度,由 extract 统一报告)""" import re # 打开查询界面 inner_frame.locator(".search-name-wrapper > .iconfont").click() inner_frame.get_by_text("订单号查询").click() # 选择"全部"标签 inner_frame.get_by_role("tab", name="全部").click() # 填充并验证 max_retries = 3 expected_value = "5000" for attempt in range(max_retries): inner_frame.locator("#rc_select_0").fill(expected_value) inner_frame.locator("#rc_select_0").press("Enter") actual_value = inner_frame.locator("#rc_select_0").input_value() if actual_value == expected_value: self._print(f"文本框填充成功: {expected_value}") break else: self._print( f"第 {attempt + 1} 次填充失败,实际值: {actual_value},重试..." ) if attempt == max_retries - 1: self._print( f"警告: {max_retries} 次尝试后仍未成功填充,继续执行..." ) def extract( self, production_id_file, data_dir="D:/python/playwrite/data", output_file="D:/python/playwrite/data/离散备料计划维护_合并.xlsx", debug_mode=False, debug_batch=None, progress_callback=None, ): """执行完整的数据提取流程""" original_callback = self.progress_callback self.progress_callback = progress_callback or self.progress_callback try: with sync_playwright() as playwright: # 步骤1:启动浏览器并登录 self._report_progress( "login", 1, 3, # 保持 3 步 "启动浏览器并登录...", action="launch_browser", ) browser, context, page, main_frame = login( playwright=playwright, username=self.username, password=self.password, headless=self.headless, ignore_https_errors=True, ) # 步骤2:打开功能页面 self._report_progress( "login", 2, 3, # 保持 3 步 "登录成功,打开功能页面...", action="open_function_page", ) self._print("=" * 80) self._print("开始执行离散备料计划维护数据提取") self._print("=" * 80) main_frame.locator("i").first.click() with page.expect_popup() as page1_info: main_frame.get_by_title( "离散备料计划维护", exact=True ).first.click() page1 = page1_info.value main_frame = page1.locator("#forwardFrame").content_frame inner_frame_locator = main_frame.locator("#mainiframe") inner_frame_locator.wait_for(state="visible", timeout=15000) inner_frame = inner_frame_locator.content_frame # 步骤3:设置查询界面 self._report_progress( "login", 3, 3, # 保持 3 步 "配置查询界面...", action="setup_query_interface", ) self.setup_query_interface(inner_frame) # 后续代码保持不变... order_ids = self.get_production_order_numbers( production_id_file, report_progress=True ) downloaded_files = [] total_batches = sum( 1 for _ in self.group_order_ids(order_ids, self.batch_size) ) for batch_index, order_ids_batch in enumerate( self.group_order_ids(order_ids, self.batch_size) ): self._print( f"\n=== 开始处理第 {batch_index + 1} 批,共 {len(order_ids_batch)} 个订单号 ===" ) downloaded_file = self.download_batch( inner_frame, order_ids_batch, batch_index, total_batches, page1, debug_mode=debug_mode, debug_batch=debug_batch, ) downloaded_files.append(downloaded_file) self._print("\n开始执行账号注销...") self._report_progress( "logout", 1, 2, "正在注销账号...", action="logout_start", ) logout(main_frame, verbose=self.verbose) self._report_progress( "logout", 2, 2, "注销完成 ✓", action="logout_complete", ) if downloaded_files: self._print( f"\n=== 开始转换并合并 {len(downloaded_files)} 个文件 ===" ) output_path, merged_df = self.convert_and_merge_files(downloaded_files, output_file) # 数据库保存步骤(独立阶段) if self.enable_db_persistence and self.dao and merged_df is not None: self._print(f"\n=== 开始保存数据到数据库 ===") self._save_to_database(merged_df) else: self._print("\n没有下载到任何文件") self._print(f"\n=== 全部完成 ===") self._print(f"最终文件: {output_file}") self._report_progress( "complete", 1, 1, "数据提取完成 ✓", output_file=output_file, action="all_complete", ) context.close() browser.close() return output_file finally: # Close database connection if open if self.dao: try: self.dao.__exit__(None, None, None) except Exception: pass self.progress_callback = original_callback def main(): """测试函数""" extractor = DiscreteMaterialPlanExtractor( username="BLDpengqiangqiang", password="Cqbld123456.", headless=False, verbose=True, ) production_id_file = os.path.join(os.path.dirname(__file__), "productionID.txt") output_file = "D:/python/playwrite/data/离散备料计划维护_合并.xlsx" extractor.extract(production_id_file, output_file) input("按回车退出...") if __name__ == "__main__": main()