# site_yunda.py import os import re import yaml from datetime import datetime, timedelta import pandas as pd from paths import DOWNLOAD_DIR, CONFIG_PATH def yunda_login(page): """ 韵达大运系统智能自动登录流 支持‘未登录自动填充提交’与‘已登录静默过客’双工模式 """ print(">> 正在探测韵达当前的登录会话状态...") try: # 定位“账号密码登录”切换按钮 switch_btn = page.locator("span", has_text="账号密码登录") # 设定 5 秒探针延迟。如果 5 秒内按钮可见,说明处于未登录的初始状态 if switch_btn.is_visible(timeout=5000): print(" -> 捕获到标准的未登录界面,正在强切至【账号密码登录】模式...") switch_btn.click() page.wait_for_timeout(500) # 从配置读取凭证(默认空串;真实凭据仅存于被忽略的 config.yaml) username = "" password = "" if os.path.exists(CONFIG_PATH): with open(CONFIG_PATH, "r", encoding="utf-8") as f: config = yaml.safe_load(f) or {} yd_cfg = config.get("yunda", {}) username = str(yd_cfg.get("username", "")) password = str(yd_cfg.get("password", "")) print(f" -> 正在程序化充填表单凭证 (账号: {username})...") page.locator("#username").fill(username) page.locator("#password").fill(password) page.wait_for_timeout(300) print(" -> 正在点击【登录】按钮并提交表单...") page.locator('button[type="submit"]', has_text="登录").click() page.wait_for_timeout(1000) else: print( " -> 探针未发现登录按钮,判定当前会话已处于持久化登录状态,静默穿透。" ) except Exception as e: print(f" ⚠️ 自动登录状态机探测发生波动(可能已处于工作台内部): {e}") def yunda_smart_menu_click(page, menu_path): """ 韵达排他性多级树形菜单智能导航器(带防折叠状态断言机制) 采用 XPath 亲子轴绝对隔离技术,彻底斩断嵌套手风琴菜单的向下漏包干扰 """ print(f">> 正在智能路由韵达菜单: {' -> '.join(menu_path)}") for item in menu_path: title_locator = page.locator( f"xpath=//div[contains(@class, 'el-submenu__title') and .//span[normalize-space(.)='{item}']]" ).first parent_li = page.locator( f"xpath=//div[contains(@class, 'el-submenu__title') and .//span[normalize-space(.)='{item}']]/.." ).first leaf_locator = page.locator( f"xpath=//li[contains(@class, 'el-menu-item') and .//span[normalize-space(.)='{item}']]" ).first if title_locator.is_visible(): current_class = parent_li.get_attribute("class") or "" is_opened = "is-opened" in current_class if not is_opened: print(f" -> 探测到父菜单 [{item}] 当前处于【收起】状态,执行点击展开") title_locator.click() page.wait_for_timeout(500) else: print( f" -> 探测到父菜单 [{item}] 当前已经处于【展开】状态,安全跳过点击(防折叠保护激活)" ) elif leaf_locator.is_visible(): print(f" -> 成功对焦目标叶子节点 [{item}],直接执行跳转点击") leaf_locator.click() page.wait_for_timeout(1000) def yunda_expected_download(page): """韵达:应到货物数据下载""" print("\n▶ 开始执行【韵达 - 应到货物数据下载】任务...") target_task_title = "进站主单表" download_dir = DOWNLOAD_DIR if not os.path.exists(download_dir): os.makedirs(download_dir) export_times = [] try: # 1. 验证首页并智能路由 page.locator(".el-menu-item", has_text="首页").wait_for( state="visible", timeout=15000 ) print("✅ 韵达工作台首页成功加载") yunda_smart_menu_click(page, ["运营管理", "进站管理", "进站交接单查询"]) print(">> 正在跨域动态追踪【进站交接单查询】业务窗体...") # 彻底移除会引起竞速超时坑的 custom 探测器,使用原生懒加载 frame_locator ws_frame = page.frame_locator("section iframe") ws_frame.locator("#startTime").wait_for(state="attached", timeout=15000) print("✅ 进站交接单查询工作区就绪") # 2. 从配置文件中解析并计算绝对日期跨度 query_days = 1 try: if os.path.exists(CONFIG_PATH): with open(CONFIG_PATH, "r", encoding="utf-8") as f: config = yaml.safe_load(f) or {} query_days = int(config.get("yunda", {}).get("query_days", 1)) except Exception as e: print(f" ⚠️ 读取 config.yaml 失败,默认查询 1 天: {e}") today = datetime.now() start_date = today - timedelta(days=(query_days - 1)) today_ymd = f"{today.year}-{today.month}-{today.day}" start_date_ymd = f"{start_date.year}-{start_date.month}-{start_date.day}" print(f">> 正在精准对焦时间区间: [{start_date_ymd}] 至 [{today_ymd}]") # 设定起始时间 print(" >> 呼出起始时间控件...") page.wait_for_timeout(1000) ws_frame.locator("#startTime").click(force=True) calendar1 = ws_frame.locator(".layui-laydate:visible").first calendar1.wait_for(state="visible", timeout=5000) calendar1.locator(f"td[lay-ymd='{start_date_ymd}']").click() calendar1.locator(".laydate-btns-confirm").click() page.wait_for_timeout(400) # 设定截止时间 print(" >> 呼出截止时间控件...") ws_frame.locator("#endTime").click(force=True) calendar2 = ws_frame.locator(".layui-laydate:visible").first calendar2.wait_for(state="visible", timeout=5000) calendar2.locator(f"td[lay-ymd='{today_ymd}']").click() calendar2.locator(".laydate-btns-confirm").click() page.wait_for_timeout(500) # 3. 多维状态机静默加载校验 print(">> 正在触发现场查询数据流...") ws_frame.locator("a.btn-success", has_text="查询").click() page.wait_for_timeout(800) # ==================================================================== # 🛡️ 核心修复:将 Loading 蒙层和数据断言限制在当前的 #tab-1 结界内部 # 彻底解决多标签页导致 Strict Mode (5 elements found) 的污染问题 # ==================================================================== loading_mask = ws_frame.locator( "#tab-1 .fixed-table-loading", has_text="正在努力地加载数据中" ).first if loading_mask.is_visible(): print(" ⏳ 检测到专属数据加载罩,正在等待后端重载返回...") loading_mask.wait_for(state="hidden", timeout=30000) page.wait_for_timeout(500) sum_panel = ws_frame.locator("#sum").first has_data = False if sum_panel.is_visible(): sum_text = sum_panel.inner_text() match_tickets = re.search(r"进站实际票数:(\d+)", sum_text) if match_tickets and int(match_tickets.group(1)) > 0: has_data = True print( f" ✅ 深度断言:局部统计面板加载完毕,实际票数: [{match_tickets.group(1)}],执行穿透。" ) if not has_data: # 同样将空数据提示隔离在当前 tab 中 if ws_frame.locator( "#tab-1 .no-records-found", has_text="没有找到匹配的记录" ).first.is_visible(): print( " ⚠️ 确认为空数据环境:系统提示【没有找到匹配的记录】,正在执行复位净化..." ) page.locator(".tags-view-item", has_text="进站交接单查询").locator( ".el-icon-close" ).click() return # 4. 深度等待表格第一行数据行渲染就绪 ws_frame.locator("#exampleTable1 tbody tr[data-index='0']").wait_for( state="visible", timeout=10000 ) main_rows = ws_frame.locator("#exampleTable1 tbody tr[data-index]") row_count = main_rows.count() print(f">> 当前视窗共捕获到活跃交接单记录: {row_count} 条") # 5. 循环双击穿透提交 for i in range(row_count): print(f" ⏳ 正在处理第 {i+1}/{row_count} 个交接单模块...") current_row = ws_frame.locator("#exampleTable1 tbody tr[data-index]").nth(i) raw_no = current_row.locator("td").nth(1).inner_text().strip() # ==================================================================== # 🛡️ 状态机拦截卫语句:识别绑定状态,过滤已绑定交接单 # ==================================================================== bind_status = current_row.locator("td").nth(2).inner_text().strip() print(f" -> 锁定提取单号: {raw_no} [绑定状态: {bind_status}]") if bind_status == "已绑定": print(" ⏭️ 状态拦截:该交接单处于【已绑定】状态,安全跳过。") continue # ==================================================================== current_row.dblclick() ws_frame.locator("#docSum").wait_for(state="visible", timeout=15000) page.wait_for_timeout(500) ws_frame.locator('a.btn-info[onclick*="exportFile"]').click() ws_frame.locator(".layui-layer-title", has_text="数据导出").wait_for( state="visible", timeout=15000 ) export_frame = ws_frame.frame_locator('iframe[name="target1"]') export_frame.locator(".allRight").click() page.wait_for_timeout(400) print(" >> 正在建立后台离线任务...") task_success = False for attempt in range(5): export_frame.locator("#submitbutton", has_text="导出数据").click() confirm_link = export_frame.get_by_role("link", name="确定") try: confirm_link.wait_for(state="visible", timeout=6000) if export_frame.get_by_text("导出任务建立成功").is_visible(): print(" ✅ 判定通过:成功捕获到【导出任务建立成功】特征!") confirm_link.click() task_success = True break elif export_frame.get_by_text( "请选择格式相应的导出字段" ).is_visible(): print( " ⚠️ 警告:检测到【未选择字段】错误,重新触发补点全选..." ) confirm_link.click() page.wait_for_timeout(500) export_frame.locator(".allRight").click() page.wait_for_timeout(500) else: confirm_link.click() page.wait_for_timeout(1000) except Exception: page.wait_for_timeout(1000) if not task_success: raise RuntimeError( "致命异常:连续 5 次尝试均无法成功建立应到数据离线任务。" ) ws_frame.locator(".layui-layer-close1").click() page.wait_for_timeout(500) ws_frame.locator("#myTab a", has_text="交接单信息").click() page.wait_for_timeout(800) export_times.append(datetime.now()) print(">> 📤 任务提交流闭环,正在执行【进站交接单查询】工作台销毁...") page.locator(".tags-view-item", has_text="进站交接单查询").locator( ".el-icon-close" ).click() page.wait_for_timeout(500) # 防线:如果全部记录都被跳过了,export_times 为空,直接结束 if not export_times: print( ">> ⚠️ 本次查询未产生任何有效的离线下载任务(全部空单或已被跳过),中止后端收割流。" ) return _yunda_poll_and_download_tasks( page, export_times, target_task_title, download_dir, "韵达-应到货物数据.xlsx", ) except Exception as e: print(f"\n❌ 任务执行过程中发生异常: {e}") def yunda_actual_download(page): """韵达:实到货物数据下载""" print("\n▶ 开始执行【韵达 - 实到货物数据下载】任务...") target_task_title = "扫描记录数据" download_dir = DOWNLOAD_DIR if not os.path.exists(download_dir): os.makedirs(download_dir) export_times = [] try: page.locator(".el-menu-item", has_text="首页").wait_for( state="visible", timeout=15000 ) print("✅ 韵达工作台首页成功加载") yunda_smart_menu_click(page, ["报表管理", "扫描记录查询"]) print(">> 正在跨域动态追踪【扫描记录查询】业务窗体...") ws_frame = page.frame_locator("section iframe") ws_frame.locator( ".no-records-found", has_text="没有找到匹配的记录" ).first.wait_for(state="visible", timeout=15000) print("✅ 扫描记录查询工作区初始化完毕") query_days = 1 try: if os.path.exists(CONFIG_PATH): with open(CONFIG_PATH, "r", encoding="utf-8") as f: config = yaml.safe_load(f) or {} query_days = int(config.get("yunda", {}).get("query_days", 1)) except Exception: pass today = datetime.now() start_date = today - timedelta(days=(query_days - 1)) print(f">> 正在精准对焦实到时间区间: [近 {query_days} 天]") print(" >> 正在设定起始时间...") ws_frame.locator("#startDate").click() page.wait_for_timeout(400) box1 = ws_frame.locator("#laydate_box:visible").first box1.locator( f"td[y='{start_date.year}'][m='{start_date.month}'][d='{start_date.day}']" ).click() page.wait_for_timeout(400) print(" >> 正在设定截止时间...") ws_frame.locator("#endDate").click() page.wait_for_timeout(400) box2 = ws_frame.locator("#laydate_box:visible").first box2.locator( f"td[y='{today.year}'][m='{today.month}'][d='{today.day}']" ).click() page.wait_for_timeout(500) print(" >> 正在变更扫描类型为【到件】...") ws_frame.locator("#scanRecordTyp").select_option(value="03") page.wait_for_timeout(500) print(">> 正在触发现场查询数据流...") ws_frame.locator('input[type="button"][value="查询"]').click() page.wait_for_timeout(800) loading_mask = ws_frame.locator( ".fixed-table-loading", has_text="正在努力地加载数据中" ).first if loading_mask.is_visible(): print(" ⏳ 检测到专属数据加载罩,正在等待后端重载返回...") loading_mask.wait_for(state="hidden", timeout=30000) page.wait_for_timeout(500) pg_info = ws_frame.locator(".pagination-info").first has_records = False if pg_info.is_visible(): info_text = pg_info.inner_text() match_total = re.search(r"总共\s*(\d+)\s*条记录", info_text) if match_total and int(match_total.group(1)) > 0: has_records = True print( f" ✅ 深度断言:实到数据渲染完毕,总记录数: [{match_total.group(1)}] 条。" ) if not has_records: if ws_frame.locator( ".no-records-found", has_text="没有找到匹配的记录" ).first.is_visible(): print( " ⚠️ 当前查询范围内确认为空数据:系统提示【没有找到匹配的记录】,终止并关闭环境。" ) page.locator(".tags-view-item", has_text="扫描记录查询").locator( ".el-icon-close" ).click() return else: print(" ⚠️ 发生渲染异常,未找到数据也未找到空记录提示。安全收尾...") page.locator(".tags-view-item", has_text="扫描记录查询").locator( ".el-icon-close" ).click() return # 5. 执行导出流 print(">> 正在发起【导出】申请指令...") ws_frame.locator('input[type="button"][id="export"]').click() ws_frame.locator(".layui-layer-title", has_text="数据导出").wait_for( state="visible", timeout=15000 ) export_frame = ws_frame.frame_locator('iframe[name="myFrame"]') export_frame.locator(".allRight").click() page.wait_for_timeout(400) print(" >> 正在建立后台离线任务...") task_success = False for attempt in range(5): export_frame.locator("#submitbutton", has_text="导出数据").click() confirm_link = export_frame.get_by_role("link", name="确定") try: confirm_link.wait_for(state="visible", timeout=6000) if export_frame.get_by_text("导出任务建立成功").is_visible(): print(" ✅ 判定通过:成功捕获到【导出任务建立成功】特征!") confirm_link.click() task_success = True break elif export_frame.get_by_text("请选择格式相应的导出字段").is_visible(): print(" ⚠️ 警告:检测到字段未全选,执行强补点击...") confirm_link.click() page.wait_for_timeout(500) export_frame.locator(".allRight").click() page.wait_for_timeout(500) else: confirm_link.click() page.wait_for_timeout(1000) except Exception: page.wait_for_timeout(1000) if not task_success: raise RuntimeError( "致命异常:连续 5 次尝试均无法成功建立实到数据离线任务。" ) ws_frame.locator(".layui-layer-close1").click() page.wait_for_timeout(500) export_times.append(datetime.now()) print(">> 📤 任务提交流闭环,正在执行【扫描记录查询】工作台销毁...") page.locator(".tags-view-item", has_text="扫描记录查询").locator( ".el-icon-close" ).click() page.wait_for_timeout(500) # 防线:双重保护 if not export_times: print(">> ⚠️ 本次查询未产生有效的离线下载任务,中止后端收割流。") return # 6. 收割下载 _yunda_poll_and_download_tasks( page, export_times, target_task_title, download_dir, "韵达-实到货物数据.xlsx", ) except Exception as e: print(f"\n❌ 任务执行过程中发生异常: {e}") def _yunda_poll_and_download_tasks( page, export_times, target_task_title, download_dir, final_filename ): """韵达专属离线任务轮询下载引擎""" print("\n>> 正在前往【导出服务】中心...") yunda_smart_menu_click(page, ["基础数据", "导出服务"]) export_ws_frame = page.frame_locator("section iframe") export_ws_frame.get_by_role("cell", name="模块名称", exact=True).wait_for( state="visible", timeout=15000 ) page.wait_for_timeout(1000) print(">> 离线文件队列已对接,启动【1分钟高频精确校对+缺单局部重载刷新】断言...") total_expected = len(export_times) while True: task_rows = export_ws_frame.locator( ".datagrid-view2 .datagrid-btable tbody tr.datagrid-row" ) row_count = task_rows.count() ready_indices = [] processing_indices = [] for idx in range(row_count): row = task_rows.nth(idx) module_name = row.locator("td[field='modueName']").inner_text().strip() status_name = row.locator("td[field='fileStatus']").inner_text().strip() create_time_str = ( row.locator("td[field='createdTime']").inner_text().strip() ) if module_name == target_task_title: try: row_time = datetime.strptime(create_time_str, "%Y-%m-%d %H:%M:%S") matched = any( abs((row_time - et).total_seconds()) <= 60 for et in export_times ) if matched: if status_name == "导出完成": ready_indices.append(idx) else: processing_indices.append(idx) except Exception: pass total_found = len(ready_indices) + len(processing_indices) print( f" 📊 状态研判:自建期望数 [{total_expected}],实际入表 [{total_found}] (完成 [{len(ready_indices)}],生成中 [{len(processing_indices)}])" ) if total_found < total_expected or len(processing_indices) > 0: print(" ⏳ 队列未齐,执行【点击查询按钮】触发局部无痕刷新...") export_ws_frame.locator( "#ydkyimport_basic_export_searchData1_ky_export_common" ).click() page.wait_for_timeout(3000) else: print(">> ✅ 所有目标离线任务全量就绪!开始依序接入文件流...") break downloaded_files = [] for row_idx in ready_indices: try: target_row = export_ws_frame.locator( ".datagrid-view2 .datagrid-btable tbody tr.datagrid-row" ).nth(row_idx) time_flag = ( target_row.locator("td[field='createdTime']").inner_text().strip() ) print(f" 🎯 触发下载 -> 离线任务时间节点: [{time_flag}] ...") with page.expect_download() as download_info: target_row.locator("td[field='extreFile'] a").get_by_text( "下载" ).first.click() download = download_info.value safe_timestamp = datetime.now().strftime("%Y%m%d_%H%M%S_%f") custom_filename = f"韵达_temp_{safe_timestamp}.xlsx" save_path = os.path.join(download_dir, custom_filename) download.save_as(save_path) downloaded_files.append(save_path) print(f" ⬇️ 文件已用安全序列号落盘: downloads/{custom_filename}") page.wait_for_timeout(500) except Exception as e: print(f" ❌ 文件流接收失败: {e}") print(">> 📥 【导出服务】数据提取链闭环,正在执行当前 Tab 窗口销毁...") try: page.locator(".tags-view-item", has_text="导出服务").locator( ".el-icon-close" ).click() print(" ✅ 【导出服务】工作区已安全关闭。") except Exception: pass if downloaded_files: print("\n>> 🧪 正在启动离线数据清洗与高能扁平合并流...") all_dfs = [] for file_path in downloaded_files: try: df = pd.read_excel(file_path, dtype=str) if not df.empty: all_dfs.append(df) except Exception: pass if all_dfs: combined_df = pd.concat(all_dfs, ignore_index=True) final_output = os.path.join(download_dir, final_filename) combined_df.to_excel(final_output, index=False) print(f"====================================================") print(f" 🎉 恭喜!韵达网点数据拉取合流大功告成!") print(f" 📁 最终存储位置: {final_output}") print(f"====================================================") for file_path in downloaded_files: os.remove(file_path) print(" ✅ 临时缓存阵列已无缝净化。")