- Implement yunda_actual_download via the 报表管理 -> 扫描记录查询 menu: set date range, switch scan type to 到件, then reuse the shared poll/download/merge engine - Add a three-stage query buffer (yield, wait for the Bootstrap Table loading mask, settle) before the data/empty verdict to avoid stale DOM reads in the expected-arrival flow - Trim the offline-export retry state machine's verbose logs - Fix a typo (研编 -> 研判) and drop the 待开发 tag on menu [7] Co-Authored-By: Claude <noreply@anthropic.com>
568 lines
22 KiB
Python
568 lines
22 KiB
Python
# site_yunda.py
|
||
|
||
import os
|
||
import re
|
||
import yaml
|
||
from datetime import datetime, timedelta
|
||
import pandas as pd
|
||
|
||
|
||
def _wait_and_get_frame(page, text_indicator, timeout_ms=20000):
|
||
"""【文本雷达探测器】全域跨框架检索包含特定文本的活动上下文"""
|
||
start_time = datetime.now()
|
||
while (datetime.now() - start_time).total_seconds() * 1000 < timeout_ms:
|
||
try:
|
||
if page.get_by_text(text_indicator).count() > 0:
|
||
return page
|
||
except:
|
||
pass
|
||
|
||
for frame in page.frames:
|
||
try:
|
||
if frame.get_by_text(text_indicator).count() > 0:
|
||
return frame
|
||
except:
|
||
pass
|
||
|
||
page.wait_for_timeout(300)
|
||
raise TimeoutError(
|
||
f"爆栈超时:未能在活动框架中死守到包含 [{text_indicator}] 的视窗。"
|
||
)
|
||
|
||
|
||
def _wait_and_get_frame_by_selector(page, selector, timeout_ms=20000):
|
||
"""【控件雷达探测器】无视 IFrame 的 src 或层级,直接扫描谁包含指定 CSS 控件"""
|
||
start_time = datetime.now()
|
||
while (datetime.now() - start_time).total_seconds() * 1000 < timeout_ms:
|
||
try:
|
||
if page.locator(selector).count() > 0:
|
||
return page
|
||
except:
|
||
pass
|
||
|
||
for frame in page.frames:
|
||
try:
|
||
if frame.locator(selector).count() > 0:
|
||
return frame
|
||
except:
|
||
pass
|
||
|
||
page.wait_for_timeout(300)
|
||
raise TimeoutError(f"爆栈超时:全域未探测到包含控件 [{selector}] 的业务框架。")
|
||
|
||
|
||
def yunda_smart_menu_click(page, menu_path):
|
||
"""韵达排他性二级树形菜单智能展开器"""
|
||
print(f">> 正在导航韵达菜单: {' -> '.join(menu_path)}")
|
||
for item in menu_path:
|
||
locator = page.locator(".el-submenu__title, .el-menu-item", has_text=item).first
|
||
locator.click()
|
||
page.wait_for_timeout(600)
|
||
|
||
|
||
def yunda_expected_download(page):
|
||
"""韵达:应到货物数据下载"""
|
||
print("\n▶ 开始执行【韵达 - 应到货物数据下载】任务...")
|
||
target_task_title = "进站主单表"
|
||
|
||
download_dir = os.path.join(os.getcwd(), "downloads")
|
||
if not os.path.exists(download_dir):
|
||
os.makedirs(download_dir)
|
||
|
||
export_times = []
|
||
|
||
try:
|
||
# 1. 验证首页并智能路由
|
||
page.locator(".el-menu-item", has_text="首页").wait_for(
|
||
state="visible", timeout=15000
|
||
)
|
||
print("✅ 韵达工作台首页成功加载")
|
||
|
||
yunda_smart_menu_click(page, ["运营管理", "进站管理", "进站交接单查询"])
|
||
|
||
print(">> [雷达扫描] 正在跨域动态追踪【进站交接单查询】业务窗体...")
|
||
ws_frame = _wait_and_get_frame_by_selector(page, "#startTime")
|
||
|
||
ws_frame.locator("#startTime").wait_for(state="attached", timeout=15000)
|
||
print("✅ 进站交接单查询工作区就绪")
|
||
|
||
# 2. 从配置文件中解析并计算绝对日期跨度
|
||
query_days = 1
|
||
try:
|
||
if os.path.exists("config.yaml"):
|
||
with open("config.yaml", "r", encoding="utf-8") as f:
|
||
config = yaml.safe_load(f) or {}
|
||
query_days = int(config.get("yunda", {}).get("query_days", 1))
|
||
except Exception as e:
|
||
print(f" ⚠️ 读取 config.yaml 失败,默认查询 1 天: {e}")
|
||
|
||
today = datetime.now()
|
||
start_date = today - timedelta(days=(query_days - 1))
|
||
|
||
today_ymd = f"{today.year}-{today.month}-{today.day}"
|
||
start_date_ymd = f"{start_date.year}-{start_date.month}-{start_date.day}"
|
||
|
||
print(f">> 正在精准对焦时间区间: [{start_date_ymd}] 至 [{today_ymd}]")
|
||
|
||
# 设定起始时间
|
||
print(" >> 呼出起始时间控件...")
|
||
page.wait_for_timeout(1000)
|
||
ws_frame.locator("#startTime").click(force=True)
|
||
|
||
calendar1 = ws_frame.locator(".layui-laydate:visible").first
|
||
calendar1.wait_for(state="visible", timeout=5000)
|
||
calendar1.locator(f"td[lay-ymd='{start_date_ymd}']").click()
|
||
calendar1.locator(".laydate-btns-confirm").click()
|
||
page.wait_for_timeout(400)
|
||
|
||
# 设定截止时间
|
||
print(" >> 呼出截止时间控件...")
|
||
ws_frame.locator("#endTime").click(force=True)
|
||
|
||
calendar2 = ws_frame.locator(".layui-laydate:visible").first
|
||
calendar2.wait_for(state="visible", timeout=5000)
|
||
calendar2.locator(f"td[lay-ymd='{today_ymd}']").click()
|
||
calendar2.locator(".laydate-btns-confirm").click()
|
||
page.wait_for_timeout(500)
|
||
|
||
# ====================================================================
|
||
# 🛡️ 深度优化:三段式安全缓冲,彻底规避 DOM 旧状态穿透陷阱
|
||
# ====================================================================
|
||
print(">> 正在触发现场查询数据流...")
|
||
ws_frame.locator("a.btn-success", has_text="查询").click()
|
||
|
||
# 阶段1:物理安全挂起,赋予浏览器触发渲染周期和 Ajax 的生命力
|
||
page.wait_for_timeout(800)
|
||
|
||
# 阶段2:精准抓取 Bootstrap Table 专属的加载遮罩层
|
||
loading_mask = ws_frame.locator(
|
||
".fixed-table-loading", has_text="正在努力地加载数据中"
|
||
)
|
||
if loading_mask.is_visible():
|
||
print(" ⏳ 检测到专属数据加载罩,正在等待后端重载返回...")
|
||
loading_mask.wait_for(state="hidden", timeout=30000)
|
||
|
||
page.wait_for_timeout(500) # DOM 重排完毕后再缓冲半秒
|
||
|
||
# 阶段3:进入绝对真实的终态判定世界
|
||
sum_panel = ws_frame.locator("#sum")
|
||
has_data = False
|
||
|
||
if sum_panel.is_visible():
|
||
sum_text = sum_panel.inner_text()
|
||
match_tickets = re.search(r"进站实际票数:(\d+)", sum_text)
|
||
if match_tickets and int(match_tickets.group(1)) > 0:
|
||
has_data = True
|
||
print(
|
||
f" ✅ 深度断言:局部统计面板加载完毕,实际票数: [{match_tickets.group(1)}],执行穿透。"
|
||
)
|
||
|
||
if not has_data:
|
||
if ws_frame.locator(
|
||
".no-records-found", has_text="没有找到匹配的记录"
|
||
).is_visible():
|
||
print(
|
||
" ⚠️ 确认为空数据环境:系统提示【没有找到匹配的记录】,正在执行复位净化..."
|
||
)
|
||
page.locator(".tags-view-item", has_text="进站交接单查询").locator(
|
||
".el-icon-close"
|
||
).click()
|
||
return
|
||
|
||
# 4. 深度等待表格第一行数据行渲染就绪
|
||
ws_frame.locator("#exampleTable1 tbody tr[data-index='0']").wait_for(
|
||
state="visible", timeout=10000
|
||
)
|
||
|
||
main_rows = ws_frame.locator("#exampleTable1 tbody tr[data-index]")
|
||
row_count = main_rows.count()
|
||
print(f">> 当前视窗共捕获到活跃交接单记录: {row_count} 条")
|
||
|
||
# 5. 循环双击穿透提交
|
||
for i in range(row_count):
|
||
print(f" ⏳ 正在处理第 {i+1}/{row_count} 个交接单模块...")
|
||
current_row = ws_frame.locator("#exampleTable1 tbody tr[data-index]").nth(i)
|
||
|
||
raw_no = current_row.locator("td").nth(1).inner_text().strip()
|
||
print(f" -> 锁定提取单号: {raw_no}")
|
||
|
||
current_row.dblclick()
|
||
|
||
ws_frame.locator("#docSum").wait_for(state="visible", timeout=15000)
|
||
page.wait_for_timeout(500)
|
||
|
||
ws_frame.locator('a.btn-info[onclick*="exportFile"]').click()
|
||
|
||
ws_frame.locator(".layui-layer-title", has_text="数据导出").wait_for(
|
||
state="visible", timeout=15000
|
||
)
|
||
|
||
export_frame = ws_frame.frame_locator('iframe[name="target1"]')
|
||
export_frame.locator(".allRight").click()
|
||
page.wait_for_timeout(400)
|
||
|
||
# 智能重试容错状态机
|
||
print(" >> 正在建立后台离线任务...")
|
||
task_success = False
|
||
for attempt in range(5):
|
||
export_frame.locator("#submitbutton", has_text="导出数据").click()
|
||
confirm_link = export_frame.get_by_role("link", name="确定")
|
||
try:
|
||
confirm_link.wait_for(state="visible", timeout=6000)
|
||
if export_frame.get_by_text("导出任务建立成功").is_visible():
|
||
print(" ✅ 判定通过:成功捕获到【导出任务建立成功】特征!")
|
||
confirm_link.click()
|
||
task_success = True
|
||
break
|
||
elif export_frame.get_by_text(
|
||
"请选择格式相应的导出字段"
|
||
).is_visible():
|
||
print(
|
||
" ⚠️ 警告:检测到【未选择字段】错误,重新触发补点全选..."
|
||
)
|
||
confirm_link.click()
|
||
page.wait_for_timeout(500)
|
||
export_frame.locator(".allRight").click()
|
||
page.wait_for_timeout(500)
|
||
else:
|
||
confirm_link.click()
|
||
page.wait_for_timeout(1000)
|
||
except Exception:
|
||
page.wait_for_timeout(1000)
|
||
|
||
if not task_success:
|
||
raise RuntimeError(
|
||
"致命异常:连续 5 次尝试均无法成功建立应到数据离线任务。"
|
||
)
|
||
|
||
ws_frame.locator(".layui-layer-close1").click()
|
||
page.wait_for_timeout(500)
|
||
|
||
ws_frame.locator("#myTab a", has_text="交接单信息").click()
|
||
page.wait_for_timeout(800)
|
||
|
||
export_times.append(datetime.now())
|
||
|
||
# 6. 一阶段全量数据提交闭环,销毁当前业务 Tab
|
||
print(">> 📤 任务提交流闭环,正在执行【进站交接单查询】工作台销毁...")
|
||
page.locator(".tags-view-item", has_text="进站交接单查询").locator(
|
||
".el-icon-close"
|
||
).click()
|
||
page.wait_for_timeout(500)
|
||
|
||
# 7. 进入【导出服务】队列收割
|
||
_yunda_poll_and_download_tasks(
|
||
page,
|
||
export_times,
|
||
target_task_title,
|
||
download_dir,
|
||
"韵达-应到货物数据.xlsx",
|
||
)
|
||
|
||
except Exception as e:
|
||
print(f"\n❌ 任务执行过程中发生异常: {e}")
|
||
|
||
|
||
def yunda_actual_download(page):
|
||
"""韵达:实到货物数据下载"""
|
||
print("\n▶ 开始执行【韵达 - 实到货物数据下载】任务...")
|
||
target_task_title = "扫描记录数据"
|
||
|
||
download_dir = os.path.join(os.getcwd(), "downloads")
|
||
if not os.path.exists(download_dir):
|
||
os.makedirs(download_dir)
|
||
|
||
export_times = []
|
||
|
||
try:
|
||
page.locator(".el-menu-item", has_text="首页").wait_for(
|
||
state="visible", timeout=15000
|
||
)
|
||
print("✅ 韵达工作台首页成功加载")
|
||
|
||
yunda_smart_menu_click(page, ["报表管理", "扫描记录查询"])
|
||
|
||
print(">> [雷达扫描] 正在跨域动态追踪【扫描记录查询】业务窗体...")
|
||
ws_frame = _wait_and_get_frame_by_selector(page, "#startDate")
|
||
|
||
ws_frame.locator(".no-records-found", has_text="没有找到匹配的记录").wait_for(
|
||
state="visible", timeout=15000
|
||
)
|
||
print("✅ 扫描记录查询工作区初始化完毕")
|
||
|
||
query_days = 1
|
||
try:
|
||
if os.path.exists("config.yaml"):
|
||
with open("config.yaml", "r", encoding="utf-8") as f:
|
||
config = yaml.safe_load(f) or {}
|
||
query_days = int(config.get("yunda", {}).get("query_days", 1))
|
||
except Exception:
|
||
pass
|
||
|
||
today = datetime.now()
|
||
start_date = today - timedelta(days=(query_days - 1))
|
||
|
||
print(f">> 正在精准对焦实到时间区间: [近 {query_days} 天]")
|
||
|
||
print(" >> 正在设定起始时间...")
|
||
ws_frame.locator("#startDate").click()
|
||
page.wait_for_timeout(400)
|
||
box1 = ws_frame.locator("#laydate_box:visible").first
|
||
box1.locator(
|
||
f"td[y='{start_date.year}'][m='{start_date.month}'][d='{start_date.day}']"
|
||
).click()
|
||
page.wait_for_timeout(400)
|
||
|
||
print(" >> 正在设定截止时间...")
|
||
ws_frame.locator("#endDate").click()
|
||
page.wait_for_timeout(400)
|
||
box2 = ws_frame.locator("#laydate_box:visible").first
|
||
box2.locator(
|
||
f"td[y='{today.year}'][m='{today.month}'][d='{today.day}']"
|
||
).click()
|
||
page.wait_for_timeout(500)
|
||
|
||
print(" >> 正在变更扫描类型为【到件】...")
|
||
ws_frame.locator("#scanRecordTyp").select_option(value="03")
|
||
page.wait_for_timeout(500)
|
||
|
||
# ====================================================================
|
||
# 🛡️ 深度优化:实到数据的三段式安全缓冲判定
|
||
# ====================================================================
|
||
print(">> 正在触发现场查询数据流...")
|
||
ws_frame.locator('input[type="button"][value="查询"]').click()
|
||
|
||
page.wait_for_timeout(800)
|
||
|
||
loading_mask = ws_frame.locator(
|
||
".fixed-table-loading", has_text="正在努力地加载数据中"
|
||
)
|
||
if loading_mask.is_visible():
|
||
print(" ⏳ 检测到专属数据加载罩,正在等待后端重载返回...")
|
||
loading_mask.wait_for(state="hidden", timeout=30000)
|
||
|
||
page.wait_for_timeout(500)
|
||
|
||
pg_info = ws_frame.locator(".pagination-info")
|
||
has_records = False
|
||
|
||
if pg_info.is_visible():
|
||
info_text = pg_info.inner_text()
|
||
match_total = re.search(r"总共\s*(\d+)\s*条记录", info_text)
|
||
if match_total and int(match_total.group(1)) > 0:
|
||
has_records = True
|
||
print(
|
||
f" ✅ 深度断言:实到数据渲染完毕,总记录数: [{match_total.group(1)}] 条。"
|
||
)
|
||
|
||
if not has_records:
|
||
if ws_frame.locator(
|
||
".no-records-found", has_text="没有找到匹配的记录"
|
||
).is_visible():
|
||
print(
|
||
" ⚠️ 当前查询范围内确认为空数据:系统提示【没有找到匹配的记录】,终止并关闭环境。"
|
||
)
|
||
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
|
||
".el-icon-close"
|
||
).click()
|
||
return
|
||
else:
|
||
print(" ⚠️ 发生渲染异常,未找到数据也未找到空记录提示。安全收尾...")
|
||
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
|
||
".el-icon-close"
|
||
).click()
|
||
return
|
||
|
||
# 5. 执行导出流
|
||
print(">> 正在发起【导出】申请指令...")
|
||
ws_frame.locator('input[type="button"][id="export"]').click()
|
||
|
||
ws_frame.locator(".layui-layer-title", has_text="数据导出").wait_for(
|
||
state="visible", timeout=15000
|
||
)
|
||
export_frame = ws_frame.frame_locator('iframe[name="myFrame"]')
|
||
|
||
export_frame.locator(".allRight").click()
|
||
page.wait_for_timeout(400)
|
||
|
||
print(" >> 正在建立后台离线任务...")
|
||
task_success = False
|
||
for attempt in range(5):
|
||
export_frame.locator("#submitbutton", has_text="导出数据").click()
|
||
confirm_link = export_frame.get_by_role("link", name="确定")
|
||
try:
|
||
confirm_link.wait_for(state="visible", timeout=6000)
|
||
if export_frame.get_by_text("导出任务建立成功").is_visible():
|
||
print(" ✅ 判定通过:成功捕获到【导出任务建立成功】特征!")
|
||
confirm_link.click()
|
||
task_success = True
|
||
break
|
||
elif export_frame.get_by_text("请选择格式相应的导出字段").is_visible():
|
||
print(" ⚠️ 警告:检测到字段未全选,执行强补点击...")
|
||
confirm_link.click()
|
||
page.wait_for_timeout(500)
|
||
export_frame.locator(".allRight").click()
|
||
page.wait_for_timeout(500)
|
||
else:
|
||
confirm_link.click()
|
||
page.wait_for_timeout(1000)
|
||
except Exception:
|
||
page.wait_for_timeout(1000)
|
||
|
||
if not task_success:
|
||
raise RuntimeError(
|
||
"致命异常:连续 5 次尝试均无法成功建立实到数据离线任务。"
|
||
)
|
||
|
||
ws_frame.locator(".layui-layer-close1").click()
|
||
page.wait_for_timeout(500)
|
||
export_times.append(datetime.now())
|
||
|
||
print(">> 📤 任务提交流闭环,正在执行【扫描记录查询】工作台销毁...")
|
||
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
|
||
".el-icon-close"
|
||
).click()
|
||
page.wait_for_timeout(500)
|
||
|
||
# 6. 收割下载
|
||
_yunda_poll_and_download_tasks(
|
||
page,
|
||
export_times,
|
||
target_task_title,
|
||
download_dir,
|
||
"韵达-实到货物数据.xlsx",
|
||
)
|
||
|
||
except Exception as e:
|
||
print(f"\n❌ 任务执行过程中发生异常: {e}")
|
||
|
||
|
||
def _yunda_poll_and_download_tasks(
|
||
page, export_times, target_task_title, download_dir, final_filename
|
||
):
|
||
"""韵达专属离线任务轮询下载引擎"""
|
||
print("\n>> 正在前往【导出服务】中心...")
|
||
yunda_smart_menu_click(page, ["基础数据", "导出服务"])
|
||
|
||
export_ws_frame = page.frame_locator("section iframe")
|
||
export_ws_frame.get_by_role("cell", name="模块名称", exact=True).wait_for(
|
||
state="visible", timeout=15000
|
||
)
|
||
page.wait_for_timeout(1000)
|
||
|
||
print(">> 离线文件队列已对接,启动【1分钟高频精确校对+缺单局部重载刷新】断言...")
|
||
total_expected = len(export_times)
|
||
|
||
while True:
|
||
task_rows = export_ws_frame.locator(
|
||
".datagrid-view2 .datagrid-btable tbody tr.datagrid-row"
|
||
)
|
||
row_count = task_rows.count()
|
||
|
||
ready_indices = []
|
||
processing_indices = []
|
||
|
||
for idx in range(row_count):
|
||
row = task_rows.nth(idx)
|
||
module_name = row.locator("td[field='modueName']").inner_text().strip()
|
||
status_name = row.locator("td[field='fileStatus']").inner_text().strip()
|
||
create_time_str = (
|
||
row.locator("td[field='createdTime']").inner_text().strip()
|
||
)
|
||
|
||
if module_name == target_task_title:
|
||
try:
|
||
row_time = datetime.strptime(create_time_str, "%Y-%m-%d %H:%M:%S")
|
||
matched = any(
|
||
abs((row_time - et).total_seconds()) <= 60
|
||
for et in export_times
|
||
)
|
||
|
||
if matched:
|
||
if status_name == "导出完成":
|
||
ready_indices.append(idx)
|
||
else:
|
||
processing_indices.append(idx)
|
||
except Exception:
|
||
pass
|
||
|
||
total_found = len(ready_indices) + len(processing_indices)
|
||
print(
|
||
f" 📊 状态研判:自建期望数 [{total_expected}],实际入表 [{total_found}] (完成 [{len(ready_indices)}],生成中 [{len(processing_indices)}])"
|
||
)
|
||
|
||
if total_found < total_expected or len(processing_indices) > 0:
|
||
print(" ⏳ 队列未齐,执行【点击查询按钮】触发局部无痕刷新...")
|
||
export_ws_frame.locator(
|
||
"#ydkyimport_basic_export_searchData1_ky_export_common"
|
||
).click()
|
||
page.wait_for_timeout(3000)
|
||
else:
|
||
print(">> ✅ 所有目标离线任务全量就绪!开始依序接入文件流...")
|
||
break
|
||
|
||
# 下载成果流
|
||
downloaded_files = []
|
||
for row_idx in ready_indices:
|
||
try:
|
||
target_row = export_ws_frame.locator(
|
||
".datagrid-view2 .datagrid-btable tbody tr.datagrid-row"
|
||
).nth(row_idx)
|
||
time_flag = (
|
||
target_row.locator("td[field='createdTime']").inner_text().strip()
|
||
)
|
||
print(f" 🎯 触发下载 -> 离线任务时间节点: [{time_flag}] ...")
|
||
|
||
with page.expect_download() as download_info:
|
||
target_row.locator("td[field='extreFile'] a").get_by_text(
|
||
"下载"
|
||
).first.click()
|
||
|
||
download = download_info.value
|
||
|
||
safe_timestamp = datetime.now().strftime("%Y%m%d_%H%M%S_%f")
|
||
custom_filename = f"韵达_temp_{safe_timestamp}.xlsx"
|
||
save_path = os.path.join(download_dir, custom_filename)
|
||
|
||
download.save_as(save_path)
|
||
downloaded_files.append(save_path)
|
||
print(f" ⬇️ 文件已用安全序列号落盘: downloads/{custom_filename}")
|
||
page.wait_for_timeout(500)
|
||
except Exception as e:
|
||
print(f" ❌ 文件流接收失败: {e}")
|
||
|
||
# 环境清理
|
||
print(">> 📥 【导出服务】数据提取链闭环,正在执行当前 Tab 窗口销毁...")
|
||
try:
|
||
page.locator(".tags-view-item", has_text="导出服务").locator(
|
||
".el-icon-close"
|
||
).click()
|
||
print(" ✅ 【导出服务】工作区已安全关闭。")
|
||
except:
|
||
pass
|
||
|
||
# 合并扁平数据集
|
||
if downloaded_files:
|
||
print("\n>> 🧪 正在启动离线数据清洗与高能扁平合并流...")
|
||
all_dfs = []
|
||
for file_path in downloaded_files:
|
||
try:
|
||
df = pd.read_excel(file_path, dtype=str)
|
||
if not df.empty:
|
||
all_dfs.append(df)
|
||
except:
|
||
pass
|
||
|
||
if all_dfs:
|
||
combined_df = pd.concat(all_dfs, ignore_index=True)
|
||
final_output = os.path.join(download_dir, final_filename)
|
||
combined_df.to_excel(final_output, index=False)
|
||
print(f"====================================================")
|
||
print(f" 🎉 恭喜!韵达网点数据拉取合流大功告成!")
|
||
print(f" 📁 最终存储位置: {final_output}")
|
||
print(f"====================================================")
|
||
|
||
for file_path in downloaded_files:
|
||
os.remove(file_path)
|
||
print(" ✅ 临时缓存阵列已无缝净化。")
|