Files
InboundVerify/site_yunda.py
Misaka 604d1822e4 Implement Yunda actual-arrival download and harden query loading
- Implement yunda_actual_download via the 报表管理 -> 扫描记录查询
  menu: set date range, switch scan type to 到件, then reuse the
  shared poll/download/merge engine
- Add a three-stage query buffer (yield, wait for the Bootstrap Table
  loading mask, settle) before the data/empty verdict to avoid stale
  DOM reads in the expected-arrival flow
- Trim the offline-export retry state machine's verbose logs
- Fix a typo (研编 -> 研判) and drop the 待开发 tag on menu [7]

Co-Authored-By: Claude <noreply@anthropic.com>
2026-06-20 18:59:49 +08:00

568 lines
22 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# site_yunda.py
import os
import re
import yaml
from datetime import datetime, timedelta
import pandas as pd
def _wait_and_get_frame(page, text_indicator, timeout_ms=20000):
"""【文本雷达探测器】全域跨框架检索包含特定文本的活动上下文"""
start_time = datetime.now()
while (datetime.now() - start_time).total_seconds() * 1000 < timeout_ms:
try:
if page.get_by_text(text_indicator).count() > 0:
return page
except:
pass
for frame in page.frames:
try:
if frame.get_by_text(text_indicator).count() > 0:
return frame
except:
pass
page.wait_for_timeout(300)
raise TimeoutError(
f"爆栈超时:未能在活动框架中死守到包含 [{text_indicator}] 的视窗。"
)
def _wait_and_get_frame_by_selector(page, selector, timeout_ms=20000):
"""【控件雷达探测器】无视 IFrame 的 src 或层级,直接扫描谁包含指定 CSS 控件"""
start_time = datetime.now()
while (datetime.now() - start_time).total_seconds() * 1000 < timeout_ms:
try:
if page.locator(selector).count() > 0:
return page
except:
pass
for frame in page.frames:
try:
if frame.locator(selector).count() > 0:
return frame
except:
pass
page.wait_for_timeout(300)
raise TimeoutError(f"爆栈超时:全域未探测到包含控件 [{selector}] 的业务框架。")
def yunda_smart_menu_click(page, menu_path):
"""韵达排他性二级树形菜单智能展开器"""
print(f">> 正在导航韵达菜单: {' -> '.join(menu_path)}")
for item in menu_path:
locator = page.locator(".el-submenu__title, .el-menu-item", has_text=item).first
locator.click()
page.wait_for_timeout(600)
def yunda_expected_download(page):
"""韵达:应到货物数据下载"""
print("\n▶ 开始执行【韵达 - 应到货物数据下载】任务...")
target_task_title = "进站主单表"
download_dir = os.path.join(os.getcwd(), "downloads")
if not os.path.exists(download_dir):
os.makedirs(download_dir)
export_times = []
try:
# 1. 验证首页并智能路由
page.locator(".el-menu-item", has_text="首页").wait_for(
state="visible", timeout=15000
)
print("✅ 韵达工作台首页成功加载")
yunda_smart_menu_click(page, ["运营管理", "进站管理", "进站交接单查询"])
print(">> [雷达扫描] 正在跨域动态追踪【进站交接单查询】业务窗体...")
ws_frame = _wait_and_get_frame_by_selector(page, "#startTime")
ws_frame.locator("#startTime").wait_for(state="attached", timeout=15000)
print("✅ 进站交接单查询工作区就绪")
# 2. 从配置文件中解析并计算绝对日期跨度
query_days = 1
try:
if os.path.exists("config.yaml"):
with open("config.yaml", "r", encoding="utf-8") as f:
config = yaml.safe_load(f) or {}
query_days = int(config.get("yunda", {}).get("query_days", 1))
except Exception as e:
print(f" ⚠️ 读取 config.yaml 失败,默认查询 1 天: {e}")
today = datetime.now()
start_date = today - timedelta(days=(query_days - 1))
today_ymd = f"{today.year}-{today.month}-{today.day}"
start_date_ymd = f"{start_date.year}-{start_date.month}-{start_date.day}"
print(f">> 正在精准对焦时间区间: [{start_date_ymd}] 至 [{today_ymd}]")
# 设定起始时间
print(" >> 呼出起始时间控件...")
page.wait_for_timeout(1000)
ws_frame.locator("#startTime").click(force=True)
calendar1 = ws_frame.locator(".layui-laydate:visible").first
calendar1.wait_for(state="visible", timeout=5000)
calendar1.locator(f"td[lay-ymd='{start_date_ymd}']").click()
calendar1.locator(".laydate-btns-confirm").click()
page.wait_for_timeout(400)
# 设定截止时间
print(" >> 呼出截止时间控件...")
ws_frame.locator("#endTime").click(force=True)
calendar2 = ws_frame.locator(".layui-laydate:visible").first
calendar2.wait_for(state="visible", timeout=5000)
calendar2.locator(f"td[lay-ymd='{today_ymd}']").click()
calendar2.locator(".laydate-btns-confirm").click()
page.wait_for_timeout(500)
# ====================================================================
# 🛡️ 深度优化:三段式安全缓冲,彻底规避 DOM 旧状态穿透陷阱
# ====================================================================
print(">> 正在触发现场查询数据流...")
ws_frame.locator("a.btn-success", has_text="查询").click()
# 阶段1物理安全挂起赋予浏览器触发渲染周期和 Ajax 的生命力
page.wait_for_timeout(800)
# 阶段2精准抓取 Bootstrap Table 专属的加载遮罩层
loading_mask = ws_frame.locator(
".fixed-table-loading", has_text="正在努力地加载数据中"
)
if loading_mask.is_visible():
print(" ⏳ 检测到专属数据加载罩,正在等待后端重载返回...")
loading_mask.wait_for(state="hidden", timeout=30000)
page.wait_for_timeout(500) # DOM 重排完毕后再缓冲半秒
# 阶段3进入绝对真实的终态判定世界
sum_panel = ws_frame.locator("#sum")
has_data = False
if sum_panel.is_visible():
sum_text = sum_panel.inner_text()
match_tickets = re.search(r"进站实际票数:(\d+)", sum_text)
if match_tickets and int(match_tickets.group(1)) > 0:
has_data = True
print(
f" ✅ 深度断言:局部统计面板加载完毕,实际票数: [{match_tickets.group(1)}],执行穿透。"
)
if not has_data:
if ws_frame.locator(
".no-records-found", has_text="没有找到匹配的记录"
).is_visible():
print(
" ⚠️ 确认为空数据环境:系统提示【没有找到匹配的记录】,正在执行复位净化..."
)
page.locator(".tags-view-item", has_text="进站交接单查询").locator(
".el-icon-close"
).click()
return
# 4. 深度等待表格第一行数据行渲染就绪
ws_frame.locator("#exampleTable1 tbody tr[data-index='0']").wait_for(
state="visible", timeout=10000
)
main_rows = ws_frame.locator("#exampleTable1 tbody tr[data-index]")
row_count = main_rows.count()
print(f">> 当前视窗共捕获到活跃交接单记录: {row_count}")
# 5. 循环双击穿透提交
for i in range(row_count):
print(f" ⏳ 正在处理第 {i+1}/{row_count} 个交接单模块...")
current_row = ws_frame.locator("#exampleTable1 tbody tr[data-index]").nth(i)
raw_no = current_row.locator("td").nth(1).inner_text().strip()
print(f" -> 锁定提取单号: {raw_no}")
current_row.dblclick()
ws_frame.locator("#docSum").wait_for(state="visible", timeout=15000)
page.wait_for_timeout(500)
ws_frame.locator('a.btn-info[onclick*="exportFile"]').click()
ws_frame.locator(".layui-layer-title", has_text="数据导出").wait_for(
state="visible", timeout=15000
)
export_frame = ws_frame.frame_locator('iframe[name="target1"]')
export_frame.locator(".allRight").click()
page.wait_for_timeout(400)
# 智能重试容错状态机
print(" >> 正在建立后台离线任务...")
task_success = False
for attempt in range(5):
export_frame.locator("#submitbutton", has_text="导出数据").click()
confirm_link = export_frame.get_by_role("link", name="确定")
try:
confirm_link.wait_for(state="visible", timeout=6000)
if export_frame.get_by_text("导出任务建立成功").is_visible():
print(" ✅ 判定通过:成功捕获到【导出任务建立成功】特征!")
confirm_link.click()
task_success = True
break
elif export_frame.get_by_text(
"请选择格式相应的导出字段"
).is_visible():
print(
" ⚠️ 警告:检测到【未选择字段】错误,重新触发补点全选..."
)
confirm_link.click()
page.wait_for_timeout(500)
export_frame.locator(".allRight").click()
page.wait_for_timeout(500)
else:
confirm_link.click()
page.wait_for_timeout(1000)
except Exception:
page.wait_for_timeout(1000)
if not task_success:
raise RuntimeError(
"致命异常:连续 5 次尝试均无法成功建立应到数据离线任务。"
)
ws_frame.locator(".layui-layer-close1").click()
page.wait_for_timeout(500)
ws_frame.locator("#myTab a", has_text="交接单信息").click()
page.wait_for_timeout(800)
export_times.append(datetime.now())
# 6. 一阶段全量数据提交闭环,销毁当前业务 Tab
print(">> 📤 任务提交流闭环,正在执行【进站交接单查询】工作台销毁...")
page.locator(".tags-view-item", has_text="进站交接单查询").locator(
".el-icon-close"
).click()
page.wait_for_timeout(500)
# 7. 进入【导出服务】队列收割
_yunda_poll_and_download_tasks(
page,
export_times,
target_task_title,
download_dir,
"韵达-应到货物数据.xlsx",
)
except Exception as e:
print(f"\n❌ 任务执行过程中发生异常: {e}")
def yunda_actual_download(page):
"""韵达:实到货物数据下载"""
print("\n▶ 开始执行【韵达 - 实到货物数据下载】任务...")
target_task_title = "扫描记录数据"
download_dir = os.path.join(os.getcwd(), "downloads")
if not os.path.exists(download_dir):
os.makedirs(download_dir)
export_times = []
try:
page.locator(".el-menu-item", has_text="首页").wait_for(
state="visible", timeout=15000
)
print("✅ 韵达工作台首页成功加载")
yunda_smart_menu_click(page, ["报表管理", "扫描记录查询"])
print(">> [雷达扫描] 正在跨域动态追踪【扫描记录查询】业务窗体...")
ws_frame = _wait_and_get_frame_by_selector(page, "#startDate")
ws_frame.locator(".no-records-found", has_text="没有找到匹配的记录").wait_for(
state="visible", timeout=15000
)
print("✅ 扫描记录查询工作区初始化完毕")
query_days = 1
try:
if os.path.exists("config.yaml"):
with open("config.yaml", "r", encoding="utf-8") as f:
config = yaml.safe_load(f) or {}
query_days = int(config.get("yunda", {}).get("query_days", 1))
except Exception:
pass
today = datetime.now()
start_date = today - timedelta(days=(query_days - 1))
print(f">> 正在精准对焦实到时间区间: [近 {query_days} 天]")
print(" >> 正在设定起始时间...")
ws_frame.locator("#startDate").click()
page.wait_for_timeout(400)
box1 = ws_frame.locator("#laydate_box:visible").first
box1.locator(
f"td[y='{start_date.year}'][m='{start_date.month}'][d='{start_date.day}']"
).click()
page.wait_for_timeout(400)
print(" >> 正在设定截止时间...")
ws_frame.locator("#endDate").click()
page.wait_for_timeout(400)
box2 = ws_frame.locator("#laydate_box:visible").first
box2.locator(
f"td[y='{today.year}'][m='{today.month}'][d='{today.day}']"
).click()
page.wait_for_timeout(500)
print(" >> 正在变更扫描类型为【到件】...")
ws_frame.locator("#scanRecordTyp").select_option(value="03")
page.wait_for_timeout(500)
# ====================================================================
# 🛡️ 深度优化:实到数据的三段式安全缓冲判定
# ====================================================================
print(">> 正在触发现场查询数据流...")
ws_frame.locator('input[type="button"][value="查询"]').click()
page.wait_for_timeout(800)
loading_mask = ws_frame.locator(
".fixed-table-loading", has_text="正在努力地加载数据中"
)
if loading_mask.is_visible():
print(" ⏳ 检测到专属数据加载罩,正在等待后端重载返回...")
loading_mask.wait_for(state="hidden", timeout=30000)
page.wait_for_timeout(500)
pg_info = ws_frame.locator(".pagination-info")
has_records = False
if pg_info.is_visible():
info_text = pg_info.inner_text()
match_total = re.search(r"总共\s*(\d+)\s*条记录", info_text)
if match_total and int(match_total.group(1)) > 0:
has_records = True
print(
f" ✅ 深度断言:实到数据渲染完毕,总记录数: [{match_total.group(1)}] 条。"
)
if not has_records:
if ws_frame.locator(
".no-records-found", has_text="没有找到匹配的记录"
).is_visible():
print(
" ⚠️ 当前查询范围内确认为空数据:系统提示【没有找到匹配的记录】,终止并关闭环境。"
)
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
".el-icon-close"
).click()
return
else:
print(" ⚠️ 发生渲染异常,未找到数据也未找到空记录提示。安全收尾...")
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
".el-icon-close"
).click()
return
# 5. 执行导出流
print(">> 正在发起【导出】申请指令...")
ws_frame.locator('input[type="button"][id="export"]').click()
ws_frame.locator(".layui-layer-title", has_text="数据导出").wait_for(
state="visible", timeout=15000
)
export_frame = ws_frame.frame_locator('iframe[name="myFrame"]')
export_frame.locator(".allRight").click()
page.wait_for_timeout(400)
print(" >> 正在建立后台离线任务...")
task_success = False
for attempt in range(5):
export_frame.locator("#submitbutton", has_text="导出数据").click()
confirm_link = export_frame.get_by_role("link", name="确定")
try:
confirm_link.wait_for(state="visible", timeout=6000)
if export_frame.get_by_text("导出任务建立成功").is_visible():
print(" ✅ 判定通过:成功捕获到【导出任务建立成功】特征!")
confirm_link.click()
task_success = True
break
elif export_frame.get_by_text("请选择格式相应的导出字段").is_visible():
print(" ⚠️ 警告:检测到字段未全选,执行强补点击...")
confirm_link.click()
page.wait_for_timeout(500)
export_frame.locator(".allRight").click()
page.wait_for_timeout(500)
else:
confirm_link.click()
page.wait_for_timeout(1000)
except Exception:
page.wait_for_timeout(1000)
if not task_success:
raise RuntimeError(
"致命异常:连续 5 次尝试均无法成功建立实到数据离线任务。"
)
ws_frame.locator(".layui-layer-close1").click()
page.wait_for_timeout(500)
export_times.append(datetime.now())
print(">> 📤 任务提交流闭环,正在执行【扫描记录查询】工作台销毁...")
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
".el-icon-close"
).click()
page.wait_for_timeout(500)
# 6. 收割下载
_yunda_poll_and_download_tasks(
page,
export_times,
target_task_title,
download_dir,
"韵达-实到货物数据.xlsx",
)
except Exception as e:
print(f"\n❌ 任务执行过程中发生异常: {e}")
def _yunda_poll_and_download_tasks(
page, export_times, target_task_title, download_dir, final_filename
):
"""韵达专属离线任务轮询下载引擎"""
print("\n>> 正在前往【导出服务】中心...")
yunda_smart_menu_click(page, ["基础数据", "导出服务"])
export_ws_frame = page.frame_locator("section iframe")
export_ws_frame.get_by_role("cell", name="模块名称", exact=True).wait_for(
state="visible", timeout=15000
)
page.wait_for_timeout(1000)
print(">> 离线文件队列已对接启动【1分钟高频精确校对+缺单局部重载刷新】断言...")
total_expected = len(export_times)
while True:
task_rows = export_ws_frame.locator(
".datagrid-view2 .datagrid-btable tbody tr.datagrid-row"
)
row_count = task_rows.count()
ready_indices = []
processing_indices = []
for idx in range(row_count):
row = task_rows.nth(idx)
module_name = row.locator("td[field='modueName']").inner_text().strip()
status_name = row.locator("td[field='fileStatus']").inner_text().strip()
create_time_str = (
row.locator("td[field='createdTime']").inner_text().strip()
)
if module_name == target_task_title:
try:
row_time = datetime.strptime(create_time_str, "%Y-%m-%d %H:%M:%S")
matched = any(
abs((row_time - et).total_seconds()) <= 60
for et in export_times
)
if matched:
if status_name == "导出完成":
ready_indices.append(idx)
else:
processing_indices.append(idx)
except Exception:
pass
total_found = len(ready_indices) + len(processing_indices)
print(
f" 📊 状态研判:自建期望数 [{total_expected}],实际入表 [{total_found}] (完成 [{len(ready_indices)}],生成中 [{len(processing_indices)}])"
)
if total_found < total_expected or len(processing_indices) > 0:
print(" ⏳ 队列未齐,执行【点击查询按钮】触发局部无痕刷新...")
export_ws_frame.locator(
"#ydkyimport_basic_export_searchData1_ky_export_common"
).click()
page.wait_for_timeout(3000)
else:
print(">> ✅ 所有目标离线任务全量就绪!开始依序接入文件流...")
break
# 下载成果流
downloaded_files = []
for row_idx in ready_indices:
try:
target_row = export_ws_frame.locator(
".datagrid-view2 .datagrid-btable tbody tr.datagrid-row"
).nth(row_idx)
time_flag = (
target_row.locator("td[field='createdTime']").inner_text().strip()
)
print(f" 🎯 触发下载 -> 离线任务时间节点: [{time_flag}] ...")
with page.expect_download() as download_info:
target_row.locator("td[field='extreFile'] a").get_by_text(
"下载"
).first.click()
download = download_info.value
safe_timestamp = datetime.now().strftime("%Y%m%d_%H%M%S_%f")
custom_filename = f"韵达_temp_{safe_timestamp}.xlsx"
save_path = os.path.join(download_dir, custom_filename)
download.save_as(save_path)
downloaded_files.append(save_path)
print(f" ⬇️ 文件已用安全序列号落盘: downloads/{custom_filename}")
page.wait_for_timeout(500)
except Exception as e:
print(f" ❌ 文件流接收失败: {e}")
# 环境清理
print(">> 📥 【导出服务】数据提取链闭环,正在执行当前 Tab 窗口销毁...")
try:
page.locator(".tags-view-item", has_text="导出服务").locator(
".el-icon-close"
).click()
print(" ✅ 【导出服务】工作区已安全关闭。")
except:
pass
# 合并扁平数据集
if downloaded_files:
print("\n>> 🧪 正在启动离线数据清洗与高能扁平合并流...")
all_dfs = []
for file_path in downloaded_files:
try:
df = pd.read_excel(file_path, dtype=str)
if not df.empty:
all_dfs.append(df)
except:
pass
if all_dfs:
combined_df = pd.concat(all_dfs, ignore_index=True)
final_output = os.path.join(download_dir, final_filename)
combined_df.to_excel(final_output, index=False)
print(f"====================================================")
print(f" 🎉 恭喜!韵达网点数据拉取合流大功告成!")
print(f" 📁 最终存储位置: {final_output}")
print(f"====================================================")
for file_path in downloaded_files:
os.remove(file_path)
print(" ✅ 临时缓存阵列已无缝净化。")