Files
InboundVerify/site_zto.py
Misaka 358b8327b9 Implement ZTO actual-arrival download and extract shared poll engine
- Implement zto_actual_download via the 到件扫描监控 menu with 子单
  number-type filter, reusing the common download flow
- Extract the export-task poll/verify/download/merge steps into
  _zto_poll_and_download_tasks so both ZTO flows share it
- Tighten task time-match window from 120s to 60s and add a
  refresh-and-recheck loop when expected tasks are missing or pending
- Drop unused _ensure_menu_expanded helper
- Update menu [6] label since actual-arrival is now implemented

Co-Authored-By: Claude <noreply@anthropic.com>
2026-06-20 13:48:28 +08:00

486 lines
19 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# site_zto.py
import os
import re
import yaml
from datetime import datetime
import pandas as pd
def _wait_and_get_frame(page, text_indicator, timeout_ms=20000):
"""动态雷达探测器:全域扫描所有视窗"""
start_time = datetime.now()
while (datetime.now() - start_time).total_seconds() * 1000 < timeout_ms:
try:
if page.get_by_text(text_indicator).count() > 0:
return page
except:
pass
for frame in page.frames:
try:
if frame.get_by_text(text_indicator).count() > 0:
return frame
except:
pass
page.wait_for_timeout(300)
raise TimeoutError(f"爆栈超时:全域未死守到包含 [{text_indicator}] 的弹窗视窗。")
def zto_smart_menu_click(page, menu_path):
"""中通智能菜单导航器"""
print(f">> 正在导航: {' -> '.join(menu_path)}")
for i in range(len(menu_path)):
current_menu = menu_path[i]
if i < len(menu_path) - 1:
next_menu = menu_path[i + 1]
next_locator = page.locator("span.menu-name", has_text=next_menu).first
if not next_locator.is_visible():
page.locator("span.menu-name", has_text=current_menu).first.click()
page.wait_for_timeout(800)
else:
page.locator("span.menu-name", has_text=current_menu).first.click()
page.wait_for_timeout(1000)
def zto_expected_download(page):
"""中通:应到货物数据下载"""
print("\n▶ 开始执行【中通 - 应到货物数据下载】任务...")
target_task_title = "进站交接单查询-运单信息"
download_dir = os.path.join(os.getcwd(), "downloads")
if not os.path.exists(download_dir):
os.makedirs(download_dir)
export_times = []
try:
# 1. 智能导航
zto_smart_menu_click(page, ["运营管理", "进站管理", "进站交接单查询"])
ewb_frame = page.frame_locator('iframe[src*="inEwbsListNoSearch"]')
print(">> 正在探针检测表格统计面板 (#inEwbCount)...")
ewb_frame.locator("#inEwbCount").wait_for(state="attached", timeout=15000)
# 读取 YAML 配置设定天数
query_days = 1
try:
if os.path.exists("config.yaml"):
with open("config.yaml", "r", encoding="utf-8") as f:
config = yaml.safe_load(f) or {}
query_days = int(config.get("zto", {}).get("query_days", 1))
except Exception as e:
print(f" ⚠️ 读取 config.yaml 失败,默认查询 1 天: {e}")
print(f">> 正在设定查询时间范围为近【{query_days}】天...")
ewb_frame.locator("#beginDate").click()
page.wait_for_timeout(500)
today_cell = ewb_frame.locator("td div.day.real-today").first
today_cell.wait_for(state="visible")
today_time_str = today_cell.get_attribute("time")
if today_time_str:
today_time = int(today_time_str)
start_time = today_time - (query_days - 1) * 86400000
start_cell = ewb_frame.locator(f"td div.day[time='{start_time}']").first
if start_cell.is_visible():
start_cell.click()
page.wait_for_timeout(300)
today_cell.click()
else:
print(" ⚠️ 设定的天数过大,不在当前日历视窗内,自动降级为查询当天。")
today_cell.click()
page.wait_for_timeout(300)
today_cell.click()
else:
today_cell.click()
page.wait_for_timeout(300)
today_cell.click()
page.wait_for_timeout(500)
# 3. 触发查询与深度状态机判定
print(">> 正在点击【查询】按钮并等待数据响应...")
ewb_frame.locator("#searchbtn").click()
page.wait_for_timeout(1000)
loading_mask = ewb_frame.locator(".mini-mask-loading", has_text="加载中")
if loading_mask.is_visible():
print(" ⏳ 检测到数据加载遮罩层,等待系统渲染...")
loading_mask.wait_for(state="hidden", timeout=30000)
page.wait_for_timeout(1000)
empty_flag = ewb_frame.locator("span", has_text="没有搜索到符合条件的数据记录")
if empty_flag.is_visible():
print(" ⚠️ 当前查询范围内【没有搜索到符合条件的数据记录】,终止导出。")
return
ticket_count_str = ewb_frame.locator("#inEwbCount").inner_text().strip()
print(f" ✅ 数据渲染完毕!当前进站实际票数为: [{ticket_count_str}]")
# 4. 提取主表记录并循环双击
main_rows = ewb_frame.locator("#datagrid1 .mini-grid-rows-view .mini-grid-row")
count = main_rows.count()
print(f">> 共发现 {count} 个交接单需要导出。")
for i in range(count):
print(f" ⏳ 正在处理第 {i+1}/{count} 个交接单...")
row = ewb_frame.locator(
"#datagrid1 .mini-grid-rows-view .mini-grid-row"
).nth(i)
# 正则表达式清洗交接单号
raw_text = row.locator("td").nth(3).inner_text()
match = re.search(r"\d{18}", raw_text)
handover_no = match.group(0) if match else raw_text.strip()
print(f" -> 锁定提取单号:{handover_no}")
row.dblclick()
ewb_frame.locator("#datagrid2").get_by_text("运单号").wait_for(
state="visible"
)
ewb_frame.locator(
"#datagrid2 .mini-grid-row", has_text=handover_no
).first.wait_for(state="visible")
# 5. 执行导出流程
ewb_frame.locator("#exportExcel").click()
page.locator(".mini-panel-title", has_text="导出选择列").wait_for(
state="visible"
)
export_frame = page.frame_locator('iframe[src*="download"]')
export_frame.locator(".mini-button-text", has_text=">>").click()
page.wait_for_timeout(300)
export_frame.locator(".mini-button-text", has_text="确定").click()
print(" >> [雷达扫描] 正在跨域动态追踪【温馨提示】所在的隐身视窗...")
ctx_alert = _wait_and_get_frame(page, "温馨提示")
ctx_alert.locator(
".mini-messagebox-buttons .mini-button-text", has_text="确定"
).click()
print(" ✅ 【温馨提示】已成功通过父级过滤并确认!")
print(" >> 正在等待服务器建立后台离线任务...")
try:
ctx_tips = _wait_and_get_frame(
page, "生成离线导出任务成功", timeout_ms=10000
)
ctx_tips.locator(".mini-tips-success").wait_for(
state="hidden", timeout=15000
)
print(" ✅ 成功提示框已平滑隐藏。")
except Exception:
print(" ✅ 成功提示框及其容器已自动销毁 (触发 Detached 拦截)。")
export_times.append(datetime.now())
print(" >> 切换回【交接单信息】标签页...")
ewb_frame.locator("#ewbsListNo").click()
page.wait_for_timeout(1000)
# 核心逻辑交由统一的轮询下载引擎处理
_zto_poll_and_download_tasks(
page,
export_times,
target_task_title,
download_dir,
"中通-应到货物数据.xlsx",
)
except Exception as e:
print(f"\n❌ 任务执行过程中发生异常: {e}")
def zto_actual_download(page):
"""中通:实到货物数据下载"""
print("\n▶ 开始执行【中通 - 实到货物数据下载】任务...")
target_task_title = "到件扫描管理"
download_dir = os.path.join(os.getcwd(), "downloads")
if not os.path.exists(download_dir):
os.makedirs(download_dir)
export_times = []
try:
# 1. 智能导航
zto_smart_menu_click(page, ["运营管理", "扫描操作与监控", "到件扫描监控"])
arr_frame = page.frame_locator('iframe[src*="ArriveScan"]')
print(">> 正在探针检测主页面 (#daterange)...")
arr_frame.locator("#daterange").wait_for(state="attached", timeout=15000)
# 2. 读取 YAML 并设定时间范围
query_days = 1
try:
if os.path.exists("config.yaml"):
with open("config.yaml", "r", encoding="utf-8") as f:
config = yaml.safe_load(f) or {}
query_days = int(config.get("zto", {}).get("query_days", 1))
except Exception:
pass
print(f">> 正在设定查询时间范围为近【{query_days}】天...")
arr_frame.locator("#daterange").click()
page.wait_for_timeout(500)
today_cell = arr_frame.locator("td div.day.real-today").first
today_cell.wait_for(state="visible")
today_time_str = today_cell.get_attribute("time")
if today_time_str:
today_time = int(today_time_str)
start_time = today_time - (query_days - 1) * 86400000
start_cell = arr_frame.locator(f"td div.day[time='{start_time}']").first
if start_cell.is_visible():
start_cell.click()
page.wait_for_timeout(300)
today_cell.click()
else:
today_cell.click()
page.wait_for_timeout(300)
today_cell.click()
else:
today_cell.click()
page.wait_for_timeout(300)
today_cell.click()
page.wait_for_timeout(500)
# 3. 设定单号类型(树形下拉框选择子单)
print(">> 正在设定单号类型为【子单】...")
arr_frame.locator('[id="bandEwbType$text"]').click()
page.wait_for_timeout(500)
arr_frame.locator(".mini-tree-nodeshow").filter(
has_text=re.compile(r"^子单$")
).locator(".mini-tree-checkbox").click()
page.wait_for_timeout(300)
# 4. 触发查询与深度状态机判定
print(">> 正在点击【查询】按钮并等待数据响应...")
arr_frame.locator("#searchbtn").click()
page.wait_for_timeout(1000)
loading_mask = arr_frame.locator(".mini-mask-loading", has_text="加载中")
if loading_mask.is_visible():
print(" ⏳ 检测到数据加载遮罩层,等待系统渲染...")
loading_mask.wait_for(state="hidden", timeout=30000)
page.wait_for_timeout(1000)
empty_flag = arr_frame.locator("span", has_text="没有搜索到符合条件的数据记录")
if empty_flag.is_visible():
print(" ⚠️ 当前查询范围内【没有搜索到符合条件的数据记录】,终止导出。")
return
arr_frame.locator("#page1").wait_for(state="visible", timeout=15000)
print(" ✅ 数据渲染完毕!(底部分页统计控件已就绪)")
# 5. 执行导出流程
arr_frame.locator("#exportExcel").click()
page.locator(".mini-panel-title", has_text="导出选择列").wait_for(
state="visible"
)
export_frame = page.frame_locator('iframe[src*="download"]')
export_frame.locator(".mini-button-text", has_text=">>").click()
page.wait_for_timeout(300)
export_frame.locator(".mini-button-text", has_text="确定").click()
print(" >> [雷达扫描] 正在跨域动态追踪【温馨提示】所在的隐身视窗...")
ctx_alert = _wait_and_get_frame(page, "温馨提示")
ctx_alert.locator(
".mini-messagebox-buttons .mini-button-text", has_text="确定"
).click()
print(" ✅ 【温馨提示】已成功通过父级过滤并确认!")
print(" >> 正在等待服务器建立后台离线任务...")
try:
ctx_tips = _wait_and_get_frame(
page, "生成离线导出任务成功", timeout_ms=10000
)
ctx_tips.locator(".mini-tips-success").wait_for(
state="hidden", timeout=15000
)
print(" ✅ 成功提示框已平滑隐藏。")
except Exception:
print(" ✅ 成功提示框及其容器已自动销毁 (触发 Detached 拦截)。")
export_times.append(datetime.now())
# 核心逻辑交由统一的轮询下载引擎处理
_zto_poll_and_download_tasks(
page,
export_times,
target_task_title,
download_dir,
"中通-实到货物数据.xlsx",
)
except Exception as e:
print(f"\n❌ 任务执行过程中发生异常: {e}")
# ====================================================================
# 🛡️ 优化核心:多网点通用任务匹配、界面重载刷新、并行状态机引擎
# ====================================================================
def _zto_poll_and_download_tasks(
page, export_times, target_task_title, download_dir, final_filename
):
"""进站中通任务专属轮询、验证及即选即隐下载引擎"""
print("\n>> 正在前往【导出任务管理】界面...")
zto_smart_menu_click(page, ["系统配置", "导出任务管理"])
taskdone_frame = page.frame_locator('iframe[src*="taskdone"]')
taskdone_frame.locator("#taskdoneDatagrid").get_by_text("任务标题").wait_for(
state="visible"
)
page.wait_for_timeout(1000)
print(
">> 列表中已渲染开始执行【1分钟误差+标题判定+缺单自动点击查询刷新】高级断言..."
)
total_expected = len(export_times)
while True:
task_rows = taskdone_frame.locator(
"#taskdoneDatagrid .mini-grid-rows-view .mini-grid-row"
)
row_count = task_rows.count()
ready_timestamps = set()
processing_timestamps = set()
for i in range(row_count):
tds = task_rows.nth(i).locator("td")
if tds.count() < 10:
continue
title_str = tds.nth(3).inner_text().strip()
submit_time_str = tds.nth(4).inner_text().strip()
status_str = tds.nth(6).inner_text().strip()
# 严格核对任务标题
if title_str == target_task_title:
try:
row_time = datetime.strptime(submit_time_str, "%Y-%m-%d %H:%M:%S")
# 【核心优化】:核实当前行是否落在任何一个发起的任务的 1 分钟之内 (<= 60秒)
matched = any(
abs((row_time - et).total_seconds()) <= 60
for et in export_times
)
if matched:
if status_str == "成功执行":
ready_timestamps.add(submit_time_str)
else:
processing_timestamps.add(submit_time_str)
except Exception:
pass
# 状态研判
total_found = len(ready_timestamps) + len(processing_timestamps)
print(
f" 📊 状态盘点:期望任务数 [{total_expected}],实际入表 [{total_found}] (就绪 [{len(ready_timestamps)}],处理中 [{len(processing_timestamps)}])"
)
# 【核心分支】:如果列表中出现的数量还不够,或者有任务在“执行中”,说明数据还没完全落盘
if total_found < total_expected or len(processing_timestamps) > 0:
print(" ⏳ 发现任务缺失或正在生成,执行【点击查询按钮】触发局部重载...")
try:
# 寻找并在 iframe 内部执行局部刷新按钮的点击
taskdone_frame.locator(
".mini-button-text", has_text="查询"
).first.click()
except Exception as e:
print(f" ⚠️ 局部刷新按钮失效,使用菜单后备刷新: {e}")
page.locator("li.leaf span.menu-name", has_text="导出任务管理").click()
page.wait_for_timeout(3000) # 挂起 3 秒进入下一轮状态机检测
else:
# 数量全满,且全部为“成功执行”
target_task_timestamps = list(ready_timestamps)
print(">> ✅ 目标任务已全部生成,解除锁定,开始接收文件流...")
break
# 8. 下载逻辑
downloaded_files = []
for time_str in target_task_timestamps:
try:
target_row = (
taskdone_frame.locator(
"#taskdoneDatagrid .mini-grid-rows-view .mini-grid-row"
)
.filter(
has=taskdone_frame.locator(
f"td:nth-child(5):has-text('{time_str}')"
)
)
.first
)
print(f" 🎯 触发下载 -> 任务 [{time_str}] ...")
checkbox = target_row.locator(".mini-grid-checkbox")
if checkbox.is_visible():
checkbox.click()
print(" -> 已勾选当前记录的 Checkbox")
page.wait_for_timeout(500)
with page.expect_download() as download_info:
target_row.locator("td").nth(10).locator(".ui-btn-download").click()
download = download_info.value
save_path = os.path.join(download_dir, download.suggested_filename)
download.save_as(save_path)
downloaded_files.append(save_path)
print(f" ⬇️ 文件已落盘: downloads/{download.suggested_filename}")
if checkbox.is_visible():
checkbox.click()
print(" -> 已取消勾选,清理现场进入下一条")
page.wait_for_timeout(500)
except Exception as e:
print(f" ❌ 下载任务 [{time_str}] 失败: {e}")
# 9. 合并数据
if downloaded_files:
print("\n>> 🧪 正在开始执行扁平数据高能合并与清洗流程...")
all_data_frames = []
for file_path in downloaded_files:
try:
df = pd.read_excel(file_path, dtype=str)
if not df.empty:
all_data_frames.append(df)
except Exception:
pass
if all_data_frames:
combined_df = pd.concat(all_data_frames, ignore_index=True)
final_output_path = os.path.join(download_dir, final_filename)
combined_df.to_excel(final_output_path, index=False)
print(f"====================================================")
print(f" 🎉 恭喜!数据流闭环完成!")
print(f" 📁 最终输出成果归档至: {final_output_path}")
print(f"====================================================")
for file_path in downloaded_files:
os.remove(file_path)
print(" ✅ 临时缓存数据清理完毕。")