- Add site_zto.py: ZTO expected-arrival data download with menu navigation, export polling, and Excel merge - Wire ZTO into main_router: site URL, login detection, menu items [5]/[6], and a global offline Left Anti-Join compare entry [9] - Read config.yaml debug flags to launch only the target site for focused debugging; guard each site init and route handler behind site-loaded checks via is_site_ready() - Expand config.example.yaml with documented debug, baishi, and zto config keys Co-Authored-By: Claude <noreply@anthropic.com>
360 lines
15 KiB
Python
360 lines
15 KiB
Python
# site_zto.py
|
|
|
|
import os
|
|
import re
|
|
import yaml
|
|
from datetime import datetime
|
|
import pandas as pd
|
|
|
|
|
|
def _wait_and_get_frame(page, text_indicator, timeout_ms=20000):
|
|
"""动态雷达探测器:全域扫描所有视窗"""
|
|
start_time = datetime.now()
|
|
while (datetime.now() - start_time).total_seconds() * 1000 < timeout_ms:
|
|
# 1. 探测最外层
|
|
try:
|
|
if page.get_by_text(text_indicator).count() > 0:
|
|
return page
|
|
except:
|
|
pass
|
|
|
|
# 2. 穿透扫描所有活跃的子 iframe
|
|
for frame in page.frames:
|
|
try:
|
|
if frame.get_by_text(text_indicator).count() > 0:
|
|
return frame
|
|
except:
|
|
pass
|
|
|
|
page.wait_for_timeout(300)
|
|
raise TimeoutError(f"爆栈超时:全域未死守到包含 [{text_indicator}] 的弹窗视窗。")
|
|
|
|
|
|
def _ensure_menu_expanded(page, level1_name, level2_name=None):
|
|
"""MiniUI 智能菜单展开器"""
|
|
print(f">> 正在智能路由菜单...")
|
|
if level2_name:
|
|
if not page.locator(
|
|
"li.shrink span.menu-name", has_text=level2_name
|
|
).is_visible():
|
|
page.locator(
|
|
"a.treeview-title span.menu-name", has_text=level1_name
|
|
).click()
|
|
page.wait_for_timeout(500)
|
|
else:
|
|
if not page.locator(
|
|
"a.treeview-title span.menu-name", has_text=level1_name
|
|
).is_visible():
|
|
page.locator(
|
|
"a.treeview-title span.menu-name", has_text=level1_name
|
|
).click()
|
|
page.wait_for_timeout(500)
|
|
|
|
|
|
def zto_smart_menu_click(page, menu_path):
|
|
"""中通智能菜单导航器"""
|
|
print(f">> 正在导航: {' -> '.join(menu_path)}")
|
|
for i in range(len(menu_path)):
|
|
current_menu = menu_path[i]
|
|
if i < len(menu_path) - 1:
|
|
next_menu = menu_path[i + 1]
|
|
next_locator = page.locator("span.menu-name", has_text=next_menu).first
|
|
if not next_locator.is_visible():
|
|
page.locator("span.menu-name", has_text=current_menu).first.click()
|
|
page.wait_for_timeout(800)
|
|
else:
|
|
page.locator("span.menu-name", has_text=current_menu).first.click()
|
|
page.wait_for_timeout(1000)
|
|
|
|
|
|
def zto_expected_download(page):
|
|
"""中通:应到货物数据下载"""
|
|
print("\n▶ 开始执行【中通 - 应到货物数据下载】任务...")
|
|
|
|
download_dir = os.path.join(os.getcwd(), "downloads")
|
|
if not os.path.exists(download_dir):
|
|
os.makedirs(download_dir)
|
|
|
|
export_times = []
|
|
|
|
try:
|
|
# 1. 智能导航
|
|
zto_smart_menu_click(page, ["运营管理", "进站管理", "进站交接单查询"])
|
|
|
|
ewb_frame = page.frame_locator('iframe[src*="inEwbsListNoSearch"]')
|
|
|
|
print(">> 正在探针检测表格统计面板 (#inEwbCount)...")
|
|
ewb_frame.locator("#inEwbCount").wait_for(state="attached", timeout=15000)
|
|
|
|
# 读取 YAML 配置设定天数
|
|
query_days = 1
|
|
try:
|
|
if os.path.exists("config.yaml"):
|
|
with open("config.yaml", "r", encoding="utf-8") as f:
|
|
config = yaml.safe_load(f) or {}
|
|
query_days = int(config.get("zto", {}).get("query_days", 1))
|
|
except Exception as e:
|
|
print(f" ⚠️ 读取 config.yaml 失败,默认查询 1 天: {e}")
|
|
|
|
print(f">> 正在设定查询时间范围为近【{query_days}】天...")
|
|
ewb_frame.locator("#beginDate").click()
|
|
page.wait_for_timeout(500)
|
|
|
|
today_cell = ewb_frame.locator("td div.day.real-today").first
|
|
today_cell.wait_for(state="visible")
|
|
|
|
today_time_str = today_cell.get_attribute("time")
|
|
if today_time_str:
|
|
today_time = int(today_time_str)
|
|
start_time = today_time - (query_days - 1) * 86400000
|
|
start_cell = ewb_frame.locator(f"td div.day[time='{start_time}']").first
|
|
|
|
if start_cell.is_visible():
|
|
start_cell.click()
|
|
page.wait_for_timeout(300)
|
|
today_cell.click()
|
|
else:
|
|
print(" ⚠️ 设定的天数过大,不在当前日历视窗内,自动降级为查询当天。")
|
|
today_cell.click()
|
|
page.wait_for_timeout(300)
|
|
today_cell.click()
|
|
else:
|
|
today_cell.click()
|
|
page.wait_for_timeout(300)
|
|
today_cell.click()
|
|
|
|
page.wait_for_timeout(500)
|
|
|
|
# 3. 触发查询与深度状态机判定
|
|
print(">> 正在点击【查询】按钮并等待数据响应...")
|
|
ewb_frame.locator("#searchbtn").click()
|
|
|
|
page.wait_for_timeout(1000)
|
|
|
|
loading_mask = ewb_frame.locator(".mini-mask-loading", has_text="加载中")
|
|
if loading_mask.is_visible():
|
|
print(" ⏳ 检测到数据加载遮罩层,等待系统渲染...")
|
|
loading_mask.wait_for(state="hidden", timeout=30000)
|
|
|
|
page.wait_for_timeout(1000)
|
|
|
|
empty_flag = ewb_frame.locator("span", has_text="没有搜索到符合条件的数据记录")
|
|
if empty_flag.is_visible():
|
|
print(" ⚠️ 当前查询范围内【没有搜索到符合条件的数据记录】,终止导出。")
|
|
return
|
|
|
|
ticket_count_str = ewb_frame.locator("#inEwbCount").inner_text().strip()
|
|
print(f" ✅ 数据渲染完毕!当前进站实际票数为: [{ticket_count_str}]")
|
|
|
|
# 4. 提取主表记录并循环双击
|
|
main_rows = ewb_frame.locator("#datagrid1 .mini-grid-rows-view .mini-grid-row")
|
|
count = main_rows.count()
|
|
print(f">> 共发现 {count} 个交接单需要导出。")
|
|
|
|
for i in range(count):
|
|
print(f" ⏳ 正在处理第 {i+1}/{count} 个交接单...")
|
|
|
|
row = ewb_frame.locator(
|
|
"#datagrid1 .mini-grid-rows-view .mini-grid-row"
|
|
).nth(i)
|
|
|
|
# 正则表达式清洗交接单号
|
|
raw_text = row.locator("td").nth(3).inner_text()
|
|
match = re.search(r"\d{18}", raw_text)
|
|
handover_no = match.group(0) if match else raw_text.strip()
|
|
print(f" -> 锁定提取单号:{handover_no}")
|
|
|
|
row.dblclick()
|
|
|
|
ewb_frame.locator("#datagrid2").get_by_text("运单号").wait_for(
|
|
state="visible"
|
|
)
|
|
ewb_frame.locator(
|
|
"#datagrid2 .mini-grid-row", has_text=handover_no
|
|
).first.wait_for(state="visible")
|
|
|
|
# 5. 执行导出流程
|
|
ewb_frame.locator("#exportExcel").click()
|
|
|
|
page.locator(".mini-panel-title", has_text="导出选择列").wait_for(
|
|
state="visible"
|
|
)
|
|
export_frame = page.frame_locator('iframe[src*="download"]')
|
|
|
|
export_frame.locator(".mini-button-text", has_text=">>").click()
|
|
page.wait_for_timeout(300)
|
|
export_frame.locator(".mini-button-text", has_text="确定").click()
|
|
|
|
print(" >> [雷达扫描] 正在跨域动态追踪【温馨提示】所在的隐身视窗...")
|
|
ctx_alert = _wait_and_get_frame(page, "温馨提示")
|
|
|
|
ctx_alert.locator(
|
|
".mini-messagebox-buttons .mini-button-text", has_text="确定"
|
|
).click()
|
|
print(" ✅ 【温馨提示】已成功通过父级过滤并确认!")
|
|
|
|
print(" >> 正在等待服务器建立后台离线任务...")
|
|
try:
|
|
ctx_tips = _wait_and_get_frame(
|
|
page, "生成离线导出任务成功", timeout_ms=10000
|
|
)
|
|
ctx_tips.locator(".mini-tips-success").wait_for(
|
|
state="hidden", timeout=15000
|
|
)
|
|
print(" ✅ 成功提示框已平滑隐藏。")
|
|
except Exception as e:
|
|
print(" ✅ 成功提示框及其容器已自动销毁 (触发 Detached 拦截)。")
|
|
|
|
export_times.append(datetime.now())
|
|
|
|
# 同域无缝 Tab 切换回交接单信息
|
|
print(" >> 切换回【交接单信息】标签页...")
|
|
ewb_frame.locator("#ewbsListNo").click()
|
|
page.wait_for_timeout(1000)
|
|
|
|
if count > 0:
|
|
print("✅ 所有交接单的导出任务已成功提交!")
|
|
else:
|
|
print("⚠️ 未发现任何数据,直接跳转至下载环节。")
|
|
|
|
# 6. 前往【导出任务管理】
|
|
print("\n>> 正在前往【导出任务管理】界面...")
|
|
zto_smart_menu_click(page, ["系统配置", "导出任务管理"])
|
|
|
|
taskdone_frame = page.frame_locator('iframe[src*="taskdone"]')
|
|
|
|
taskdone_frame.locator("#taskdoneDatagrid").get_by_text("任务标题").wait_for(
|
|
state="visible"
|
|
)
|
|
page.wait_for_timeout(2000)
|
|
|
|
# 7. 轮询任务状态
|
|
print(">> 列表中已渲染,开始匹配并检查后端处理状态...")
|
|
while True:
|
|
task_rows = taskdone_frame.locator(
|
|
"#taskdoneDatagrid .mini-grid-rows-view .mini-grid-row"
|
|
)
|
|
row_count = task_rows.count()
|
|
pending_tasks = 0
|
|
current_ready_timestamps = []
|
|
|
|
for i in range(row_count):
|
|
tds = task_rows.nth(i).locator("td")
|
|
if tds.count() < 10:
|
|
continue
|
|
|
|
title_str = tds.nth(3).inner_text().strip()
|
|
submit_time_str = tds.nth(4).inner_text().strip()
|
|
status_str = tds.nth(6).inner_text().strip()
|
|
|
|
if title_str == "进站交接单查询-运单信息":
|
|
try:
|
|
row_time = datetime.strptime(
|
|
submit_time_str, "%Y-%m-%d %H:%M:%S"
|
|
)
|
|
matched = any(
|
|
abs((row_time - et).total_seconds()) <= 120
|
|
for et in export_times
|
|
)
|
|
|
|
if matched:
|
|
if status_str != "成功执行":
|
|
pending_tasks += 1
|
|
if submit_time_str not in current_ready_timestamps:
|
|
print(
|
|
f" ⏳ 任务 [{submit_time_str}] 状态为【{status_str}】,数据生成中..."
|
|
)
|
|
else:
|
|
if submit_time_str not in current_ready_timestamps:
|
|
current_ready_timestamps.append(submit_time_str)
|
|
except Exception as e:
|
|
print(f" ⚠️ 解析时间时出错: {e}")
|
|
|
|
if pending_tasks > 0:
|
|
print(
|
|
f">> 共有 {pending_tasks} 个匹配任务还在处理中,等待 5 秒后刷新..."
|
|
)
|
|
page.wait_for_timeout(5000)
|
|
page.locator("li.leaf span.menu-name", has_text="导出任务管理").click()
|
|
page.wait_for_timeout(2000)
|
|
else:
|
|
target_task_timestamps = current_ready_timestamps
|
|
if len(target_task_timestamps) > 0:
|
|
print(">> ✅ 所有目标任务已就绪!开始并行下载...")
|
|
break
|
|
|
|
# 8. 下载逻辑
|
|
downloaded_files = []
|
|
for time_str in target_task_timestamps:
|
|
try:
|
|
target_row = taskdone_frame.locator(
|
|
"#taskdoneDatagrid .mini-grid-rows-view .mini-grid-row"
|
|
).filter(
|
|
has=taskdone_frame.locator(
|
|
f"td:nth-child(5):has-text('{time_str}')"
|
|
)
|
|
)
|
|
print(f" 🎯 触发下载 -> 任务 [{time_str}] ...")
|
|
|
|
# 闭环 Checkbox 勾选逻辑
|
|
checkbox = target_row.locator(".mini-grid-checkbox")
|
|
if checkbox.is_visible():
|
|
checkbox.click()
|
|
print(" -> 已勾选当前记录的 Checkbox")
|
|
page.wait_for_timeout(500)
|
|
|
|
with page.expect_download() as download_info:
|
|
target_row.locator("td").nth(10).locator(".ui-btn-download").click()
|
|
|
|
download = download_info.value
|
|
save_path = os.path.join(download_dir, download.suggested_filename)
|
|
download.save_as(save_path)
|
|
downloaded_files.append(save_path)
|
|
print(f" ⬇️ 文件已落盘: downloads/{download.suggested_filename}")
|
|
|
|
if checkbox.is_visible():
|
|
checkbox.click()
|
|
print(" -> 已取消勾选,清理现场进入下一条")
|
|
page.wait_for_timeout(500)
|
|
|
|
except Exception as e:
|
|
print(f" ❌ 下载任务 [{time_str}] 失败: {e}")
|
|
|
|
# 9. 合并数据
|
|
if downloaded_files:
|
|
print("\n>> 🧪 正在开始执行扁平数据高能合并流程...")
|
|
all_data_frames = []
|
|
for file_path in downloaded_files:
|
|
try:
|
|
# ====================================================================
|
|
# 🛡️ 核心修复:强制以字符串类型 (str) 读取所有列,彻底防止长单号变科学计数法或丢失精度
|
|
# ====================================================================
|
|
df = pd.read_excel(file_path, dtype=str)
|
|
if not df.empty:
|
|
all_data_frames.append(df)
|
|
except Exception as e:
|
|
pass
|
|
|
|
if all_data_frames:
|
|
combined_df = pd.concat(all_data_frames, ignore_index=True)
|
|
final_output_path = os.path.join(download_dir, "中通-应到货物数据.xlsx")
|
|
combined_df.to_excel(final_output_path, index=False)
|
|
print(f"====================================================")
|
|
print(f" 🎉 恭喜!合并成功!最终输出路径: {final_output_path}")
|
|
print(f"====================================================")
|
|
|
|
for file_path in downloaded_files:
|
|
os.remove(file_path)
|
|
print("✅ 临时数据清理完毕。")
|
|
print("\n🎉 【中通 - 应到货物数据下载】全流程测试完毕!")
|
|
|
|
except Exception as e:
|
|
print(f"\n❌ 任务执行过程中发生异常: {e}")
|
|
|
|
|
|
def zto_actual_download(page):
|
|
"""中通:实到货物数据下载"""
|
|
print("\n▶ 开始执行【中通 - 实到货物数据下载】任务...")
|
|
print("🚧 逻辑开发中 (pass)...")
|
|
pass
|