Files
InboundVerify/site_yunda.py
Misaka ab3f25a10e Centralize paths in paths.py and harden waybill-number handling
- Add paths.py: BASE_DIR/DOWNLOAD_DIR/CONFIG_PATH anchored on __file__
  so paths resolve regardless of the launch cwd
- Route main_router and all site modules through DOWNLOAD_DIR/CONFIG_PATH,
  replacing os.getcwd()-based download dirs and the "config.yaml" literal
- main_router compare engine: read Excel as str with keep_default_na,
  strip/normalize 运单号, drop blanks, and isin against a set so
  int/float-vs-str mismatches no longer produce false "undelivered"
- Shunxin: give downloaded files unique microsecond temp names and read
  Excel as str to preserve long-waybill precision
- Normalize bare except: to except Exception: across affected files

Co-Authored-By: Claude <noreply@anthropic.com>
2026-06-20 22:28:30 +08:00

609 lines
25 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# site_yunda.py
import os
import re
import yaml
from datetime import datetime, timedelta
import pandas as pd
from paths import DOWNLOAD_DIR, CONFIG_PATH
def yunda_login(page):
"""
韵达大运系统智能自动登录流
支持‘未登录自动填充提交’与‘已登录静默过客’双工模式
"""
print(">> 正在探测韵达当前的登录会话状态...")
try:
# 定位“账号密码登录”切换按钮
switch_btn = page.locator("span", has_text="账号密码登录")
# 设定 5 秒探针延迟。如果 5 秒内按钮可见,说明处于未登录的初始状态
if switch_btn.is_visible(timeout=5000):
print(" -> 捕获到标准的未登录界面,正在强切至【账号密码登录】模式...")
switch_btn.click()
page.wait_for_timeout(500)
# 从配置读取凭证(默认空串;真实凭据仅存于被忽略的 config.yaml
username = ""
password = ""
if os.path.exists(CONFIG_PATH):
with open(CONFIG_PATH, "r", encoding="utf-8") as f:
config = yaml.safe_load(f) or {}
yd_cfg = config.get("yunda", {})
username = str(yd_cfg.get("username", ""))
password = str(yd_cfg.get("password", ""))
print(f" -> 正在程序化充填表单凭证 (账号: {username})...")
page.locator("#username").fill(username)
page.locator("#password").fill(password)
page.wait_for_timeout(300)
print(" -> 正在点击【登录】按钮并提交表单...")
page.locator('button[type="submit"]', has_text="登录").click()
page.wait_for_timeout(1000)
else:
print(
" -> 探针未发现登录按钮,判定当前会话已处于持久化登录状态,静默穿透。"
)
except Exception as e:
print(f" ⚠️ 自动登录状态机探测发生波动(可能已处于工作台内部): {e}")
def yunda_smart_menu_click(page, menu_path):
"""
韵达排他性多级树形菜单智能导航器(带防折叠状态断言机制)
采用 XPath 亲子轴绝对隔离技术,彻底斩断嵌套手风琴菜单的向下漏包干扰
"""
print(f">> 正在智能路由韵达菜单: {' -> '.join(menu_path)}")
for item in menu_path:
title_locator = page.locator(
f"xpath=//div[contains(@class, 'el-submenu__title') and .//span[normalize-space(.)='{item}']]"
).first
parent_li = page.locator(
f"xpath=//div[contains(@class, 'el-submenu__title') and .//span[normalize-space(.)='{item}']]/.."
).first
leaf_locator = page.locator(
f"xpath=//li[contains(@class, 'el-menu-item') and .//span[normalize-space(.)='{item}']]"
).first
if title_locator.is_visible():
current_class = parent_li.get_attribute("class") or ""
is_opened = "is-opened" in current_class
if not is_opened:
print(f" -> 探测到父菜单 [{item}] 当前处于【收起】状态,执行点击展开")
title_locator.click()
page.wait_for_timeout(500)
else:
print(
f" -> 探测到父菜单 [{item}] 当前已经处于【展开】状态,安全跳过点击(防折叠保护激活)"
)
elif leaf_locator.is_visible():
print(f" -> 成功对焦目标叶子节点 [{item}],直接执行跳转点击")
leaf_locator.click()
page.wait_for_timeout(1000)
def yunda_expected_download(page):
"""韵达:应到货物数据下载"""
print("\n▶ 开始执行【韵达 - 应到货物数据下载】任务...")
target_task_title = "进站主单表"
download_dir = DOWNLOAD_DIR
if not os.path.exists(download_dir):
os.makedirs(download_dir)
export_times = []
try:
# 1. 验证首页并智能路由
page.locator(".el-menu-item", has_text="首页").wait_for(
state="visible", timeout=15000
)
print("✅ 韵达工作台首页成功加载")
yunda_smart_menu_click(page, ["运营管理", "进站管理", "进站交接单查询"])
print(">> 正在跨域动态追踪【进站交接单查询】业务窗体...")
# 彻底移除会引起竞速超时坑的 custom 探测器,使用原生懒加载 frame_locator
ws_frame = page.frame_locator("section iframe")
ws_frame.locator("#startTime").wait_for(state="attached", timeout=15000)
print("✅ 进站交接单查询工作区就绪")
# 2. 从配置文件中解析并计算绝对日期跨度
query_days = 1
try:
if os.path.exists(CONFIG_PATH):
with open(CONFIG_PATH, "r", encoding="utf-8") as f:
config = yaml.safe_load(f) or {}
query_days = int(config.get("yunda", {}).get("query_days", 1))
except Exception as e:
print(f" ⚠️ 读取 config.yaml 失败,默认查询 1 天: {e}")
today = datetime.now()
start_date = today - timedelta(days=(query_days - 1))
today_ymd = f"{today.year}-{today.month}-{today.day}"
start_date_ymd = f"{start_date.year}-{start_date.month}-{start_date.day}"
print(f">> 正在精准对焦时间区间: [{start_date_ymd}] 至 [{today_ymd}]")
# 设定起始时间
print(" >> 呼出起始时间控件...")
page.wait_for_timeout(1000)
ws_frame.locator("#startTime").click(force=True)
calendar1 = ws_frame.locator(".layui-laydate:visible").first
calendar1.wait_for(state="visible", timeout=5000)
calendar1.locator(f"td[lay-ymd='{start_date_ymd}']").click()
calendar1.locator(".laydate-btns-confirm").click()
page.wait_for_timeout(400)
# 设定截止时间
print(" >> 呼出截止时间控件...")
ws_frame.locator("#endTime").click(force=True)
calendar2 = ws_frame.locator(".layui-laydate:visible").first
calendar2.wait_for(state="visible", timeout=5000)
calendar2.locator(f"td[lay-ymd='{today_ymd}']").click()
calendar2.locator(".laydate-btns-confirm").click()
page.wait_for_timeout(500)
# 3. 多维状态机静默加载校验
print(">> 正在触发现场查询数据流...")
ws_frame.locator("a.btn-success", has_text="查询").click()
page.wait_for_timeout(800)
# ====================================================================
# 🛡️ 核心修复:将 Loading 蒙层和数据断言限制在当前的 #tab-1 结界内部
# 彻底解决多标签页导致 Strict Mode (5 elements found) 的污染问题
# ====================================================================
loading_mask = ws_frame.locator(
"#tab-1 .fixed-table-loading", has_text="正在努力地加载数据中"
).first
if loading_mask.is_visible():
print(" ⏳ 检测到专属数据加载罩,正在等待后端重载返回...")
loading_mask.wait_for(state="hidden", timeout=30000)
page.wait_for_timeout(500)
sum_panel = ws_frame.locator("#sum").first
has_data = False
if sum_panel.is_visible():
sum_text = sum_panel.inner_text()
match_tickets = re.search(r"进站实际票数:(\d+)", sum_text)
if match_tickets and int(match_tickets.group(1)) > 0:
has_data = True
print(
f" ✅ 深度断言:局部统计面板加载完毕,实际票数: [{match_tickets.group(1)}],执行穿透。"
)
if not has_data:
# 同样将空数据提示隔离在当前 tab 中
if ws_frame.locator(
"#tab-1 .no-records-found", has_text="没有找到匹配的记录"
).first.is_visible():
print(
" ⚠️ 确认为空数据环境:系统提示【没有找到匹配的记录】,正在执行复位净化..."
)
page.locator(".tags-view-item", has_text="进站交接单查询").locator(
".el-icon-close"
).click()
return
# 4. 深度等待表格第一行数据行渲染就绪
ws_frame.locator("#exampleTable1 tbody tr[data-index='0']").wait_for(
state="visible", timeout=10000
)
main_rows = ws_frame.locator("#exampleTable1 tbody tr[data-index]")
row_count = main_rows.count()
print(f">> 当前视窗共捕获到活跃交接单记录: {row_count}")
# 5. 循环双击穿透提交
for i in range(row_count):
print(f" ⏳ 正在处理第 {i+1}/{row_count} 个交接单模块...")
current_row = ws_frame.locator("#exampleTable1 tbody tr[data-index]").nth(i)
raw_no = current_row.locator("td").nth(1).inner_text().strip()
# ====================================================================
# 🛡️ 状态机拦截卫语句:识别绑定状态,过滤已绑定交接单
# ====================================================================
bind_status = current_row.locator("td").nth(2).inner_text().strip()
print(f" -> 锁定提取单号: {raw_no} [绑定状态: {bind_status}]")
if bind_status == "已绑定":
print(" ⏭️ 状态拦截:该交接单处于【已绑定】状态,安全跳过。")
continue
# ====================================================================
current_row.dblclick()
ws_frame.locator("#docSum").wait_for(state="visible", timeout=15000)
page.wait_for_timeout(500)
ws_frame.locator('a.btn-info[onclick*="exportFile"]').click()
ws_frame.locator(".layui-layer-title", has_text="数据导出").wait_for(
state="visible", timeout=15000
)
export_frame = ws_frame.frame_locator('iframe[name="target1"]')
export_frame.locator(".allRight").click()
page.wait_for_timeout(400)
print(" >> 正在建立后台离线任务...")
task_success = False
for attempt in range(5):
export_frame.locator("#submitbutton", has_text="导出数据").click()
confirm_link = export_frame.get_by_role("link", name="确定")
try:
confirm_link.wait_for(state="visible", timeout=6000)
if export_frame.get_by_text("导出任务建立成功").is_visible():
print(" ✅ 判定通过:成功捕获到【导出任务建立成功】特征!")
confirm_link.click()
task_success = True
break
elif export_frame.get_by_text(
"请选择格式相应的导出字段"
).is_visible():
print(
" ⚠️ 警告:检测到【未选择字段】错误,重新触发补点全选..."
)
confirm_link.click()
page.wait_for_timeout(500)
export_frame.locator(".allRight").click()
page.wait_for_timeout(500)
else:
confirm_link.click()
page.wait_for_timeout(1000)
except Exception:
page.wait_for_timeout(1000)
if not task_success:
raise RuntimeError(
"致命异常:连续 5 次尝试均无法成功建立应到数据离线任务。"
)
ws_frame.locator(".layui-layer-close1").click()
page.wait_for_timeout(500)
ws_frame.locator("#myTab a", has_text="交接单信息").click()
page.wait_for_timeout(800)
export_times.append(datetime.now())
print(">> 📤 任务提交流闭环,正在执行【进站交接单查询】工作台销毁...")
page.locator(".tags-view-item", has_text="进站交接单查询").locator(
".el-icon-close"
).click()
page.wait_for_timeout(500)
# 防线如果全部记录都被跳过了export_times 为空,直接结束
if not export_times:
print(
">> ⚠️ 本次查询未产生任何有效的离线下载任务(全部空单或已被跳过),中止后端收割流。"
)
return
_yunda_poll_and_download_tasks(
page,
export_times,
target_task_title,
download_dir,
"韵达-应到货物数据.xlsx",
)
except Exception as e:
print(f"\n❌ 任务执行过程中发生异常: {e}")
def yunda_actual_download(page):
"""韵达:实到货物数据下载"""
print("\n▶ 开始执行【韵达 - 实到货物数据下载】任务...")
target_task_title = "扫描记录数据"
download_dir = DOWNLOAD_DIR
if not os.path.exists(download_dir):
os.makedirs(download_dir)
export_times = []
try:
page.locator(".el-menu-item", has_text="首页").wait_for(
state="visible", timeout=15000
)
print("✅ 韵达工作台首页成功加载")
yunda_smart_menu_click(page, ["报表管理", "扫描记录查询"])
print(">> 正在跨域动态追踪【扫描记录查询】业务窗体...")
ws_frame = page.frame_locator("section iframe")
ws_frame.locator(
".no-records-found", has_text="没有找到匹配的记录"
).first.wait_for(state="visible", timeout=15000)
print("✅ 扫描记录查询工作区初始化完毕")
query_days = 1
try:
if os.path.exists(CONFIG_PATH):
with open(CONFIG_PATH, "r", encoding="utf-8") as f:
config = yaml.safe_load(f) or {}
query_days = int(config.get("yunda", {}).get("query_days", 1))
except Exception:
pass
today = datetime.now()
start_date = today - timedelta(days=(query_days - 1))
print(f">> 正在精准对焦实到时间区间: [近 {query_days} 天]")
print(" >> 正在设定起始时间...")
ws_frame.locator("#startDate").click()
page.wait_for_timeout(400)
box1 = ws_frame.locator("#laydate_box:visible").first
box1.locator(
f"td[y='{start_date.year}'][m='{start_date.month}'][d='{start_date.day}']"
).click()
page.wait_for_timeout(400)
print(" >> 正在设定截止时间...")
ws_frame.locator("#endDate").click()
page.wait_for_timeout(400)
box2 = ws_frame.locator("#laydate_box:visible").first
box2.locator(
f"td[y='{today.year}'][m='{today.month}'][d='{today.day}']"
).click()
page.wait_for_timeout(500)
print(" >> 正在变更扫描类型为【到件】...")
ws_frame.locator("#scanRecordTyp").select_option(value="03")
page.wait_for_timeout(500)
print(">> 正在触发现场查询数据流...")
ws_frame.locator('input[type="button"][value="查询"]').click()
page.wait_for_timeout(800)
loading_mask = ws_frame.locator(
".fixed-table-loading", has_text="正在努力地加载数据中"
).first
if loading_mask.is_visible():
print(" ⏳ 检测到专属数据加载罩,正在等待后端重载返回...")
loading_mask.wait_for(state="hidden", timeout=30000)
page.wait_for_timeout(500)
pg_info = ws_frame.locator(".pagination-info").first
has_records = False
if pg_info.is_visible():
info_text = pg_info.inner_text()
match_total = re.search(r"总共\s*(\d+)\s*条记录", info_text)
if match_total and int(match_total.group(1)) > 0:
has_records = True
print(
f" ✅ 深度断言:实到数据渲染完毕,总记录数: [{match_total.group(1)}] 条。"
)
if not has_records:
if ws_frame.locator(
".no-records-found", has_text="没有找到匹配的记录"
).first.is_visible():
print(
" ⚠️ 当前查询范围内确认为空数据:系统提示【没有找到匹配的记录】,终止并关闭环境。"
)
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
".el-icon-close"
).click()
return
else:
print(" ⚠️ 发生渲染异常,未找到数据也未找到空记录提示。安全收尾...")
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
".el-icon-close"
).click()
return
# 5. 执行导出流
print(">> 正在发起【导出】申请指令...")
ws_frame.locator('input[type="button"][id="export"]').click()
ws_frame.locator(".layui-layer-title", has_text="数据导出").wait_for(
state="visible", timeout=15000
)
export_frame = ws_frame.frame_locator('iframe[name="myFrame"]')
export_frame.locator(".allRight").click()
page.wait_for_timeout(400)
print(" >> 正在建立后台离线任务...")
task_success = False
for attempt in range(5):
export_frame.locator("#submitbutton", has_text="导出数据").click()
confirm_link = export_frame.get_by_role("link", name="确定")
try:
confirm_link.wait_for(state="visible", timeout=6000)
if export_frame.get_by_text("导出任务建立成功").is_visible():
print(" ✅ 判定通过:成功捕获到【导出任务建立成功】特征!")
confirm_link.click()
task_success = True
break
elif export_frame.get_by_text("请选择格式相应的导出字段").is_visible():
print(" ⚠️ 警告:检测到字段未全选,执行强补点击...")
confirm_link.click()
page.wait_for_timeout(500)
export_frame.locator(".allRight").click()
page.wait_for_timeout(500)
else:
confirm_link.click()
page.wait_for_timeout(1000)
except Exception:
page.wait_for_timeout(1000)
if not task_success:
raise RuntimeError(
"致命异常:连续 5 次尝试均无法成功建立实到数据离线任务。"
)
ws_frame.locator(".layui-layer-close1").click()
page.wait_for_timeout(500)
export_times.append(datetime.now())
print(">> 📤 任务提交流闭环,正在执行【扫描记录查询】工作台销毁...")
page.locator(".tags-view-item", has_text="扫描记录查询").locator(
".el-icon-close"
).click()
page.wait_for_timeout(500)
# 防线:双重保护
if not export_times:
print(">> ⚠️ 本次查询未产生有效的离线下载任务,中止后端收割流。")
return
# 6. 收割下载
_yunda_poll_and_download_tasks(
page,
export_times,
target_task_title,
download_dir,
"韵达-实到货物数据.xlsx",
)
except Exception as e:
print(f"\n❌ 任务执行过程中发生异常: {e}")
def _yunda_poll_and_download_tasks(
page, export_times, target_task_title, download_dir, final_filename
):
"""韵达专属离线任务轮询下载引擎"""
print("\n>> 正在前往【导出服务】中心...")
yunda_smart_menu_click(page, ["基础数据", "导出服务"])
export_ws_frame = page.frame_locator("section iframe")
export_ws_frame.get_by_role("cell", name="模块名称", exact=True).wait_for(
state="visible", timeout=15000
)
page.wait_for_timeout(1000)
print(">> 离线文件队列已对接启动【1分钟高频精确校对+缺单局部重载刷新】断言...")
total_expected = len(export_times)
while True:
task_rows = export_ws_frame.locator(
".datagrid-view2 .datagrid-btable tbody tr.datagrid-row"
)
row_count = task_rows.count()
ready_indices = []
processing_indices = []
for idx in range(row_count):
row = task_rows.nth(idx)
module_name = row.locator("td[field='modueName']").inner_text().strip()
status_name = row.locator("td[field='fileStatus']").inner_text().strip()
create_time_str = (
row.locator("td[field='createdTime']").inner_text().strip()
)
if module_name == target_task_title:
try:
row_time = datetime.strptime(create_time_str, "%Y-%m-%d %H:%M:%S")
matched = any(
abs((row_time - et).total_seconds()) <= 60
for et in export_times
)
if matched:
if status_name == "导出完成":
ready_indices.append(idx)
else:
processing_indices.append(idx)
except Exception:
pass
total_found = len(ready_indices) + len(processing_indices)
print(
f" 📊 状态研判:自建期望数 [{total_expected}],实际入表 [{total_found}] (完成 [{len(ready_indices)}],生成中 [{len(processing_indices)}])"
)
if total_found < total_expected or len(processing_indices) > 0:
print(" ⏳ 队列未齐,执行【点击查询按钮】触发局部无痕刷新...")
export_ws_frame.locator(
"#ydkyimport_basic_export_searchData1_ky_export_common"
).click()
page.wait_for_timeout(3000)
else:
print(">> ✅ 所有目标离线任务全量就绪!开始依序接入文件流...")
break
downloaded_files = []
for row_idx in ready_indices:
try:
target_row = export_ws_frame.locator(
".datagrid-view2 .datagrid-btable tbody tr.datagrid-row"
).nth(row_idx)
time_flag = (
target_row.locator("td[field='createdTime']").inner_text().strip()
)
print(f" 🎯 触发下载 -> 离线任务时间节点: [{time_flag}] ...")
with page.expect_download() as download_info:
target_row.locator("td[field='extreFile'] a").get_by_text(
"下载"
).first.click()
download = download_info.value
safe_timestamp = datetime.now().strftime("%Y%m%d_%H%M%S_%f")
custom_filename = f"韵达_temp_{safe_timestamp}.xlsx"
save_path = os.path.join(download_dir, custom_filename)
download.save_as(save_path)
downloaded_files.append(save_path)
print(f" ⬇️ 文件已用安全序列号落盘: downloads/{custom_filename}")
page.wait_for_timeout(500)
except Exception as e:
print(f" ❌ 文件流接收失败: {e}")
print(">> 📥 【导出服务】数据提取链闭环,正在执行当前 Tab 窗口销毁...")
try:
page.locator(".tags-view-item", has_text="导出服务").locator(
".el-icon-close"
).click()
print(" ✅ 【导出服务】工作区已安全关闭。")
except Exception:
pass
if downloaded_files:
print("\n>> 🧪 正在启动离线数据清洗与高能扁平合并流...")
all_dfs = []
for file_path in downloaded_files:
try:
df = pd.read_excel(file_path, dtype=str)
if not df.empty:
all_dfs.append(df)
except Exception:
pass
if all_dfs:
combined_df = pd.concat(all_dfs, ignore_index=True)
final_output = os.path.join(download_dir, final_filename)
combined_df.to_excel(final_output, index=False)
print(f"====================================================")
print(f" 🎉 恭喜!韵达网点数据拉取合流大功告成!")
print(f" 📁 最终存储位置: {final_output}")
print(f"====================================================")
for file_path in downloaded_files:
os.remove(file_path)
print(" ✅ 临时缓存阵列已无缝净化。")