Files
playwrite/utils/离散备料计划维护数据提取.py
Misaka_Company 53a1e33e45 feat: add SQL Server database persistence for extracted material plan data
Add optional database persistence feature that automatically saves extracted
discrete material plan data to SQL Server. Users can enable this feature in
the settings tab.

Changes:
- Add enable_db_persistence flag to ExtractionConfig (default: disabled)
- Create DiscreteMaterialPlanDAO for database operations with REPLACE pattern
- Update progress tracking to include database persistence stage (90-100%)
- Add database persistence checkbox in settings UI
- Remove verbose logging checkbox from data extraction UI (config-only now)
- Update extraction workflow to save merged DataFrame to database

Progress weights adjusted:
- download: 65% -> 60%
- database: 10% (new stage)
- Other stages adjusted accordingly

Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
2026-02-06 12:21:58 +08:00

599 lines
21 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
离散备料计划维护数据提取工具
负责登录、批量下载、转换数据
"""
import os
import pandas as pd
from playwright.sync_api import sync_playwright
from utils.excel_converter import ExcelConverter
from utils.auth import login, logout
from db.production_order_query import (
read_production_ids,
query_production_order_numbers,
)
from typing import Callable, Optional
class DiscreteMaterialPlanExtractor:
"""离散备料计划维护数据提取器"""
def __init__(
self, username, password, headless=False, verbose=True, batch_size=100,
enable_db_persistence=False
):
"""
初始化提取器
Args:
username: 登录用户名
password: 登录密码
headless: 是否无头模式运行
verbose: 是否打印详细日志
batch_size: 批次大小
enable_db_persistence: 是否启用数据库持久化
"""
self.username = username
self.password = password
self.headless = headless
self.verbose = verbose
self.batch_size = batch_size
self.progress_callback = None
self.converter = ExcelConverter(verbose=verbose)
self.enable_db_persistence = enable_db_persistence
self.dao = None
if self.enable_db_persistence:
from db.discrete_material_plan_dao import DiscreteMaterialPlanDAO
self.dao = DiscreteMaterialPlanDAO()
self.dao.__enter__() # Enter context manager
def _print(self, *args, **kwargs):
"""打印日志(如果 verbose=True"""
if self.verbose:
print(*args, **kwargs)
def _report_progress(
self, stage: str, current: int, total: int, message: str, **detail
):
"""
报告进度
Args:
stage: 阶段标识
current: 当前进度值
total: 总量
message: 显示消息
**detail: 额外详细信息
"""
if self.progress_callback:
try:
from gui.progress import ProgressInfo
progress_info = ProgressInfo(
stage=stage,
current=current,
total=total,
message=message,
detail=detail,
)
self.progress_callback(progress_info)
except Exception:
# 如果进度回调失败,忽略错误,不影响主流程
pass
def get_production_order_numbers(self, production_id_file, report_progress=False):
"""
读取总排号文件并查询数据库获取生产订单号
Args:
production_id_file: ProductionID.txt 文件路径
report_progress: 是否报告进度
Returns:
生产订单号列表
"""
if report_progress:
self._report_progress(
"query",
1,
3,
"正在读取总排号文件...",
action="read_file",
)
# 读取总排号
production_ids = read_production_ids(production_id_file)
self._print(f"从文件读取到 {len(production_ids)} 个总排号")
if report_progress:
self._report_progress(
"query",
2,
3,
f"正在查询数据库({len(production_ids)} 个总排号)...",
action="query_database",
production_id_count=len(production_ids),
)
# 查询数据库获取生产订单号
order_ids = query_production_order_numbers(production_ids)
self._print(f"查询到 {len(order_ids)} 个生产订单号")
if report_progress:
self._report_progress(
"query",
3,
3,
f"查询完成:获取到 {len(order_ids)} 个生产订单号",
action="query_complete",
order_id_count=len(order_ids),
)
return order_ids
def group_order_ids(self, order_ids, group_size=100):
"""将订单号分组"""
for i in range(0, len(order_ids), group_size):
yield order_ids[i : i + group_size]
def download_batch(
self,
inner_frame,
order_ids,
batch_index,
total_batches,
page1,
debug_mode=False,
debug_batch=None,
):
"""下载一批订单号的数据"""
from playwright.sync_api import TimeoutError
import re
import time
# 步骤1清空文本框
self._report_progress(
"download",
batch_index * 7 + 1,
total_batches * 7,
f"{batch_index + 1}/{total_batches} 批 - 准备输入订单号",
batch_index=batch_index + 1,
action="clear_textbox",
)
textbox = inner_frame.get_by_role("textbox", name="来源生产订单号")
textbox.fill("")
# 步骤2填充订单号
self._report_progress(
"download",
batch_index * 7 + 2,
total_batches * 7,
f"{batch_index + 1}/{total_batches} 批 - 输入 {len(order_ids)} 个订单号",
batch_index=batch_index + 1,
action="fill_order_ids",
order_count=len(order_ids),
)
textbox.fill(",".join(order_ids))
# 步骤3点击查询
self._report_progress(
"download",
batch_index * 7 + 3,
total_batches * 7,
f"{batch_index + 1}/{total_batches} 批 - 提交查询请求",
batch_index=batch_index + 1,
action="click_search",
)
inner_frame.locator(".search-component-searchBtn").click()
self._print(f"{batch_index + 1} 批查询完成,等待加载结果...")
# 步骤4等待加载完成
self._report_progress(
"download",
batch_index * 7 + 4,
total_batches * 7,
f"{batch_index + 1}/{total_batches} 批 - 等待数据加载...",
batch_index=batch_index + 1,
action="wait_loading",
)
loading_locator = inner_frame.locator("div").filter(has_text="加载中").nth(1)
try:
loading_locator.wait_for(state="visible", timeout=3000)
loading_locator.wait_for(state="hidden", timeout=0) # 无限等待,直到消失
except TimeoutError:
# 加载很快完成,或者没有出现加载提示
pass
self._print(f"{batch_index + 1} 批加载完成,开始选择数据...")
# 调试模式:只在指定批次暂停
if debug_mode and (debug_batch is None or batch_index == debug_batch):
self._print(f"=== 调试暂停:第 {batch_index + 1} 批 ===")
page1.pause()
# 步骤5选择所有数据
self._report_progress(
"download",
batch_index * 7 + 5,
total_batches * 7,
f"{batch_index + 1}/{total_batches} 批 - 选择所有数据行",
batch_index=batch_index + 1,
action="select_all_rows",
)
inner_frame.get_by_role("row", name="序号").get_by_label("").click()
# 步骤6配置并触发导出
self._report_progress(
"download",
batch_index * 7 + 6,
total_batches * 7,
f"{batch_index + 1}/{total_batches} 批 - 配置导出参数",
batch_index=batch_index + 1,
action="configure_export",
)
# 点击输出
inner_frame.get_by_role("button", name="更多").hover()
inner_frame.get_by_text("输出", exact=True).click()
# 设置行数阈值
input_box = (
inner_frame.locator("div")
.filter(has_text=re.compile(r"^行数阈值$"))
.locator("input[type='text']")
)
input_box.fill("300000")
# 步骤7下载文件
self._report_progress(
"download",
batch_index * 7 + 7,
total_batches * 7,
f"{batch_index + 1}/{total_batches} 批 - 正在下载文件...",
batch_index=batch_index + 1,
action="downloading_file",
)
download_path = f"D:/python/playwrite/data/temp_batch_{batch_index + 1}.xlsx"
with page1.expect_download() as download_info:
inner_frame.get_by_role("button", name="确定(Y)").click()
download = download_info.value
download.save_as(download_path)
self._print(f"{batch_index + 1} 批下载完成: {download_path}")
# 报告批次完成
self._report_progress(
"download",
(batch_index + 1) * 7,
total_batches * 7,
f"{batch_index + 1}/{total_batches} 批下载完成 ✓",
batch_index=batch_index + 1,
action="batch_complete",
file_path=download_path,
)
# 等待页面恢复,准备下一次查询
time.sleep(1)
return download_path
def convert_and_merge_files(self, file_paths, output_path):
"""使用 ExcelConverter 转换并合并所有文件,返回合并后的 DataFrame"""
# 确保输出文件路径是正确的格式
output_path = os.path.normpath(output_path)
output_dir = os.path.dirname(output_path)
output_filename = os.path.basename(output_path)
self._print(f"输出文件路径: {output_path}")
self._print(f"输出目录: {output_dir}")
self._print(f"输出文件名: {output_filename}")
# 步骤1检查并创建输出目录
self._report_progress(
"convert",
1,
len(file_paths) * 2 + 3,
"准备转换:检查输出目录",
action="check_directory",
)
if output_dir and not os.path.exists(output_dir):
self._print(f"创建输出目录: {output_dir}")
os.makedirs(output_dir)
all_dataframes = []
# 步骤2-N转换每个文件
for i, file_path in enumerate(file_paths, 1):
self._print(f"转换第 {i} 个文件: {file_path}")
# 报告开始转换
self._report_progress(
"convert",
1 + (i - 1) * 2 + 1,
len(file_paths) * 2 + 3,
f"正在转换文件 {i}/{len(file_paths)}",
file_index=i,
file_path=file_path,
action="converting_file",
)
df = self.converter.convert(file_path, output_file=None) # 只转换,不保存
all_dataframes.append(df)
self._print(f" 提取到 {len(df)} 条记录")
# 报告转换完成
self._report_progress(
"convert",
1 + (i - 1) * 2 + 2,
len(file_paths) * 2 + 3,
f"文件 {i}/{len(file_paths)} 转换完成({len(df)} 条记录)",
file_index=i,
record_count=len(df),
action="file_converted",
)
merged_df = None
if all_dataframes:
# 步骤N+1合并数据
self._report_progress(
"convert",
len(file_paths) * 2 + 2,
len(file_paths) * 2 + 3,
f"正在合并 {len(all_dataframes)} 个文件的数据...",
action="merging_data",
file_count=len(all_dataframes),
)
self._print(f"\n合并 {len(all_dataframes)} 个文件的数据...")
merged_df = pd.concat(all_dataframes, ignore_index=True)
merged_df.to_excel(output_path, index=False)
self._print(f"合并完成: {output_path}, 总共 {len(merged_df)} 条记录")
# 步骤N+2删除临时文件
self._report_progress(
"convert",
len(file_paths) * 2 + 3,
len(file_paths) * 2 + 3,
f"清理临时文件...",
action="cleanup",
total_records=len(merged_df),
)
for file_path in file_paths:
os.remove(file_path)
self._print(f"已删除临时文件: {file_path}")
return output_path, merged_df
return None, None
def _save_to_database(self, df: pd.DataFrame):
"""Save DataFrame to database with progress reporting"""
try:
self._report_progress(
"database", 0, 3, "准备保存到数据库...",
action="db_start"
)
stats = self.dao.save_dataframe_with_replace(df)
self._report_progress(
"database", 3, 3,
f"数据库保存完成: 删除 {stats['deleted']} 条, 新增 {stats['inserted']}",
action="db_complete",
stats=stats
)
self._print(f"\n数据库保存成功:")
self._print(f" 删除旧记录: {stats['deleted']}")
self._print(f" 新增记录: {stats['inserted']}")
except Exception as e:
self._print(f"\n警告: 数据库保存失败: {e}")
self._report_progress(
"database", 3, 3,
f"数据库保存失败: {str(e)}",
action="db_error",
error=str(e)
)
def setup_query_interface(self, inner_frame):
"""设置查询界面(不报告进度,由 extract 统一报告)"""
import re
# 打开查询界面
inner_frame.locator(".search-name-wrapper > .iconfont").click()
inner_frame.get_by_text("订单号查询").click()
# 选择"全部"标签
inner_frame.get_by_role("tab", name="全部").click()
# 填充并验证
max_retries = 3
expected_value = "5000"
for attempt in range(max_retries):
inner_frame.locator("#rc_select_0").fill(expected_value)
inner_frame.locator("#rc_select_0").press("Enter")
actual_value = inner_frame.locator("#rc_select_0").input_value()
if actual_value == expected_value:
self._print(f"文本框填充成功: {expected_value}")
break
else:
self._print(
f"{attempt + 1} 次填充失败,实际值: {actual_value},重试..."
)
if attempt == max_retries - 1:
self._print(
f"警告: {max_retries} 次尝试后仍未成功填充,继续执行..."
)
def extract(
self,
production_id_file,
data_dir="D:/python/playwrite/data",
output_file="D:/python/playwrite/data/离散备料计划维护_合并.xlsx",
debug_mode=False,
debug_batch=None,
progress_callback=None,
):
"""执行完整的数据提取流程"""
original_callback = self.progress_callback
self.progress_callback = progress_callback or self.progress_callback
try:
with sync_playwright() as playwright:
# 步骤1启动浏览器并登录
self._report_progress(
"login",
1,
3, # 保持 3 步
"启动浏览器并登录...",
action="launch_browser",
)
browser, context, page, main_frame = login(
playwright=playwright,
username=self.username,
password=self.password,
headless=self.headless,
ignore_https_errors=True,
)
# 步骤2打开功能页面
self._report_progress(
"login",
2,
3, # 保持 3 步
"登录成功,打开功能页面...",
action="open_function_page",
)
self._print("=" * 80)
self._print("开始执行离散备料计划维护数据提取")
self._print("=" * 80)
main_frame.locator("i").first.click()
with page.expect_popup() as page1_info:
main_frame.get_by_title(
"离散备料计划维护", exact=True
).first.click()
page1 = page1_info.value
main_frame = page1.locator("#forwardFrame").content_frame
inner_frame_locator = main_frame.locator("#mainiframe")
inner_frame_locator.wait_for(state="visible", timeout=15000)
inner_frame = inner_frame_locator.content_frame
# 步骤3设置查询界面
self._report_progress(
"login",
3,
3, # 保持 3 步
"配置查询界面...",
action="setup_query_interface",
)
self.setup_query_interface(inner_frame)
# 后续代码保持不变...
order_ids = self.get_production_order_numbers(
production_id_file, report_progress=True
)
downloaded_files = []
total_batches = sum(
1 for _ in self.group_order_ids(order_ids, self.batch_size)
)
for batch_index, order_ids_batch in enumerate(
self.group_order_ids(order_ids, self.batch_size)
):
self._print(
f"\n=== 开始处理第 {batch_index + 1} 批,共 {len(order_ids_batch)} 个订单号 ==="
)
downloaded_file = self.download_batch(
inner_frame,
order_ids_batch,
batch_index,
total_batches,
page1,
debug_mode=debug_mode,
debug_batch=debug_batch,
)
downloaded_files.append(downloaded_file)
self._print("\n开始执行账号注销...")
self._report_progress(
"logout",
1,
2,
"正在注销账号...",
action="logout_start",
)
logout(main_frame, verbose=self.verbose)
self._report_progress(
"logout",
2,
2,
"注销完成 ✓",
action="logout_complete",
)
if downloaded_files:
self._print(
f"\n=== 开始转换并合并 {len(downloaded_files)} 个文件 ==="
)
output_path, merged_df = self.convert_and_merge_files(downloaded_files, output_file)
# 数据库保存步骤(独立阶段)
if self.enable_db_persistence and self.dao and merged_df is not None:
self._print(f"\n=== 开始保存数据到数据库 ===")
self._save_to_database(merged_df)
else:
self._print("\n没有下载到任何文件")
self._print(f"\n=== 全部完成 ===")
self._print(f"最终文件: {output_file}")
self._report_progress(
"complete", 1, 1, "数据提取完成 ✓",
output_file=output_file,
action="all_complete",
)
context.close()
browser.close()
return output_file
finally:
# Close database connection if open
if self.dao:
try:
self.dao.__exit__(None, None, None)
except Exception:
pass
self.progress_callback = original_callback
def main():
"""测试函数"""
extractor = DiscreteMaterialPlanExtractor(
username="BLDpengqiangqiang",
password="Cqbld123456.",
headless=False,
verbose=True,
)
production_id_file = os.path.join(os.path.dirname(__file__), "productionID.txt")
output_file = "D:/python/playwrite/data/离散备料计划维护_合并.xlsx"
extractor.extract(production_id_file, output_file)
input("按回车退出...")
if __name__ == "__main__":
main()