Rename utility files to use English names for better cross-platform compatibility: - utils/离散备料计划维护数据提取.py → utils/discrete_material_plan_extractor.py - utils/离散备料计划维护数据清理.py → utils/discrete_material_plan_cleaner.py Update all import statements across the codebase and documentation references. Co-Authored-By: Claude Sonnet 4.5 <noreply@anthropic.com>
540 lines
20 KiB
Python
540 lines
20 KiB
Python
"""
|
||
物料状态校验工具
|
||
校验订单中的物料状态,匹配待删除物料
|
||
|
||
支持两种数据源:
|
||
1. Excel 文件(原有方式)
|
||
2. 数据库驱动(新增方式)
|
||
"""
|
||
|
||
import os
|
||
import pandas as pd
|
||
from typing import List, Dict, Any, Optional, Set
|
||
from dataclasses import dataclass
|
||
from utils.discrete_material_plan_extractor import DiscreteMaterialPlanExtractor
|
||
from db.materials_to_delete import get_all_materials_to_delete
|
||
from db.production_contract_data_dao import ProductionContractDataDAO
|
||
from db.discrete_material_plan_dao import DiscreteMaterialPlanDAO
|
||
from db.materials_to_be_deleted_dao import MaterialsTypeToBeDeletedDAO
|
||
from db.materials_to_be_deleted_records_dao import MaterialsToBeDeletedDAO
|
||
|
||
|
||
# ==================== DATA STRUCTURES ====================
|
||
|
||
@dataclass
|
||
class MaterialValidationResult:
|
||
"""Enhanced material validation result with complete record information"""
|
||
material_name: str
|
||
material_code: str
|
||
specification: Optional[str] = None
|
||
model: Optional[str] = None
|
||
manager_name: Optional[str] = None
|
||
is_marked_for_deletion: bool = False
|
||
matched_type_keyword: Optional[str] = None # Matched keyword from MaterialsTypeToBeDeleted
|
||
|
||
|
||
class MaterialStatusValidator:
|
||
"""物料状态校验器"""
|
||
|
||
def __init__(self, username, password, headless=False, verbose=True):
|
||
"""
|
||
初始化校验器
|
||
|
||
Args:
|
||
username: ERP系统用户名
|
||
password: ERP系统密码
|
||
headless: 是否无头模式运行浏览器
|
||
verbose: 是否打印详细日志
|
||
"""
|
||
self.username = username
|
||
self.password = password
|
||
self.headless = headless
|
||
self.verbose = verbose
|
||
|
||
def _print(self, *args, **kwargs):
|
||
"""打印日志(如果 verbose=True)"""
|
||
if self.verbose:
|
||
print(*args, **kwargs)
|
||
|
||
def extract_material_names(self, excel_file: str) -> List[str]:
|
||
"""
|
||
从Excel文件的Q列提取材料名称并去重
|
||
|
||
Args:
|
||
excel_file: Excel文件路径
|
||
|
||
Returns:
|
||
List[str]: 去重后的材料名称列表
|
||
"""
|
||
df = pd.read_excel(excel_file)
|
||
# R列索引为17(Python从0开始)
|
||
material_names = df.iloc[:, 17].dropna().unique().tolist()
|
||
# 确保所有元素都是字符串类型
|
||
material_names = [str(name) for name in material_names]
|
||
return material_names
|
||
|
||
def match_materials(
|
||
self, material_names: List[str], db_materials: List[Dict[str, Any]]
|
||
) -> List[Dict[str, Any]]:
|
||
"""
|
||
匹配材料名称
|
||
|
||
Args:
|
||
material_names: Excel中的材料名称列表
|
||
db_materials: 数据库中的物料记录列表
|
||
|
||
Returns:
|
||
List[Dict]: 匹配结果
|
||
"""
|
||
results = []
|
||
for material_name in material_names:
|
||
matched = None
|
||
for db_record in db_materials:
|
||
# 如果数据库的MaterialName出现在Excel的材料名称中
|
||
if db_record["MaterialName"] in material_name:
|
||
matched = db_record
|
||
break
|
||
|
||
results.append(
|
||
{
|
||
"材料名称": material_name,
|
||
"匹配的MaterialName": matched["MaterialName"] if matched else None,
|
||
"负责人": matched["ManagerName"] if matched else None,
|
||
"匹配状态": "匹配成功" if matched else "未匹配",
|
||
}
|
||
)
|
||
|
||
return results
|
||
|
||
def validate(
|
||
self,
|
||
production_id_file: str,
|
||
merged_excel_file: str = None,
|
||
output_file: str = None,
|
||
) -> str:
|
||
"""
|
||
执行完整的校验流程
|
||
|
||
Args:
|
||
production_id_file: ProductionID.txt文件路径
|
||
merged_excel_file: 合并后的Excel文件路径(可选)
|
||
output_file: 输出文件路径(可选)
|
||
|
||
Returns:
|
||
输出文件路径
|
||
"""
|
||
# 设置默认文件路径
|
||
if merged_excel_file is None:
|
||
merged_excel_file = "D:/python/playwrite/data/离散备料计划维护_合并.xlsx"
|
||
if output_file is None:
|
||
output_file = "D:/python/playwrite/data/物料状态校验结果.xlsx"
|
||
|
||
# 1. 调用数据提取工具
|
||
self._print("=" * 60)
|
||
self._print("步骤1: 提取备料计划数据...")
|
||
extractor = DiscreteMaterialPlanExtractor(
|
||
username=self.username,
|
||
password=self.password,
|
||
headless=self.headless,
|
||
verbose=self.verbose,
|
||
)
|
||
extractor.extract(production_id_file, output_file=merged_excel_file)
|
||
self._print(f"数据提取完成: {merged_excel_file}")
|
||
|
||
# 2. 提取材料名称
|
||
self._print("\n步骤2: 提取材料名称...")
|
||
material_names = self.extract_material_names(merged_excel_file)
|
||
self._print(f"提取到 {len(material_names)} 个唯一材料名称")
|
||
|
||
# 3. 从数据库获取待删除物料
|
||
self._print("\n步骤3: 从数据库获取待删除物料...")
|
||
db_materials = get_all_materials_to_delete()
|
||
self._print(f"获取到 {len(db_materials)} 条待删除物料记录")
|
||
|
||
# 4. 匹配物料
|
||
self._print("\n步骤4: 匹配物料...")
|
||
results = self.match_materials(material_names, db_materials)
|
||
|
||
# 5. 输出结果
|
||
self._print("\n步骤5: 输出结果...")
|
||
result_df = pd.DataFrame(results)
|
||
result_df.to_excel(output_file, index=False)
|
||
self._print(f"结果已保存: {output_file}")
|
||
|
||
# 打印统计信息
|
||
matched_count = sum(1 for r in results if r["匹配状态"] == "匹配成功")
|
||
self._print(f"\n统计信息:")
|
||
self._print(f" 总材料数: {len(results)}")
|
||
self._print(f" 匹配成功: {matched_count}")
|
||
self._print(f" 未匹配: {len(results) - matched_count}")
|
||
|
||
return output_file
|
||
|
||
def validate_from_existing_excel(
|
||
self, excel_file: str, output_file: str = None
|
||
) -> str:
|
||
"""
|
||
从已存在的Excel文件执行校验(不需要重新提取数据)
|
||
|
||
Args:
|
||
excel_file: 已存在的Excel文件路径
|
||
output_file: 输出文件路径(可选)
|
||
|
||
Returns:
|
||
输出文件路径
|
||
"""
|
||
# 设置默认输出路径
|
||
if output_file is None:
|
||
output_file = "D:/python/playwrite/data/物料状态校验结果.xlsx"
|
||
|
||
self._print("=" * 60)
|
||
self._print("从已存在的Excel文件执行校验...")
|
||
|
||
# 1. 提取材料名称
|
||
self._print("\n步骤1: 提取材料名称...")
|
||
material_names = self.extract_material_names(excel_file)
|
||
self._print(f"提取到 {len(material_names)} 个唯一材料名称")
|
||
|
||
# 2. 从数据库获取待删除物料
|
||
self._print("\n步骤2: 从数据库获取待删除物料...")
|
||
db_materials = get_all_materials_to_delete()
|
||
self._print(f"获取到 {len(db_materials)} 条待删除物料记录")
|
||
|
||
# 3. 匹配物料
|
||
self._print("\n步骤3: 匹配物料...")
|
||
results = self.match_materials(material_names, db_materials)
|
||
|
||
# 4. 输出结果
|
||
self._print("\n步骤4: 输出结果...")
|
||
result_df = pd.DataFrame(results)
|
||
result_df.to_excel(output_file, index=False)
|
||
self._print(f"结果已保存: {output_file}")
|
||
|
||
# 打印统计信息
|
||
matched_count = sum(1 for r in results if r["匹配状态"] == "匹配成功")
|
||
self._print(f"\n统计信息:")
|
||
self._print(f" 总材料数: {len(results)}")
|
||
self._print(f" 匹配成功: {matched_count}")
|
||
self._print(f" 未匹配: {len(results) - matched_count}")
|
||
|
||
return output_file
|
||
|
||
# ==================== DATABASE-DRIVEN VALIDATION ====================
|
||
|
||
def _read_production_ids(self, production_id_file: str) -> List[str]:
|
||
"""
|
||
读取 ProductionID.txt 文件
|
||
|
||
Args:
|
||
production_id_file: ProductionID.txt 文件路径
|
||
|
||
Returns:
|
||
List[str]: 总排号列表
|
||
"""
|
||
with open(production_id_file, 'r', encoding='utf-8') as f:
|
||
production_ids = [line.strip() for line in f if line.strip()]
|
||
return production_ids
|
||
|
||
def _get_source_numbers_from_production_ids(
|
||
self, production_ids: List[str]
|
||
) -> List[str]:
|
||
"""
|
||
通过 ProductionID 查询获取 SourceNumber 列表
|
||
|
||
查询链路:
|
||
ProductionID (总排号) -> productionContractData.26年压力表合同数据.生产订单号
|
||
|
||
Args:
|
||
production_ids: 总排号列表
|
||
|
||
Returns:
|
||
List[str]: 生产订单号列表
|
||
"""
|
||
self._print(f"[INFO] 正在查询 {len(production_ids)} 个总排号对应的生产订单号...")
|
||
|
||
contract_dao = ProductionContractDataDAO()
|
||
source_numbers = contract_dao.get_source_numbers_by_总排号(production_ids)
|
||
|
||
self._print(f"[INFO] 找到 {len(source_numbers)} 个唯一的生产订单号")
|
||
return source_numbers
|
||
|
||
def _get_material_names_from_db(
|
||
self, source_numbers: List[str] = None
|
||
) -> List[str]:
|
||
"""
|
||
从数据库获取材料名称,可选按 SourceNumber 过滤
|
||
|
||
Args:
|
||
source_numbers: 可选的 SourceNumber 列表进行过滤
|
||
|
||
Returns:
|
||
List[str]: 唯一材料名称列表
|
||
"""
|
||
if source_numbers is None or not source_numbers:
|
||
self._print("[INFO] 查询所有材料的名称...")
|
||
else:
|
||
self._print(f"[INFO] 查询 {len(source_numbers)} 个生产订单对应的材料名称...")
|
||
|
||
dao = DiscreteMaterialPlanDAO()
|
||
material_names = dao.get_unique_material_names(source_numbers)
|
||
|
||
self._print(f"[INFO] 找到 {len(material_names)} 个唯一材料名称")
|
||
return material_names
|
||
|
||
def validate_from_database(
|
||
self,
|
||
production_id_file: str = None,
|
||
full_table: bool = False,
|
||
output_file: str = None
|
||
) -> str:
|
||
"""
|
||
使用数据库作为数据源执行校验
|
||
|
||
支持两种模式:
|
||
1. 全表校验 (full_table=True): 查询整个 DiscreteMaterialPlanData 表
|
||
2. ProductionID 过滤校验 (production_id_file 指定): 基于 ProductionID.txt 文件过滤
|
||
|
||
查询链路(模式2):
|
||
ProductionID.txt (总排号)
|
||
-> productionContractData.26年压力表合同数据.生产订单号 (SourceNumber)
|
||
-> DiscreteMaterialPlanData.SourceNumber
|
||
-> DiscreteMaterialPlanData.MaterialName
|
||
-> 对比 MaterialsTypeToBeDeleted.MaterialName
|
||
|
||
Args:
|
||
production_id_file: ProductionID.txt 路径(模式2)
|
||
full_table: 是否全表校验(模式1)
|
||
output_file: 输出文件路径
|
||
|
||
Returns:
|
||
输出文件路径
|
||
"""
|
||
# 设置默认输出路径
|
||
if output_file is None:
|
||
output_file = "D:/python/playwrite/data/物料状态校验结果.xlsx"
|
||
|
||
self._print("=" * 60)
|
||
self._print("使用数据库数据源执行校验...")
|
||
|
||
# 根据模式获取材料名称
|
||
if full_table:
|
||
self._print("\n模式: 全表校验")
|
||
self._print("[INFO] 查询 DiscreteMaterialPlanData 表中的所有材料...")
|
||
material_names = self._get_material_names_from_db(None)
|
||
elif production_id_file:
|
||
self._print("\n模式: ProductionID 过滤校验")
|
||
self._print(f"[INFO] 读取 ProductionID 文件: {production_id_file}")
|
||
|
||
# 1. 读取 ProductionID.txt
|
||
production_ids = self._read_production_ids(production_id_file)
|
||
self._print(f"[INFO] 读取到 {len(production_ids)} 个总排号")
|
||
|
||
# 2. 查询获取 SourceNumbers
|
||
source_numbers = self._get_source_numbers_from_production_ids(production_ids)
|
||
|
||
# 3. 获取材料名称
|
||
material_names = self._get_material_names_from_db(source_numbers)
|
||
else:
|
||
raise ValueError(
|
||
"必须指定 full_table=True 或提供 production_id_file 参数"
|
||
)
|
||
|
||
# 从数据库获取待删除物料
|
||
self._print("\n从数据库获取待删除物料...")
|
||
db_materials = get_all_materials_to_delete()
|
||
self._print(f"获取到 {len(db_materials)} 条待删除物料记录")
|
||
|
||
# 匹配物料
|
||
self._print("\n匹配物料...")
|
||
results = self.match_materials(material_names, db_materials)
|
||
|
||
# 输出结果
|
||
self._print("\n输出结果...")
|
||
result_df = pd.DataFrame(results)
|
||
result_df.to_excel(output_file, index=False)
|
||
self._print(f"结果已保存: {output_file}")
|
||
|
||
# 打印统计信息
|
||
matched_count = sum(1 for r in results if r["匹配状态"] == "匹配成功")
|
||
self._print(f"\n统计信息:")
|
||
self._print(f" 总材料数: {len(results)}")
|
||
self._print(f" 匹配成功: {matched_count}")
|
||
self._print(f" 未匹配: {len(results) - matched_count}")
|
||
|
||
return output_file
|
||
|
||
# ==================== ENHANCED VALIDATION METHODS ====================
|
||
|
||
def match_materials_detailed(
|
||
self,
|
||
material_records: List[Dict[str, Any]],
|
||
type_keywords: List[Dict[str, Any]],
|
||
marked_codes_dict: Dict[str, str] # Changed: MaterialCode -> ManagerName mapping
|
||
) -> List[MaterialValidationResult]:
|
||
"""
|
||
Match materials with detailed information.
|
||
|
||
Args:
|
||
material_records: Complete records from DiscreteMaterialPlanData
|
||
type_keywords: Records from MaterialsTypeToBeDeleted (MaterialName matching)
|
||
marked_codes_dict: MaterialCode -> ManagerName mapping from MaterialsToBeDeleted
|
||
|
||
Returns:
|
||
List of MaterialValidationResult objects
|
||
"""
|
||
results = []
|
||
|
||
for record in material_records:
|
||
material_name = record.get('MaterialName', '') or ''
|
||
material_code = record.get('MaterialCode', '') or ''
|
||
specification = record.get('Specification', '') or None
|
||
model = record.get('Model', '') or None
|
||
|
||
# Priority 1: Check MaterialsToBeDeleted (MaterialCode exact match)
|
||
# This has highest priority - if MaterialCode exists, use its ManagerName
|
||
manager_name = marked_codes_dict.get(material_code) if material_code else None
|
||
is_marked = manager_name is not None
|
||
matched_keyword = None
|
||
|
||
# Priority 2: If not in MaterialsToBeDeleted, match with MaterialsTypeToBeDeleted
|
||
# (MaterialName contains match)
|
||
if not manager_name:
|
||
for type_record in type_keywords:
|
||
type_material_name = type_record.get('MaterialName', '')
|
||
if type_material_name and type_material_name in material_name:
|
||
matched_keyword = type_material_name
|
||
manager_name = type_record.get('ManagerName')
|
||
break
|
||
|
||
result = MaterialValidationResult(
|
||
material_name=material_name,
|
||
material_code=material_code,
|
||
specification=specification,
|
||
model=model,
|
||
manager_name=manager_name,
|
||
is_marked_for_deletion=is_marked,
|
||
matched_type_keyword=matched_keyword
|
||
)
|
||
results.append(result)
|
||
|
||
return results
|
||
|
||
def validate_from_database_enhanced(
|
||
self,
|
||
production_id_file: str = None,
|
||
full_table: bool = False,
|
||
output_file: str = None
|
||
) -> tuple:
|
||
"""
|
||
Enhanced database validation with complete record information.
|
||
|
||
Args:
|
||
production_id_file: ProductionID.txt path (for filtered mode)
|
||
full_table: Whether to query full table (for full table mode)
|
||
output_file: Output Excel file path
|
||
|
||
Returns:
|
||
Tuple of (output_file_path, List[MaterialValidationResult])
|
||
"""
|
||
# Set default output path
|
||
if output_file is None:
|
||
output_file = "D:/python/playwrite/data/物料状态校验结果.xlsx"
|
||
|
||
self._print("=" * 60)
|
||
self._print("使用增强数据库校验(完整记录模式)...")
|
||
|
||
# Get material records (complete records, not just MaterialName)
|
||
if full_table:
|
||
self._print("\n模式: 全表校验")
|
||
self._print("[INFO] 查询 DiscreteMaterialPlanData 表中的所有完整记录(启用 MaterialCode 去重)...")
|
||
dao = DiscreteMaterialPlanDAO()
|
||
|
||
# Get original count for deduplication statistics
|
||
original_count = dao.count_all()
|
||
|
||
material_records = dao.query_all_distinct_by_material_code()
|
||
dedup_count = original_count - len(material_records)
|
||
|
||
self._print(f"[INFO] 获取到 {len(material_records)} 条记录")
|
||
if dedup_count > 0:
|
||
self._print(f"[INFO] 基于 MaterialCode 去重:移除了 {dedup_count} 条重复记录")
|
||
elif production_id_file:
|
||
self._print("\n模式: ProductionID 过滤校验")
|
||
self._print(f"[INFO] 读取 ProductionID 文件: {production_id_file}")
|
||
|
||
# 1. Read ProductionID.txt
|
||
production_ids = self._read_production_ids(production_id_file)
|
||
self._print(f"[INFO] 读取到 {len(production_ids)} 个总排号")
|
||
|
||
# 2. Query SourceNumbers
|
||
source_numbers = self._get_source_numbers_from_production_ids(production_ids)
|
||
|
||
# 3. Get complete material records with deduplication
|
||
self._print(f"[INFO] 查询 {len(source_numbers)} 个生产订单对应的完整物料记录(启用 MaterialCode 去重)...")
|
||
dao = DiscreteMaterialPlanDAO()
|
||
|
||
# Get original count for deduplication statistics
|
||
original_records = dao.query_by_source_numbers(source_numbers)
|
||
|
||
material_records = dao.query_by_source_numbers_distinct(source_numbers)
|
||
dedup_count = len(original_records) - len(material_records)
|
||
|
||
self._print(f"[INFO] 获取到 {len(material_records)} 条记录")
|
||
if dedup_count > 0:
|
||
self._print(f"[INFO] 基于 MaterialCode 去重:移除了 {dedup_count} 条重复记录")
|
||
else:
|
||
raise ValueError("必须指定 full_table=True 或提供 production_id_file 参数")
|
||
|
||
# Get type keywords from MaterialsTypeToBeDeleted
|
||
self._print("\n从数据库获取待删除物料类型...")
|
||
type_dao = MaterialsTypeToBeDeletedDAO()
|
||
type_keywords = type_dao.get_all_materials()
|
||
self._print(f"获取到 {len(type_keywords)} 条物料类型记录")
|
||
|
||
# Get marked material codes and manager names from MaterialsToBeDeleted
|
||
self._print("从数据库获取已标记删除的物料记录...")
|
||
record_dao = MaterialsToBeDeletedDAO()
|
||
marked_records = record_dao.get_all_records()
|
||
|
||
# Build dictionary: MaterialCode -> ManagerName
|
||
marked_codes_dict = {
|
||
r['MaterialCode']: r['ManagerName']
|
||
for r in marked_records
|
||
if r.get('MaterialCode') and r.get('ManagerName')
|
||
}
|
||
self._print(f"获取到 {len(marked_codes_dict)} 个已标记的物料代码")
|
||
|
||
# Match materials
|
||
self._print("\n匹配物料...")
|
||
results = self.match_materials_detailed(material_records, type_keywords, marked_codes_dict)
|
||
|
||
# Output to Excel
|
||
self._print("\n输出结果...")
|
||
|
||
# Convert to DataFrame for Excel export
|
||
df_data = []
|
||
for r in results:
|
||
df_data.append({
|
||
"材料名称": r.material_name,
|
||
"材料代码": r.material_code,
|
||
"规格": r.specification or '',
|
||
"型号": r.model or '',
|
||
"负责人": r.manager_name or '',
|
||
"已标记删除": "是" if r.is_marked_for_deletion else "否",
|
||
"匹配的关键词": r.matched_type_keyword or ''
|
||
})
|
||
|
||
result_df = pd.DataFrame(df_data)
|
||
result_df.to_excel(output_file, index=False)
|
||
self._print(f"结果已保存: {output_file}")
|
||
|
||
# Print statistics
|
||
marked_count = sum(1 for r in results if r.is_marked_for_deletion)
|
||
matched_count = sum(1 for r in results if r.manager_name)
|
||
self._print(f"\n统计信息:")
|
||
self._print(f" 总记录数: {len(results)}")
|
||
self._print(f" 匹配到负责人: {matched_count}")
|
||
self._print(f" 已标记删除: {marked_count}")
|
||
|
||
return output_file, results
|