add extract_vba.py for VBA code extraction and update requirements.txt

This commit is contained in:
Misaka_Company
2026-01-19 16:13:52 +08:00
parent 8f3d015ed2
commit 75c90937fa
3 changed files with 410 additions and 1 deletions

3
.gitignore vendored
View File

@@ -13,4 +13,5 @@ tmpclaude-*
*workspace* *workspace*
*.png *.png
data/ data/
Excel/ Excel/
VBA/

401
extract_vba.py Normal file
View File

@@ -0,0 +1,401 @@
"""
VBA代码提取工具
从xlsm文件中提取模块和类模块代码分类保存到VBA文件夹
自动清理Attribute信息并生成元数据JSON文件
"""
import os
import sys
import json
import re
from pathlib import Path
from typing import Dict, List, Tuple, Optional
# VBA项目相关常量
STANDARD_MODULE_DIR = "Modules"
CLASS_MODULE_DIR = "ClassModules"
DOCUMENT_MODULE_DIR = "DocumentModules"
FORMS_DIR = "Forms"
METADATA_FILE = "vba_metadata.json"
class VBAExtractor:
"""VBA代码提取器"""
def __init__(self, xlsm_path: str, output_dir: str = None):
"""
初始化VBA提取器
Args:
xlsm_path: xlsm文件路径
output_dir: 输出目录默认为项目根目录下的VBA文件夹
"""
self.xlsm_path = Path(xlsm_path)
if output_dir is None:
# 使用脚本所在目录项目根目录下的VBA文件夹
script_dir = Path(__file__).parent
self.output_dir = script_dir / "VBA"
else:
self.output_dir = Path(output_dir)
# 创建输出目录结构
self.modules_dir = self.output_dir / STANDARD_MODULE_DIR
self.class_modules_dir = self.output_dir / CLASS_MODULE_DIR
self.document_modules_dir = self.output_dir / DOCUMENT_MODULE_DIR
self.forms_dir = self.output_dir / FORMS_DIR
self.modules_dir.mkdir(parents=True, exist_ok=True)
self.class_modules_dir.mkdir(parents=True, exist_ok=True)
self.document_modules_dir.mkdir(parents=True, exist_ok=True)
self.forms_dir.mkdir(parents=True, exist_ok=True)
# 存储模块元数据
self.metadata = {
"source_file": str(self.xlsm_path),
"modules": {}
}
def parse_attributes(self, code: str) -> Tuple[Dict[str, str], str]:
"""
解析VBA代码中的Attribute信息
Args:
code: VBA代码包含Attribute行
Returns:
(attributes_dict, clean_code) - 属性字典和清理后的代码
"""
attributes = {}
lines = code.split('\n')
clean_lines = []
in_attributes = True
for line in lines:
# 检查是否为Attribute行
attr_match = re.match(r'^Attribute\s+(\w+)\s*=\s*(.+)$', line.strip())
if attr_match:
attr_name = attr_match.group(1)
attr_value = attr_match.group(2).strip().strip('"')
attributes[attr_name] = attr_value
# 继续收集Attribute暂不添加到clean_lines
continue
# 遇到非Attribute行Attribute收集结束
if not line.strip().startswith('Attribute'):
in_attributes = False
# 添加到清理后的代码跳过空行和Attribute
if not in_attributes or (line.strip() and not line.strip().startswith('Attribute')):
if not in_attributes:
clean_lines.append(line)
# 去除开头的空行
while clean_lines and not clean_lines[0].strip():
clean_lines.pop(0)
clean_code = '\n'.join(clean_lines)
return attributes, clean_code
def extract_vba_modules_olevba(self):
"""
使用olevba库提取VBA代码
需要安装: pip install oletools
"""
try:
from oletools.olevba import VBA_Parser
except ImportError:
print("错误: 未安装oletools库")
print("请运行: pip install oletools")
return False
print(f"正在解析文件: {self.xlsm_path.name}")
try:
vba_parser = VBA_Parser(str(self.xlsm_path))
if vba_parser.detect_vba_macros():
print("发现VBA代码开始提取...\n")
# 遍历所有VBA模块
for (filename, stream_path, vba_filename, vba_code) in vba_parser.extract_macros():
self._process_module(vba_filename, vba_code, stream_path)
vba_parser.close()
# 保存元数据文件
self._save_metadata()
print(f"\n提取完成!")
print(f"- 标准模块: {self.modules_dir}")
print(f"- 类模块: {self.class_modules_dir}")
print(f"- 文档模块: {self.document_modules_dir}")
print(f"- 元数据: {self.output_dir / METADATA_FILE}")
return True
else:
print("未在文件中发现VBA代码")
vba_parser.close()
return False
except Exception as e:
print(f"提取VBA代码时出错: {e}")
return False
def extract_vba_modules_com(self):
"""
使用COM接口提取VBA代码需要安装Excel
优点: 更可靠,支持更多特性
缺点: 需要安装Microsoft Excel
"""
try:
import win32com.client as win32
except ImportError:
print("错误: 未安装pywin32库")
print("请运行: pip install pywin32")
return False
print(f"正在使用COM接口解析: {self.xlsm_path.name}")
try:
excel = win32.Dispatch("Excel.Application")
excel.Visible = False
excel.DisplayAlerts = False
workbook = excel.Workbooks.Open(str(self.xlsm_path.absolute()))
# 获取VBA项目
if not workbook.VBProject:
print("错误: 无法访问VBA项目")
print("请确保: 1) Excel信任中心设置'信任对VBA工程对象模型的访问'")
print(" 2) 文件中包含VBA代码")
workbook.Close(False)
excel.Quit()
return False
vb_project = workbook.VBProject
print("开始提取VBA组件...\n")
# 遍历所有VBA组件
for component in vb_project.VBComponents:
module_name = component.Name
module_type = component.Type
# 获取代码
code_module = component.CodeModule
line_count = code_module.CountOfLines
if line_count > 0:
vba_code = code_module.Lines(1, line_count)
else:
vba_code = ""
# 根据类型分类保存
# 1=标准模块, 2=类模块, 3=MSForm, 11=Document/工作表/工作簿
type_name = {
1: STANDARD_MODULE_DIR,
2: CLASS_MODULE_DIR,
3: FORMS_DIR,
11: DOCUMENT_MODULE_DIR
}.get(module_type, "Unknown")
target_dir = {
1: self.modules_dir,
2: self.class_modules_dir,
3: self.forms_dir,
11: self.document_modules_dir
}.get(module_type, self.modules_dir)
target_dir.mkdir(exist_ok=True)
self._process_module(module_name, vba_code, type_name)
workbook.Close(False)
excel.Quit()
# 保存元数据文件
self._save_metadata()
print(f"\n提取完成!")
print(f"- 标准模块: {self.modules_dir}")
print(f"- 类模块: {self.class_modules_dir}")
print(f"- 文档模块: {self.document_modules_dir}")
print(f"- 元数据: {self.output_dir / METADATA_FILE}")
return True
except Exception as e:
print(f"使用COM提取VBA代码时出错: {e}")
print("\n提示:")
print("1. 确保已安装Microsoft Excel")
print("2. 打开Excel -> 文件 -> 选项 -> 信任中心 -> 信任中心设置")
print("3. 勾选'信任对VBA工程对象模型的访问'")
try:
excel.Quit()
except:
pass
return False
def _determine_module_type(self, module_name: str, stream_path: str) -> str:
"""
根据模块名称和流路径确定模块类型
Args:
module_name: 模块名称
stream_path: 流路径
Returns:
模块类型: Modules, ClassModules, DocumentModules
"""
name_lower = module_name.lower()
# 工作表和工作簿模块
if name_lower.startswith('sheet') or name_lower == 'thisworkbook':
return DOCUMENT_MODULE_DIR
# 标准模块
if name_lower.startswith('mod_') or name_lower.startswith('mod'):
return STANDARD_MODULE_DIR
# 类模块
if name_lower.startswith('cls') or name_lower.startswith('class'):
return CLASS_MODULE_DIR
# 根据stream_path判断
if stream_path:
path_lower = stream_path.lower()
if 'sheet' in path_lower or 'thisworkbook' in path_lower:
return DOCUMENT_MODULE_DIR
elif 'class' in path_lower or 'cls' in path_lower:
return CLASS_MODULE_DIR
# 默认为标准模块
return STANDARD_MODULE_DIR
def _process_module(self, module_name: str, vba_code: str, category: str):
"""
处理模块:解析属性、清理代码、保存文件
Args:
module_name: 模块名称
vba_code: VBA代码内容
category: 模块类别可能是stream_path或类型名称
"""
# 解析Attribute信息
attributes, clean_code = self.parse_attributes(vba_code)
# 确定实际的模块类型
module_type = self._determine_module_type(module_name, category)
# 确定目标目录
target_dir = {
STANDARD_MODULE_DIR: self.modules_dir,
CLASS_MODULE_DIR: self.class_modules_dir,
DOCUMENT_MODULE_DIR: self.document_modules_dir,
FORMS_DIR: self.forms_dir
}.get(module_type, self.modules_dir)
# 确定文件扩展名
ext = '.cls' if module_type in [CLASS_MODULE_DIR, DOCUMENT_MODULE_DIR, FORMS_DIR] else '.bas'
# 清理文件名(移除已有扩展名)
clean_name = module_name.replace('/', '_').replace('\\', '_')
# 移除已存在的扩展名
for suffix in ['.cls', '.bas', '.frm']:
if clean_name.endswith(suffix):
clean_name = clean_name[:-len(suffix)]
break
clean_name += ext
# 保存清理后的代码
file_path = target_dir / clean_name
with open(file_path, 'w', encoding='utf-8') as f:
f.write(clean_code)
# 保存元数据
self.metadata["modules"][clean_name] = {
"name": module_name,
"type": module_type,
"attributes": attributes,
"file": str(file_path.relative_to(self.output_dir))
}
print(f" [OK] 已保存: {clean_name} ({module_type})")
def _save_metadata(self):
"""保存元数据到JSON文件"""
metadata_path = self.output_dir / METADATA_FILE
with open(metadata_path, 'w', encoding='utf-8') as f:
json.dump(self.metadata, f, indent=2, ensure_ascii=False)
def main():
"""主函数"""
print("=" * 60)
print("VBA代码提取工具")
print("=" * 60)
print()
# 查找xlsm文件
excel_dir = Path("Excel")
if not excel_dir.exists():
print("错误: 未找到Excel文件夹")
return
xlsm_files = list(excel_dir.glob("*.xlsm"))
if not xlsm_files:
print("错误: Excel文件夹中没有xlsm文件")
return
# 如果有多个文件,让用户选择
if len(xlsm_files) > 1:
print("发现多个xlsm文件:")
for i, f in enumerate(xlsm_files, 1):
print(f" {i}. {f.name}")
print()
choice = input("请选择文件编号 (直接回车选择第1个): ").strip()
if not choice:
xlsm_file = xlsm_files[0]
else:
try:
idx = int(choice) - 1
xlsm_file = xlsm_files[idx]
except:
print("无效选择,使用第一个文件")
xlsm_file = xlsm_files[0]
else:
xlsm_file = xlsm_files[0]
print()
print(f"选择文件: {xlsm_file.name}")
print()
# 创建提取器
extractor = VBAExtractor(str(xlsm_file))
# 选择提取方法
print("请选择提取方法:")
print(" 1. COM接口 (推荐 - 需要安装Excel)")
print(" 2. olevba库 (不需要Excel)")
print()
method = input("请选择 (直接回车使用方法1): ").strip()
if method == "2":
print("\n使用olevba库提取...")
success = extractor.extract_vba_modules_olevba()
else:
print("\n使用COM接口提取...")
success = extractor.extract_vba_modules_com()
if success:
print("\n" + "=" * 60)
print("提取成功完成!")
print("=" * 60)
else:
print("\n" + "=" * 60)
print("提取失败")
print("=" * 60)
if __name__ == "__main__":
main()

7
requirements.txt Normal file
View File

@@ -0,0 +1,7 @@
# VBA提取工具依赖
# COM接口方法 (推荐) - 需要安装Microsoft Excel
pywin32>=306; sys_platform == 'win32'
# olevba库方法 - 不需要Excel
oletools>=0.60