add extract_vba.py for VBA code extraction and update requirements.txt
This commit is contained in:
1
.gitignore
vendored
1
.gitignore
vendored
@@ -14,3 +14,4 @@ tmpclaude-*
|
||||
*.png
|
||||
data/
|
||||
Excel/
|
||||
VBA/
|
||||
401
extract_vba.py
Normal file
401
extract_vba.py
Normal file
@@ -0,0 +1,401 @@
|
||||
"""
|
||||
VBA代码提取工具
|
||||
从xlsm文件中提取模块和类模块代码,分类保存到VBA文件夹
|
||||
自动清理Attribute信息并生成元数据JSON文件
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Tuple, Optional
|
||||
|
||||
# VBA项目相关常量
|
||||
STANDARD_MODULE_DIR = "Modules"
|
||||
CLASS_MODULE_DIR = "ClassModules"
|
||||
DOCUMENT_MODULE_DIR = "DocumentModules"
|
||||
FORMS_DIR = "Forms"
|
||||
METADATA_FILE = "vba_metadata.json"
|
||||
|
||||
|
||||
class VBAExtractor:
|
||||
"""VBA代码提取器"""
|
||||
|
||||
def __init__(self, xlsm_path: str, output_dir: str = None):
|
||||
"""
|
||||
初始化VBA提取器
|
||||
|
||||
Args:
|
||||
xlsm_path: xlsm文件路径
|
||||
output_dir: 输出目录,默认为项目根目录下的VBA文件夹
|
||||
"""
|
||||
self.xlsm_path = Path(xlsm_path)
|
||||
if output_dir is None:
|
||||
# 使用脚本所在目录(项目根目录)下的VBA文件夹
|
||||
script_dir = Path(__file__).parent
|
||||
self.output_dir = script_dir / "VBA"
|
||||
else:
|
||||
self.output_dir = Path(output_dir)
|
||||
|
||||
# 创建输出目录结构
|
||||
self.modules_dir = self.output_dir / STANDARD_MODULE_DIR
|
||||
self.class_modules_dir = self.output_dir / CLASS_MODULE_DIR
|
||||
self.document_modules_dir = self.output_dir / DOCUMENT_MODULE_DIR
|
||||
self.forms_dir = self.output_dir / FORMS_DIR
|
||||
|
||||
self.modules_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.class_modules_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.document_modules_dir.mkdir(parents=True, exist_ok=True)
|
||||
self.forms_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# 存储模块元数据
|
||||
self.metadata = {
|
||||
"source_file": str(self.xlsm_path),
|
||||
"modules": {}
|
||||
}
|
||||
|
||||
def parse_attributes(self, code: str) -> Tuple[Dict[str, str], str]:
|
||||
"""
|
||||
解析VBA代码中的Attribute信息
|
||||
|
||||
Args:
|
||||
code: VBA代码(包含Attribute行)
|
||||
|
||||
Returns:
|
||||
(attributes_dict, clean_code) - 属性字典和清理后的代码
|
||||
"""
|
||||
attributes = {}
|
||||
lines = code.split('\n')
|
||||
clean_lines = []
|
||||
in_attributes = True
|
||||
|
||||
for line in lines:
|
||||
# 检查是否为Attribute行
|
||||
attr_match = re.match(r'^Attribute\s+(\w+)\s*=\s*(.+)$', line.strip())
|
||||
if attr_match:
|
||||
attr_name = attr_match.group(1)
|
||||
attr_value = attr_match.group(2).strip().strip('"')
|
||||
attributes[attr_name] = attr_value
|
||||
# 继续收集Attribute,暂不添加到clean_lines
|
||||
continue
|
||||
|
||||
# 遇到非Attribute行,Attribute收集结束
|
||||
if not line.strip().startswith('Attribute'):
|
||||
in_attributes = False
|
||||
|
||||
# 添加到清理后的代码(跳过空行和Attribute)
|
||||
if not in_attributes or (line.strip() and not line.strip().startswith('Attribute')):
|
||||
if not in_attributes:
|
||||
clean_lines.append(line)
|
||||
|
||||
# 去除开头的空行
|
||||
while clean_lines and not clean_lines[0].strip():
|
||||
clean_lines.pop(0)
|
||||
|
||||
clean_code = '\n'.join(clean_lines)
|
||||
return attributes, clean_code
|
||||
|
||||
def extract_vba_modules_olevba(self):
|
||||
"""
|
||||
使用olevba库提取VBA代码
|
||||
|
||||
需要安装: pip install oletools
|
||||
"""
|
||||
try:
|
||||
from oletools.olevba import VBA_Parser
|
||||
except ImportError:
|
||||
print("错误: 未安装oletools库")
|
||||
print("请运行: pip install oletools")
|
||||
return False
|
||||
|
||||
print(f"正在解析文件: {self.xlsm_path.name}")
|
||||
|
||||
try:
|
||||
vba_parser = VBA_Parser(str(self.xlsm_path))
|
||||
|
||||
if vba_parser.detect_vba_macros():
|
||||
print("发现VBA代码,开始提取...\n")
|
||||
|
||||
# 遍历所有VBA模块
|
||||
for (filename, stream_path, vba_filename, vba_code) in vba_parser.extract_macros():
|
||||
self._process_module(vba_filename, vba_code, stream_path)
|
||||
|
||||
vba_parser.close()
|
||||
|
||||
# 保存元数据文件
|
||||
self._save_metadata()
|
||||
|
||||
print(f"\n提取完成!")
|
||||
print(f"- 标准模块: {self.modules_dir}")
|
||||
print(f"- 类模块: {self.class_modules_dir}")
|
||||
print(f"- 文档模块: {self.document_modules_dir}")
|
||||
print(f"- 元数据: {self.output_dir / METADATA_FILE}")
|
||||
return True
|
||||
else:
|
||||
print("未在文件中发现VBA代码")
|
||||
vba_parser.close()
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
print(f"提取VBA代码时出错: {e}")
|
||||
return False
|
||||
|
||||
def extract_vba_modules_com(self):
|
||||
"""
|
||||
使用COM接口提取VBA代码(需要安装Excel)
|
||||
|
||||
优点: 更可靠,支持更多特性
|
||||
缺点: 需要安装Microsoft Excel
|
||||
"""
|
||||
try:
|
||||
import win32com.client as win32
|
||||
except ImportError:
|
||||
print("错误: 未安装pywin32库")
|
||||
print("请运行: pip install pywin32")
|
||||
return False
|
||||
|
||||
print(f"正在使用COM接口解析: {self.xlsm_path.name}")
|
||||
|
||||
try:
|
||||
excel = win32.Dispatch("Excel.Application")
|
||||
excel.Visible = False
|
||||
excel.DisplayAlerts = False
|
||||
|
||||
workbook = excel.Workbooks.Open(str(self.xlsm_path.absolute()))
|
||||
|
||||
# 获取VBA项目
|
||||
if not workbook.VBProject:
|
||||
print("错误: 无法访问VBA项目")
|
||||
print("请确保: 1) Excel信任中心设置'信任对VBA工程对象模型的访问'")
|
||||
print(" 2) 文件中包含VBA代码")
|
||||
workbook.Close(False)
|
||||
excel.Quit()
|
||||
return False
|
||||
|
||||
vb_project = workbook.VBProject
|
||||
|
||||
print("开始提取VBA组件...\n")
|
||||
|
||||
# 遍历所有VBA组件
|
||||
for component in vb_project.VBComponents:
|
||||
module_name = component.Name
|
||||
module_type = component.Type
|
||||
|
||||
# 获取代码
|
||||
code_module = component.CodeModule
|
||||
line_count = code_module.CountOfLines
|
||||
|
||||
if line_count > 0:
|
||||
vba_code = code_module.Lines(1, line_count)
|
||||
else:
|
||||
vba_code = ""
|
||||
|
||||
# 根据类型分类保存
|
||||
# 1=标准模块, 2=类模块, 3=MSForm, 11=Document/工作表/工作簿
|
||||
type_name = {
|
||||
1: STANDARD_MODULE_DIR,
|
||||
2: CLASS_MODULE_DIR,
|
||||
3: FORMS_DIR,
|
||||
11: DOCUMENT_MODULE_DIR
|
||||
}.get(module_type, "Unknown")
|
||||
|
||||
target_dir = {
|
||||
1: self.modules_dir,
|
||||
2: self.class_modules_dir,
|
||||
3: self.forms_dir,
|
||||
11: self.document_modules_dir
|
||||
}.get(module_type, self.modules_dir)
|
||||
|
||||
target_dir.mkdir(exist_ok=True)
|
||||
self._process_module(module_name, vba_code, type_name)
|
||||
|
||||
workbook.Close(False)
|
||||
excel.Quit()
|
||||
|
||||
# 保存元数据文件
|
||||
self._save_metadata()
|
||||
|
||||
print(f"\n提取完成!")
|
||||
print(f"- 标准模块: {self.modules_dir}")
|
||||
print(f"- 类模块: {self.class_modules_dir}")
|
||||
print(f"- 文档模块: {self.document_modules_dir}")
|
||||
print(f"- 元数据: {self.output_dir / METADATA_FILE}")
|
||||
return True
|
||||
|
||||
except Exception as e:
|
||||
print(f"使用COM提取VBA代码时出错: {e}")
|
||||
print("\n提示:")
|
||||
print("1. 确保已安装Microsoft Excel")
|
||||
print("2. 打开Excel -> 文件 -> 选项 -> 信任中心 -> 信任中心设置")
|
||||
print("3. 勾选'信任对VBA工程对象模型的访问'")
|
||||
try:
|
||||
excel.Quit()
|
||||
except:
|
||||
pass
|
||||
return False
|
||||
|
||||
def _determine_module_type(self, module_name: str, stream_path: str) -> str:
|
||||
"""
|
||||
根据模块名称和流路径确定模块类型
|
||||
|
||||
Args:
|
||||
module_name: 模块名称
|
||||
stream_path: 流路径
|
||||
|
||||
Returns:
|
||||
模块类型: Modules, ClassModules, DocumentModules
|
||||
"""
|
||||
name_lower = module_name.lower()
|
||||
|
||||
# 工作表和工作簿模块
|
||||
if name_lower.startswith('sheet') or name_lower == 'thisworkbook':
|
||||
return DOCUMENT_MODULE_DIR
|
||||
|
||||
# 标准模块
|
||||
if name_lower.startswith('mod_') or name_lower.startswith('mod'):
|
||||
return STANDARD_MODULE_DIR
|
||||
|
||||
# 类模块
|
||||
if name_lower.startswith('cls') or name_lower.startswith('class'):
|
||||
return CLASS_MODULE_DIR
|
||||
|
||||
# 根据stream_path判断
|
||||
if stream_path:
|
||||
path_lower = stream_path.lower()
|
||||
if 'sheet' in path_lower or 'thisworkbook' in path_lower:
|
||||
return DOCUMENT_MODULE_DIR
|
||||
elif 'class' in path_lower or 'cls' in path_lower:
|
||||
return CLASS_MODULE_DIR
|
||||
|
||||
# 默认为标准模块
|
||||
return STANDARD_MODULE_DIR
|
||||
|
||||
def _process_module(self, module_name: str, vba_code: str, category: str):
|
||||
"""
|
||||
处理模块:解析属性、清理代码、保存文件
|
||||
|
||||
Args:
|
||||
module_name: 模块名称
|
||||
vba_code: VBA代码内容
|
||||
category: 模块类别(可能是stream_path或类型名称)
|
||||
"""
|
||||
# 解析Attribute信息
|
||||
attributes, clean_code = self.parse_attributes(vba_code)
|
||||
|
||||
# 确定实际的模块类型
|
||||
module_type = self._determine_module_type(module_name, category)
|
||||
|
||||
# 确定目标目录
|
||||
target_dir = {
|
||||
STANDARD_MODULE_DIR: self.modules_dir,
|
||||
CLASS_MODULE_DIR: self.class_modules_dir,
|
||||
DOCUMENT_MODULE_DIR: self.document_modules_dir,
|
||||
FORMS_DIR: self.forms_dir
|
||||
}.get(module_type, self.modules_dir)
|
||||
|
||||
# 确定文件扩展名
|
||||
ext = '.cls' if module_type in [CLASS_MODULE_DIR, DOCUMENT_MODULE_DIR, FORMS_DIR] else '.bas'
|
||||
|
||||
# 清理文件名(移除已有扩展名)
|
||||
clean_name = module_name.replace('/', '_').replace('\\', '_')
|
||||
# 移除已存在的扩展名
|
||||
for suffix in ['.cls', '.bas', '.frm']:
|
||||
if clean_name.endswith(suffix):
|
||||
clean_name = clean_name[:-len(suffix)]
|
||||
break
|
||||
clean_name += ext
|
||||
|
||||
# 保存清理后的代码
|
||||
file_path = target_dir / clean_name
|
||||
with open(file_path, 'w', encoding='utf-8') as f:
|
||||
f.write(clean_code)
|
||||
|
||||
# 保存元数据
|
||||
self.metadata["modules"][clean_name] = {
|
||||
"name": module_name,
|
||||
"type": module_type,
|
||||
"attributes": attributes,
|
||||
"file": str(file_path.relative_to(self.output_dir))
|
||||
}
|
||||
|
||||
print(f" [OK] 已保存: {clean_name} ({module_type})")
|
||||
|
||||
def _save_metadata(self):
|
||||
"""保存元数据到JSON文件"""
|
||||
metadata_path = self.output_dir / METADATA_FILE
|
||||
with open(metadata_path, 'w', encoding='utf-8') as f:
|
||||
json.dump(self.metadata, f, indent=2, ensure_ascii=False)
|
||||
|
||||
|
||||
def main():
|
||||
"""主函数"""
|
||||
print("=" * 60)
|
||||
print("VBA代码提取工具")
|
||||
print("=" * 60)
|
||||
print()
|
||||
|
||||
# 查找xlsm文件
|
||||
excel_dir = Path("Excel")
|
||||
if not excel_dir.exists():
|
||||
print("错误: 未找到Excel文件夹")
|
||||
return
|
||||
|
||||
xlsm_files = list(excel_dir.glob("*.xlsm"))
|
||||
if not xlsm_files:
|
||||
print("错误: Excel文件夹中没有xlsm文件")
|
||||
return
|
||||
|
||||
# 如果有多个文件,让用户选择
|
||||
if len(xlsm_files) > 1:
|
||||
print("发现多个xlsm文件:")
|
||||
for i, f in enumerate(xlsm_files, 1):
|
||||
print(f" {i}. {f.name}")
|
||||
print()
|
||||
choice = input("请选择文件编号 (直接回车选择第1个): ").strip()
|
||||
if not choice:
|
||||
xlsm_file = xlsm_files[0]
|
||||
else:
|
||||
try:
|
||||
idx = int(choice) - 1
|
||||
xlsm_file = xlsm_files[idx]
|
||||
except:
|
||||
print("无效选择,使用第一个文件")
|
||||
xlsm_file = xlsm_files[0]
|
||||
else:
|
||||
xlsm_file = xlsm_files[0]
|
||||
|
||||
print()
|
||||
print(f"选择文件: {xlsm_file.name}")
|
||||
print()
|
||||
|
||||
# 创建提取器
|
||||
extractor = VBAExtractor(str(xlsm_file))
|
||||
|
||||
# 选择提取方法
|
||||
print("请选择提取方法:")
|
||||
print(" 1. COM接口 (推荐 - 需要安装Excel)")
|
||||
print(" 2. olevba库 (不需要Excel)")
|
||||
print()
|
||||
|
||||
method = input("请选择 (直接回车使用方法1): ").strip()
|
||||
|
||||
if method == "2":
|
||||
print("\n使用olevba库提取...")
|
||||
success = extractor.extract_vba_modules_olevba()
|
||||
else:
|
||||
print("\n使用COM接口提取...")
|
||||
success = extractor.extract_vba_modules_com()
|
||||
|
||||
if success:
|
||||
print("\n" + "=" * 60)
|
||||
print("提取成功完成!")
|
||||
print("=" * 60)
|
||||
else:
|
||||
print("\n" + "=" * 60)
|
||||
print("提取失败")
|
||||
print("=" * 60)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
7
requirements.txt
Normal file
7
requirements.txt
Normal file
@@ -0,0 +1,7 @@
|
||||
# VBA提取工具依赖
|
||||
|
||||
# COM接口方法 (推荐) - 需要安装Microsoft Excel
|
||||
pywin32>=306; sys_platform == 'win32'
|
||||
|
||||
# olevba库方法 - 不需要Excel
|
||||
oletools>=0.60
|
||||
Reference in New Issue
Block a user