#!/usr/bin/env python3
"""
提取Word模板中的占位符
"""

import sys
import zipfile
import re
from pathlib import Path

def extract_placeholders(docx_path):
    """从docx文件中提取所有占位符"""
    placeholders = set()
    
    try:
        with zipfile.ZipFile(docx_path, 'r') as zip_ref:
            # 读取document.xml
            with zip_ref.open('word/document.xml') as f:
                content = f.read().decode('utf-8')
                
                # 查找所有{{...}}格式的占位符
                # 清理XML标签
                clean_content = re.sub(r'<[^>]+>', '', content)
                matches = re.findall(r'\{\{([^}]+)\}\}', clean_content)
                placeholders.update(matches)
                
    except Exception as e:
        print(f"错误: {e}", file=sys.stderr)
        return []
    
    return sorted(placeholders)

if __name__ == '__main__':
    if len(sys.argv) < 2:
        print("用法: python extract_placeholders.py <docx文件路径>")
        sys.exit(1)
    
    docx_path = sys.argv[1]
    
    if not Path(docx_path).exists():
        print(f"文件不存在: {docx_path}", file=sys.stderr)
        sys.exit(1)
    
    placeholders = extract_placeholders(docx_path)
    
    print(f"找到 {len(placeholders)} 个占位符:\n")
    for p in placeholders:
        print(f"  {{{{{p}}}}}")
