#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
PDF标准化脚本
将PDF标准化为A4尺寸（210mm x 297mm 或 297mm x 210mm）
"""

import fitz  # PyMuPDF
import sys
import os
import io

# 设置标准输出编码为UTF-8（解决Windows GBK编码问题）
if sys.platform == 'win32':
    sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8')
    sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding='utf-8')


def normalize_pdf_to_a4(input_pdf, output_pdf, preserve_orientation=True):
    """
    将PDF标准化为A4尺寸
    
    参数:
        input_pdf: 输入PDF路径
        output_pdf: 输出PDF路径
        preserve_orientation: 是否保持原始方向（横向/纵向）
    
    返回:
        成功返回True，失败返回False
    """
    try:
        # A4尺寸（points，1 point = 1/72 inch）
        A4_WIDTH = 595.0  # 210mm
        A4_HEIGHT = 842.0  # 297mm
        
        # 打开输入PDF
        doc = fitz.open(input_pdf)
        
        # 创建新PDF
        new_doc = fitz.open()
        
        for page_num in range(len(doc)):
            page = doc[page_num]
            
            # 获取原始页面尺寸
            original_rect = page.rect
            original_width = original_rect.width
            original_height = original_rect.height
            
            # 判断原始方向
            is_landscape = original_width > original_height
            
            # 选择目标尺寸（保持方向）
            if preserve_orientation and is_landscape:
                # 横向A4
                target_width = A4_HEIGHT  # 842
                target_height = A4_WIDTH  # 595
            else:
                # 纵向A4
                target_width = A4_WIDTH   # 595
                target_height = A4_HEIGHT # 842
            
            # 创建新页面
            new_page = new_doc.new_page(
                width=target_width,
                height=target_height
            )
            
            # 计算缩放比例（保持宽高比，内容适配页面）
            scale_x = target_width / original_width
            scale_y = target_height / original_height
            scale = min(scale_x, scale_y)
            
            # 计算缩放后的尺寸
            scaled_width = original_width * scale
            scaled_height = original_height * scale
            
            # 计算居中位置
            x_offset = (target_width - scaled_width) / 2
            y_offset = (target_height - scaled_height) / 2
            
            # 创建目标矩形（内容放置的位置）
            content_rect = fitz.Rect(
                x_offset, 
                y_offset, 
                x_offset + scaled_width, 
                y_offset + scaled_height
            )
            
            # 将原页面内容绘制到新页面
            new_page.show_pdf_page(
                content_rect,
                doc,
                page_num
            )
        
        # 关闭原文档
        doc.close()
        
        # 保存新PDF（如果文件已存在，先尝试删除）
        if os.path.exists(output_pdf):
            try:
                os.remove(output_pdf)
            except Exception as e:
                print(f"WARNING: Cannot remove existing file: {e}", file=sys.stderr)
                # 使用临时文件名
                import tempfile
                temp_fd, temp_path = tempfile.mkstemp(suffix='.pdf', dir=os.path.dirname(output_pdf))
                os.close(temp_fd)
                output_pdf = temp_path
        
        new_doc.save(output_pdf, garbage=4, deflate=True)
        new_doc.close()
        
        print("SUCCESS: PDF normalized to A4")
        print(f"Input: {input_pdf}")
        print(f"Output: {output_pdf}")
        return True
        
    except Exception as e:
        print(f"ERROR: {str(e)}", file=sys.stderr)
        import traceback
        traceback.print_exc(file=sys.stderr)
        return False


def get_pdf_info(pdf_path):
    """获取PDF信息"""
    try:
        doc = fitz.open(pdf_path)
        page = doc[0]
        rect = page.rect
        
        print("PDF Info:")
        print(f"  Pages: {len(doc)}")
        print(f"  Page 1 size: {rect.width:.2f} x {rect.height:.2f} points")
        print(f"  Page 1 size: {rect.width/72*25.4:.2f} x {rect.height/72*25.4:.2f} mm")
        print(f"  Orientation: {'Landscape' if rect.width > rect.height else 'Portrait'}")
        
        doc.close()
        return True
    except Exception as e:
        print(f"ERROR getting PDF info: {str(e)}", file=sys.stderr)
        return False


if __name__ == "__main__":
    if len(sys.argv) < 3:
        print("Usage: python normalize_pdf.py <input_pdf> <output_pdf> [preserve_orientation]")
        print("Parameters:")
        print("  input_pdf: Input PDF file path")
        print("  output_pdf: Output PDF file path")
        print("  preserve_orientation: Preserve original orientation (1=yes, 0=no, default=1)")
        print()
        print("Examples:")
        print("  python normalize_pdf.py input.pdf output.pdf")
        print("  python normalize_pdf.py input.pdf output.pdf 1")
        sys.exit(1)
    
    input_pdf = sys.argv[1]
    output_pdf = sys.argv[2]
    preserve_orientation = True if len(sys.argv) <= 3 else (sys.argv[3] == '1')
    
    if not os.path.exists(input_pdf):
        print(f"ERROR: Input file not found: {input_pdf}", file=sys.stderr)
        sys.exit(1)
    
    print("=" * 60)
    print("PDF Normalization Tool")
    print("=" * 60)
    print()
    
    print("Input PDF Info:")
    get_pdf_info(input_pdf)
    print()
    
    print("Normalizing...")
    success = normalize_pdf_to_a4(input_pdf, output_pdf, preserve_orientation)
    
    if success:
        print()
        print("Output PDF Info:")
        get_pdf_info(output_pdf)
        print()
        print("SUCCESS: Normalization completed!")
        sys.exit(0)
    else:
        print()
        print("ERROR: Normalization failed!")
        sys.exit(1)
