#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
PDF数据提取类
用于提取UITNODIGING TOT BETALING文件中的关键数据
"""

import re
import os
import subprocess
from pathlib import Path


class PDFDataExtractor:
    def __init__(self, file_path):
        """
        初始化PDF提取器
        :param file_path: PDF文件路径
        """
        self.file_path = file_path
        self.pdf_text = None
        
        if not os.path.exists(file_path):
            raise Exception(f"PDF文件不存在: {file_path}")
    
    def extract_text_from_pdf(self):
        """从PDF中提取文本"""
        if self.pdf_text is not None:
            return self.pdf_text
        
        # 首先尝试使用pdftotext命令行工具
        try:
            result = subprocess.run(
                ['pdftotext', self.file_path, '-'],
                capture_output=True,
                text=True,
                check=False
            )
            if result.returncode == 0 and result.stdout:
                self.pdf_text = result.stdout
                return self.pdf_text
        except (FileNotFoundError, Exception):
            pass
        
        # 如果pdftotext不可用，尝试使用Python库
        return self.extract_using_python()
    
    def extract_using_python(self):
        """使用Python库提取PDF文本"""
        try:
            import pdfplumber
            
            text_content = ""
            with pdfplumber.open(self.file_path) as pdf:
                for page in pdf.pages:
                    page_text = page.extract_text()
                    if page_text:
                        text_content += page_text + "\n"
                    
                    # 如果直接提取失败，尝试从表格中提取
                    if not page_text or page_text.strip() == "":
                        tables = page.extract_tables()
                        if tables:
                            for table in tables:
                                for row in table:
                                    text_content += " ".join(
                                        [str(cell) if cell else "" for cell in row]
                                    ) + "\n"
            
            self.pdf_text = text_content
            return text_content
        
        except ImportError:
            raise Exception("无法提取PDF文本: pdfplumber库未安装，请运行: pip install pdfplumber")
        except Exception as e:
            raise Exception(f"无法提取PDF文本: {str(e)}")
    
    def extract_data(self):
        """提取所有关键数据"""
        text = self.extract_text_from_pdf()
        print(text)
        result = {
            'document_type': self.extract_document_type(text),
            'file_id': self.extract_file_id(text),
            'date': self.extract_date(text),
            'amount': self.extract_amount(text),
            'all_21_amount': self.extract_and_sum_taxable_amounts(text)
        }
        
        return result
    
    def extract_document_type(self, text):
        """提取文档类型"""
        if re.search(r'UITNODIGING TOT BETALING', text):
            return 'UITNODIGING TOT BETALING'
        return None
    
    def extract_file_id(self, text):
        """提取文件ID (Aangiftenummer)"""
        # 提取Aangiftenummer后的ID值，支持冒号在新行的情况
        match = re.search(r'Aangiftenummer(?:\s+.*?)*?\s*:\s*(\S+)', text, re.DOTALL)
        if match:
            return match.group(1)
        return None
    
    def extract_date(self, text):
        """提取日期 (Datum)并格式化为yyyy/mm/dd格式"""
        match = re.search(r'Datum:\s*(\d{2})/(\d{2})/(\d{4})', text)
        if match:
            # 重新排列日期部分为yyyy/mm/dd格式
            day = match.group(1)
            month = match.group(2)
            year = match.group(3)
            return f"{year}/{month}/{day}"
        return None
    
    def extract_amount(self, text):
        """提取金额 (从Bedrag:)"""
        match = re.search(r'Bedrag:\s*([\d.,]+)', text)
        if match:
            return match.group(1)
        return None
    
    def extract_and_sum_taxable_amounts(self, text=None):
        """
        提取并相加所有Tarief: 21,00%上面的Belastbare maatstaf: 后面的金额
        返回保留2位小数的总金额
        """
        # 如果没有传递text参数，则从PDF中提取
        if text is None:
            text = self.extract_text_from_pdf()
        
        # 按行分割文本
        lines = text.split('\n')
        total_amount = 0.0
        found_numbers = []
        
        for i, line in enumerate(lines):
            line = line.strip()
            
            # 查找Belastbare maatstaf行
            if re.search(r'Belastbare maatstaf:\s*(?:€\s*)?', line):
                belastbare_maatstaf_value = None
                
                # 尝试从当前行提取值
                match = re.search(r'Belastbare maatstaf:\s*(?:€\s*)?([\d.,]+)', line)
                if match:
                    belastbare_maatstaf_value = match.group(1)
                
                # 如果当前行没有值，尝试从下一行提取 - 修改正则表达式以处理欧元符号
                elif i + 1 < len(lines):
                    next_line = lines[i + 1].strip()
                    # 修改前：match = re.match(r'^\s*([\d.,]+)\s*$', next_line)
                    match = re.match(r'^\s*(?:€\s*)?([\d.,]+)\s*$', next_line)
                    if match:
                        belastbare_maatstaf_value = match.group(1)
                
                # 如果找到了值，检查接下来4行内是否有Tarief: 21,00
                if belastbare_maatstaf_value is not None:
                    tarief_found = False
                    
                    # 检查当前行、下一行、下下行、下下下行（最多4行）
                    for j in range(i, min(i + 4, len(lines))):
                        check_line = lines[j].strip()
                        if re.search(r'Tarief:\s*21,00', check_line):
                            tarief_found = True
                            break
                    
                    # 如果找到了有效的Tarief，记录这个金额
                    if tarief_found:
                        clean_number = self.parse_amount_advanced(belastbare_maatstaf_value)
                        found_numbers.append(float(clean_number))
        
        if found_numbers:
            total_amount = sum(found_numbers)
        
        # 返回的总金额，保留2位小数
        return f"{total_amount:.2f}"
    
    def parse_amount_advanced(self, amount_str):
        """
        高级金额解析 - 处理多种格式
        
        规则：
        1. 只保留最后一个分隔符(. 或 ,)作为小数点，前面的全部作为千位分隔符删除
        2. 如果最后一个分隔符后面超过2位，则判定为小数点；否则也作为小数点处理
        """
        if not amount_str:
            return '0.00'
        
        clean_str = amount_str.strip()
        
        # 移除货币符号、空格和其他特殊字符（保留数字、点、逗号和负号）
        clean_str = re.sub(r'[^\d,.\-]', '', clean_str)
        
        # 如果为空，返回0
        if not clean_str:
            return '0.00'
        
        # 找最后一个分隔符（点或逗号）的位置
        last_dot_pos = clean_str.rfind('.')
        last_comma_pos = clean_str.rfind(',')
        last_separator_pos = max(last_dot_pos, last_comma_pos)
        
        if last_separator_pos == -1:
            # 没有任何分隔符
            normalized = clean_str
        else:
            # 最后一个分隔符后面的位数
            decimals = len(clean_str) - last_separator_pos - 1
            
            if decimals == 3:
                # 正好3位：作为千位分隔符处理，移除所有分隔符
                normalized = clean_str.replace('.', '').replace(',', '')
            else:
                # 不是3位：作为小数点处理
                # 移除前面的所有分隔符，只保留最后一个作为小数点
                integer_part = clean_str[:last_separator_pos]
                decimal_part = clean_str[last_separator_pos + 1:]
                
                # 移除整数部分的所有点和逗号
                integer_part = integer_part.replace('.', '').replace(',', '')
                normalized = integer_part + '.' + decimal_part
        
        # 转换为float并格式化为2位小数
        result = float(normalized)
        return f"{result:.2f}"


# ==================== 使用示例 ====================

if __name__ == '__main__':
    try:
        # 创建提取器实例
        extractor = PDFDataExtractor('D:/phpstudy_pro/WWW/vat_api_de_client/resources/tmp/OHL26035510B_douane_wegvoering_OHL26035510B.pdf')
        
        # 提取数据
        data = extractor.extract_data()
        
        # 显示结果
        print("=========================================")
        print("PDF 数据提取结果")
        print("=========================================")
        print(f"文档类型       : {data.get('document_type', '未找到')}")
        print(f"文件ID         : {data.get('file_id', '未找到')}")
        print(f"日期           : {data.get('date', '未找到')}")
        print(f"金额           : {data.get('amount', '未找到')}")
        print(f"21%税基        : {data.get('all_21_amount', '未找到')}")
        print("=========================================")
        
        # JSON格式输出
        import json
        print("\nJSON 格式:")
        print(json.dumps(data, ensure_ascii=False, indent=2))
    
    except Exception as e:
        print(f"错误: {str(e)}")