#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
PDF数据提取类 - 使用 pdfplumber 库
用于提取: 文件ID、日期、B00金额
"""

import sys
import json
import re
from pathlib import Path

try:
    import pdfplumber
except ImportError:
    print(json.dumps({
        'code': 500,
        'msg': 'pdfplumber库未安装，请运行: pip install pdfplumber',
        'data': None
    }))
    sys.exit(1)


class PDFExtractor:
    def __init__(self, file_path):
        """
        初始化PDF提取器
        :param file_path: PDF文件路径
        """
        self.file_path = file_path
        self.pdf_text = None
        
        if not Path(file_path).exists():
            raise Exception(f"PDF文件不存在: {file_path}")

    def extract_text(self):
        """使用pdfplumber提取PDF中的所有文本"""
        if self.pdf_text is not None:
            return self.pdf_text
        
        try:
            text_content = ""
            with pdfplumber.open(self.file_path) as pdf:
                #print(f"PDF文件页数: {len(pdf.pages)}")
                for i, page in enumerate(pdf.pages):
                    page_text = page.extract_text()
                    #print(f"第{i+1}页提取结果: {'有内容' if page_text else '无内容'}")
                    if page_text:
                        text_content += page_text + "\n"
                    
                    # 如果直接提取失败，尝试从表格中提取
                    if not page_text or page_text.strip() == "":
                        tables = page.extract_tables()
                        if tables:
                            #print(f"第{i+1}页检测到表格，尝试从表格提取")
                            for table in tables:
                                for row in table:
                                    text_content += " ".join([str(cell) if cell else "" for cell in row]) + "\n"
            
            # 移除这个异常检查 - 即使文本为空也继续处理
            # 恢复异常检查 - 当文本为空时抛出异常
            if not text_content.strip():
                raise Exception("PDF提取文本为空")
            
            self.pdf_text = text_content
            return self.pdf_text
        
        except Exception as e:
            raise Exception(f"提取PDF文本失败: {str(e)}")

    def extract_data(self):
        """提取所有关键数据"""
        text = self.extract_text()
        print(f"提取的文本内容:{text}")        

        # 提取目的地并验证是否为德国(DE)
        destination = self.extract_destination(text)
        
        if destination != 'DE':
            b00_amount = 0
        else:
            b00_amount = self.extract_b00_amount(text)
        
        # 修改 extract_data 方法中的 result 构建
        result = {
            'file_id': self.extract_file_id(text),
            'date': self.extract_date(text),
            'b00_amount': f"{b00_amount:.2f}" if b00_amount is not None and b00_amount > 0 else None,
            'destination': destination  # 添加目的地到返回结果
        }
        
        return result
    
    def extract_destination(self, text):
        """提取目的地国家代码"""
        # 匹配 [13 06 ] Représentant: [3] ID DE 1 格式中的国家代码
        pattern = r'\[13\s+06\s*\]\s*Représentant:\s*\[3\]\s*ID\s+([A-Z]{2})\s*1'
        match = re.search(pattern, text, re.UNICODE)
        if match:
            return match.group(1).strip()
        
        # 匹配 [13 06 ] Représentant: [ ] ID DE 1 格式中的国家代码（空括号情况）
        pattern_empty_bracket = r'\[13\s+06\s*\]\s*Représentant:\s*\[\s*\]\s*ID\s+([A-Z]{2})\s*1'
        match_empty_bracket = re.search(pattern_empty_bracket, text, re.UNICODE)
        if match_empty_bracket:
            return match_empty_bracket.group(1).strip()
        
        # 匹配第二种格式：在 [13 05 074] Contactpersoon: IDA 行后查找国家代码
        pattern2 = r'\[13\s+05\s+074\s*\]\s*Contactpersoon:\s*[A-Za-z]+\s*\n\s*([A-Z]{2})\s+'
        match2 = re.search(pattern2, text, re.UNICODE)
        if match2:
            return match2.group(1).strip()
        
        # 尝试从 [16 04] 开始的区块中提取国家代码
        destination_from_block = self.extract_destination_from_block(text)
        if destination_from_block:
            return destination_from_block
        
        return None
    
    def extract_destination_from_block(self, text):
        """从 [16 04] 开始的区块中提取国家代码"""
        lines = text.split('\n')
        
        start_index = -1
        for i, line in enumerate(lines):
            if re.search(r'\[16\s+04\s*\]', line):
                start_index = i
                break
        
        if start_index == -1:
            return None
        
        # 从 [16 04] 开始，向下查找国家代码
        for j in range(start_index, min(start_index + 15, len(lines))):
            line = lines[j].strip()
            
            # 如果遇到新的字段标签，停止查找
            if re.match(r'^\[', line) and j > start_index:
                break
            
            # 查找两位大写字母的国家代码
            match = re.search(r'\b([A-Z]{2})\b', line)
            if match:
                return match.group(1)
        
        return None

    def extract_file_id(self, text):
        """提取文件ID (25BEH1000000O9NQR0)"""
        pattern = r'25BEH1[A-Z0-9]+'
        match = re.search(pattern, text)
        if match:
            return match.group(0)
        return None
    
    def extract_date(self, text):
        """提取日期 (07/10/2025)"""
        # 方式1: 匹配[13 01] Exporteur:后面的日期，格式为YYYYMMDD
        pattern1 = r'\[13\s+01\]\s*Exporteur:[\s\r\n]+(?:\s*[\r\n]+)*(\d{8})'
        match = re.search(pattern1, text, re.UNICODE)
        if match:
            date_str = match.group(1)
            if len(date_str) == 8:
                year = date_str[0:4]
                month = date_str[4:6]
                day = date_str[6:8]
                return f"{day}/{month}/{year}"
        
        # 方式2: 匹配[13 06 074] Personne de contact:后面的日期
        pattern2 = r'\[13\s+06\s+074\]\s*Personne\s+de\s+contact:[\s\r\n]+(?:\s*[\r\n]+)*(\d{2}/\d{2}/\d{4})'
        match = re.search(pattern2, text, re.UNICODE)
        if match:
            return match.group(1)
        
        # 方式3: 匹配[15 09 ] Date d'acceptation:后面的日期
        # 更健壮的版本
        pattern3 = r"\[15\s+09\s*\]\s*Date\s+d[\'\’]acceptation:[\s\r\n]*\s*(\d{2}/\d{2}/\d{4})"
        
        # 额外添加一个更通用的模式
        pattern5 = r"Date\s+d[\'\’]acceptation:[\s\r\n]*\s*(\d{2}/\d{2}/\d{4})"
        match = re.search(pattern5, text, re.UNICODE)
        if match:
            return match.group(1)
        
        # 方式4: 匹配[15 09] Datum van aanvaarding:后面的日期，格式为YYYYMMDD
        pattern4 = r'\[15\s+09\s*\]\s*Datum\s+van\s+aanvaarding:[\s\r\n]*(\d{8})'
        match = re.search(pattern4, text, re.UNICODE)
        if match:
            date_str = match.group(1)
            if len(date_str) == 8:
                year = date_str[0:4]
                month = date_str[4:6]
                day = date_str[6:8]
                return f"{day}/{month}/{year}"
        
        return None
    
    def extract_b00_amount(self, text):
        """提取B00后面的金额并累加"""
        # 首先检查 Total payment 格式的 B00 金额
        total_payment_pattern = r'Tax type: Base Value: Total payment: Payment method:[\s\r\n]+.*?B00\s+([\d.,]+)'    
        total_payment_match = re.search(total_payment_pattern, text, re.DOTALL | re.IGNORECASE)
        
        if total_payment_match:
            amount_str = total_payment_match.group(1)
            clean_amount = self.parse_amount_advanced(amount_str)
            if clean_amount:
                return float(clean_amount)
        
        # 如果没有 Total payment 格式，则按照原来的方式提取
        lines = text.split('\n')
        total_amount = 0.0
        
        for i, line in enumerate(lines):
            current_line = line.strip()            
            # 直接搜索B00后面的金额（大小写不敏感）
            if re.search(r'\bB00\b', current_line, re.IGNORECASE):                
                # 从同一行提取B00后面的金额
                # 匹配: B00 + 可选空格 + 金额数字
                b00_match = re.search(r'\bB00\s+([\d.,]+)', current_line, re.IGNORECASE)
                
                if b00_match:
                    amount_str = b00_match.group(1)
                    clean_amount = self.parse_amount_advanced(amount_str)
                    if clean_amount:
                        total_amount += float(clean_amount)
        
        return total_amount if total_amount > 0 else None
    
    def parse_amount_advanced(self, amount_str):
        """高级金额解析 - 处理多种格式"""
        if not amount_str:
            return '0.00'
        
        clean_str = amount_str.strip()
        
        # 移除货币符号、空格和其他特殊字符
        clean_str = re.sub(r'[^\d,.\-]', '', clean_str)
        
        if not clean_str:
            return '0.00'
        
        # 找最后一个分隔符
        last_dot_pos = clean_str.rfind('.')
        last_comma_pos = clean_str.rfind(',')
        last_separator_pos = max(last_dot_pos, last_comma_pos)
        
        if last_separator_pos == -1:
            normalized = clean_str
        else:
            decimals = len(clean_str) - last_separator_pos - 1
            
            if decimals == 3:
                # 千位分隔符，移除所有分隔符
                normalized = clean_str.replace('.', '').replace(',', '')
            else:
                # 小数点处理
                integer_part = clean_str[:last_separator_pos]
                decimal_part = clean_str[last_separator_pos + 1:]
                
                integer_part = integer_part.replace('.', '').replace(',', '')
                normalized = integer_part + '.' + decimal_part
        
        result = float(normalized)
        return f"{result:.2f}"


def main():
    """主函数"""
    pdf_path = 'D:/phpstudy_pro/WWW/vat_api_de_client/resources/tmp/7-10383910+税单.pdf'
    import os
    pdf_path = os.path.normpath(pdf_path)

    try:
        extractor = PDFExtractor(pdf_path)
        data = extractor.extract_data()
        
        # 更严格的检查逻辑
        if data['destination'] != 'DE':
            raise Exception("PDF文件识别失败: 目的地不是德国(DE)")
        
        if data['b00_amount'] == 0:
            raise Exception("PDF文件识别失败: 无法提取B00金额")
        
        if data['file_id'] is None and data['date'] is None:
            raise Exception("PDF文件识别失败: 无法提取文件ID和日期")
        
        response = {
            'code': 200,
            'msg': 'success',
            'data': {
                'file_id': data['file_id'],
                'date': data['date'],
                'b00_amount': data['b00_amount']
            }
        }
        
        print(json.dumps(response, ensure_ascii=False))
        sys.exit(0)
    
    except Exception as e:
        response = {
            'code': 500,
            'msg': f'错误: {str(e)}',
            'data': None
        }
        print(json.dumps(response, ensure_ascii=False))
        sys.exit(1)


if __name__ == '__main__':
    main()