"""
香港公司查册文件检测器
用于检测 PDF 是否为香港公司查册文件 (Company Particulars)
"""

import fitz
import os
import sys
import json
from typing import List, Optional


class CompanyParticularsDetector:
    """香港公司查册文件检测器"""

    def __init__(self, keywords: Optional[List[str]] = None, use_ocr: bool = False):
        """
        初始化检测器
        :param keywords: 关键词列表
        :param use_ocr: 是否使用OCR（需要安装pytesseract）
        """
        # 查册文件关键词（中文）
        self.chinese_keywords = keywords or [
            '公司资料',
            '公司详情',
            '公司注册证书',
            '注册办事处地址',
            '公司编号',
            '股本',
            '董事',
            '公司秘书',
            '成立日期',
            'Company Particulars'
        ]

        # 查册文件关键词（英文）
        self.english_keywords = [
            'Company Particulars',
            'Company Number',
            'Registered Office Address',
            'Date of Incorporation',
            'Share Capital',
            'Directors',
            'Company Secretary',
            'Incorporation',
            'Companies Registry',
            '香港公司注册处'
        ]

        self.use_ocr = use_ocr

        # 排除关键词（用于排除其他类型的文件）
        self.exclude_keywords = [
            'Certificate of Incorporation',  # CR证书本身
            'Business Registration',  # BR证书本身
            '营业执照',
            '统一社会信用代码',
            '企业信用报告',
            '企业信用信息公示报告'
        ]

    def is_company_particulars(self, pdf_path: str, check_first_n_pages: int = 2) -> dict:
        """
        检测 PDF 是否为香港公司查册文件
        :param pdf_path: PDF 文件路径
        :param check_first_n_pages: 检查前N页（默认检查前2页）
        :return: 检测结果字典
        """
        result = {
            'is_company_particulars': False,
            'matched_keywords': [],
            'confidence': 0,
            'error': None,
            'text_length': 0,
            'is_scanned': False,
            'language': None  # 'zh', 'en', or 'mixed'
        }

        # 检查文件是否存在
        if not os.path.exists(pdf_path):
            result['error'] = '文件不存在'
            return result

        try:
            # 打开 PDF
            doc = fitz.open(pdf_path)

            if len(doc) == 0:
                result['error'] = 'PDF 无内容'
                doc.close()
                return result

            # 提取前N页文本，同时统计图片页数
            all_text = ''
            image_only_pages = 0
            pages_to_check = min(check_first_n_pages, len(doc))

            for page_num in range(pages_to_check):
                page = doc[page_num]
                text = page.get_text()
                all_text += text + '\n'
                # 检测纯图片页（有图片但无文本）
                images = page.get_images()
                if len(images) > 0 and len(text.strip()) == 0:
                    image_only_pages += 1

            result['text_length'] = len(all_text.strip())
            result['image_only_pages'] = image_only_pages

            # 关闭文档
            doc.close()

            # 检测扫描版 PDF：
            # 1. 文本极少（<50字符），说明前N页几乎没可提取文本
            # 2. 所有检查页都是纯图片无文本（完全扫描版）
            # 注意：不用 image_only_pages > 0，避免误判混合PDF（如封面图+文本正文）
            #       那种情况 keyword 匹配阶段可以正常处理
            if result['text_length'] < 50 or image_only_pages >= pages_to_check:
                result['is_scanned'] = True
                if self.use_ocr:
                    result['error'] = '扫描版PDF，需要OCR识别（功能未实现）'
                else:
                    result['error'] = '无法提取文本，可能是扫描版PDF'
                return result

            # 检查排除关键词
            exclude_score = 0
            for keyword in self.exclude_keywords:
                if keyword in all_text:
                    exclude_score += 1

            # 如果排除关键词匹配过多，可能不是查册文件
            if exclude_score >= 2:
                result['error'] = '文件类型不符合查册文件特征（检测到排除关键词）'
                return result

            # 检查中文关键词
            chinese_score = 0
            matched_chinese = []
            for keyword in self.chinese_keywords:
                if keyword in all_text:
                    chinese_score += 1
                    matched_chinese.append(keyword)

            # 检查英文关键词
            english_score = 0
            matched_english = []
            for keyword in self.english_keywords:
                if keyword in all_text:
                    english_score += 1
                    matched_english.append(keyword)

            # 判断语言类型
            if chinese_score > 0 and english_score > 0:
                result['language'] = 'mixed'
            elif chinese_score > 0:
                result['language'] = 'zh'
            elif english_score > 0:
                result['language'] = 'en'

            # 计算总分
            total_score = chinese_score + english_score
            all_matched = matched_chinese + matched_english

            # 判断逻辑：
            # 1. 匹配到 "Company Particulars" 或 "公司资料" 直接判定为查册文件
            # 2. 或者匹配到多个关键词（>=3个）
            has_primary_keyword = (
                'Company Particulars' in all_text or
                '公司资料' in all_text or
                '公司详情' in all_text
            )

            if has_primary_keyword:
                result['is_company_particulars'] = True
                result['matched_keywords'] = all_matched
                result['confidence'] = 1.0
            elif total_score >= 3:
                result['is_company_particulars'] = True
                result['matched_keywords'] = all_matched
                result['confidence'] = min(total_score / 5.0, 1.0)
            elif total_score >= 2 and exclude_score == 0:
                # 匹配到2个关键词且没有排除关键词，可能需要进一步确认
                result['is_company_particulars'] = True
                result['matched_keywords'] = all_matched
                result['confidence'] = 0.6

            return result

        except Exception as e:
            result['error'] = f'处理 PDF 出错: {str(e)}'
            return result

    def batch_detect(self, pdf_paths: List[str]) -> List[dict]:
        """
        批量检测多个 PDF 文件
        :param pdf_paths: PDF 文件路径列表
        :return: 检测结果列表
        """
        results = []
        for pdf_path in pdf_paths:
            result = self.is_company_particulars(pdf_path)
            result['file_path'] = pdf_path
            results.append(result)
        return results


# 命令行接口
if __name__ == '__main__':
    detector = CompanyParticularsDetector()

    # 必须提供PDF文件路径作为参数
    if len(sys.argv) > 1:
        pdf_path = sys.argv[1]
        result = detector.is_company_particulars(pdf_path)

        # 输出JSON格式，使用ASCII编码避免Windows命令行编码问题
        print(json.dumps(result, ensure_ascii=True))
    else:
        error_result = {
            'is_company_particulars': False,
            'error': 'Missing PDF file path parameter'
        }
        print(json.dumps(error_result, ensure_ascii=True))
        sys.exit(1)
