import warnings

# 检测器契约：stdout 只输出一行 JSON。第三方库（PyMuPDF 等）的警告/弃用提示
# 会被 PHP 端 exec 捕获混入 stdout，导致 json_decode 失败（生产曾出现"无法解析Python脚本输出"），
# 故在导入第三方库前统一抑制
warnings.filterwarnings('ignore')

import fitz
import os
import unicodedata
from typing import List, Optional

class CreditReportDetector:
    """企业信用报告检测器"""

    def __init__(self, keywords: Optional[List[str]] = None, use_ocr: bool = False):
        """
        初始化检测器
        :param keywords: 关键词列表
        :param use_ocr: 是否使用OCR（需要安装pytesseract）
        """
        self.keywords = keywords or [
            '企业信用信息公示报告',
            '企业信用报告',
            'NATIONAL ENTERPRISE CREDIT INFORMATION PUBLICITY SYSTEM'
        ]
        self.use_ocr = use_ocr

        # 营业执照关键词（用于排除）
        self.business_license_keywords = [
            '营业执照',
            '统一社会信用代码',
            '注册号',
            '市场监督管理局'
        ]

    def _is_text_readable(self, text: str) -> bool:
        """判断提取的文本是否包含可读内容（而非字体字形索引等乱码）"""
        if not text.strip():
            return False
        real_chars = [c for c in text if c.isalpha() or ord(c) > 127]
        ratio = len(real_chars) / max(len(text), 1)
        return ratio >= 0.3

    def _has_type3_fonts(self, doc: fitz.Document, check_first_n_pages: int = 1) -> bool:
        """检查PDF是否使用了Type3自定义字体"""
        pages_to_check = min(check_first_n_pages, len(doc))
        for page_num in range(pages_to_check):
            for font_info in doc[page_num].get_fonts():
                # font_info: (xref, name, type, encoding, ...)
                if font_info[2] == 'Type3':
                    return True
        return False

    def is_credit_report(self, pdf_path: str, check_first_n_pages: int = 2) -> dict:
        """
        检测 PDF 是否为企业信用报告
        :param pdf_path: PDF 文件路径
        :param check_first_n_pages: 检查前N页（默认前2页），逐页独立判断，任一页命中即判定。
               不合并多页文本，避免信用报告正文页(含统一社会信用代码/注册号等营业执照排除词)
               压掉标题页(含企业信用报告关键词)的命中。
        :return: 检测结果字典
        """
        result = {
            'is_credit_report': False,
            'matched_keyword': None,
            'confidence': 0,
            'error': None,
            'text_length': 0,
            'is_scanned': False
        }

        # 检查文件是否存在
        if not os.path.exists(pdf_path):
            result['error'] = '文件不存在'
            return result

        try:
            # 打开 PDF
            doc = fitz.open(pdf_path)

            if len(doc) == 0:
                result['error'] = 'PDF 无内容'
                doc.close()
                return result

            # 检查前N页是否使用Type3字体
            has_type3 = self._has_type3_fonts(doc, check_first_n_pages)

            # 逐页提取文本并独立判断（第3页及以后不检查）
            pages_to_check = min(check_first_n_pages, len(doc))
            scanned_pages = 0

            for page_num in range(pages_to_check):
                page = doc[page_num]
                page_text = page.get_text()

                # NFKC 归一化：将 CJK 兼容表意文字转为标准形式
                # 例如 ⽤(U+2F64)→用(U+7528), ⽰(U+2F70)→示(U+793A)
                page_text = unicodedata.normalize('NFKC', page_text)

                result['text_length'] += len(page_text.strip())

                # 判断该页文本是否可读（区分真实文本与Type3字形索引乱码）
                if len(page_text.strip()) < 10 or not self._is_text_readable(page_text):
                    scanned_pages += 1
                    continue

                # 检查该页是否包含企业信用报告关键词
                credit_report_score = 0
                matched_keywords = []

                for keyword in self.keywords:
                    if keyword in page_text:
                        credit_report_score += 1
                        matched_keywords.append(keyword)

                # 检查该页是否包含营业执照关键词（排除项）
                business_license_score = 0
                for keyword in self.business_license_keywords:
                    if keyword in page_text:
                        business_license_score += 1

                # 判断逻辑：
                # 1. 如果匹配到企业信用报告关键词，且没有匹配到太多营业执照关键词
                # 2. 或者匹配到多个企业信用报告关键词
                if credit_report_score > 0:
                    if credit_report_score >= 2 or (credit_report_score >= 1 and business_license_score < 2):
                        result['is_credit_report'] = True
                        result['matched_keyword'] = ', '.join(matched_keywords)
                        result['confidence'] = min(credit_report_score / len(self.keywords), 1.0)
                        break

            # 关闭文档
            doc.close()

            # 检查的前N页全部不可读，标记为扫描版
            if scanned_pages >= pages_to_check:
                result['is_scanned'] = True
                if has_type3:
                    result['error'] = 'Type3字体PDF，无法提取可读文本，需要OCR识别'
                elif self.use_ocr:
                    result['error'] = '扫描版PDF，需要OCR识别（功能未实现）'
                else:
                    result['error'] = '无法提取有效文本，可能是扫描版PDF'

            return result

        except Exception as e:
            result['error'] = f'处理 PDF 出错: {str(e)}'
            return result

    def batch_detect(self, pdf_paths: List[str]) -> List[dict]:
        """
        批量检测多个 PDF 文件
        :param pdf_paths: PDF 文件路径列表
        :return: 检测结果列表
        """
        results = []
        for pdf_path in pdf_paths:
            result = self.is_credit_report(pdf_path)
            result['file_path'] = pdf_path
            results.append(result)
        return results


# 命令行接口
if __name__ == '__main__':
    import sys
    import json

    detector = CreditReportDetector()

    # 必须提供PDF文件路径作为参数
    if len(sys.argv) > 1:
        pdf_path = sys.argv[1]
        # 只检查前两页：信用报告的标题与登记机关均在前两页，
        # 避免长PDF后部内容（如拼接文件）干扰判断
        result = detector.is_credit_report(pdf_path, check_first_n_pages=2)

        # 输出JSON格式，避免编码问题
        print(json.dumps(result, ensure_ascii=True))
    else:
        error_result = {
            'is_credit_report': False,
            'error': 'Missing PDF file path parameter'
        }
        print(json.dumps(error_result, ensure_ascii=True))
        sys.exit(1)