import fitz
import os
from typing import List, Optional

class CreditReportDetector:
    """企业信用报告检测器"""
    
    def __init__(self, keywords: Optional[List[str]] = None, use_ocr: bool = False):
        """
        初始化检测器
        :param keywords: 关键词列表
        :param use_ocr: 是否使用OCR（需要安装pytesseract）
        """
        self.keywords = keywords or [
            '企业信用信息公示报告',
            '企业信用报告',
            'NATIONAL ENTERPRISE CREDIT INFORMATION PUBLICITY SYSTEM'
        ]
        self.use_ocr = use_ocr
        
        # 营业执照关键词（用于排除）
        self.business_license_keywords = [
            '营业执照',
            '统一社会信用代码',
            '注册号',
            '市场监督管理局'
        ]
    
    def is_credit_report(self, pdf_path: str, check_first_n_pages: int = 1) -> dict:
        """
        检测 PDF 是否为企业信用报告
        :param pdf_path: PDF 文件路径
        :param check_first_n_pages: 检查前N页（默认只检查第1页）
        :return: 检测结果字典
        """
        result = {
            'is_credit_report': False,
            'matched_keyword': None,
            'confidence': 0,
            'error': None,
            'text_length': 0,
            'is_scanned': False
        }
        
        # 检查文件是否存在
        if not os.path.exists(pdf_path):
            result['error'] = '文件不存在'
            return result
        
        try:
            # 打开 PDF
            doc = fitz.open(pdf_path)
            
            if len(doc) == 0:
                result['error'] = 'PDF 无内容'
                doc.close()
                return result
            
            # 提取前N页文本
            all_text = ''
            pages_to_check = min(check_first_n_pages, len(doc))
            
            for page_num in range(pages_to_check):
                page = doc[page_num]
                text = page.get_text()
                all_text += text + '\n'
            
            # 关闭文档
            doc.close()
            
            result['text_length'] = len(all_text.strip())
            
            # 如果没有提取到文本,可能是扫描版 PDF
            if result['text_length'] < 10:
                result['is_scanned'] = True
                if self.use_ocr:
                    result['error'] = '扫描版PDF，需要OCR识别（功能未实现）'
                else:
                    result['error'] = '无法提取文本，可能是扫描版PDF'
                return result
            
            # 检查企业信用报告关键词
            credit_report_score = 0
            matched_keywords = []
            
            for keyword in self.keywords:
                if keyword in all_text:
                    credit_report_score += 1
                    matched_keywords.append(keyword)
            
            # 检查营业执照关键词（排除项）
            business_license_score = 0
            for keyword in self.business_license_keywords:
                if keyword in all_text:
                    business_license_score += 1
            
            # 判断逻辑：
            # 1. 如果匹配到企业信用报告关键词，且没有匹配到太多营业执照关键词
            # 2. 或者匹配到多个企业信用报告关键词
            if credit_report_score > 0:
                if credit_report_score >= 2 or (credit_report_score >= 1 and business_license_score < 2):
                    result['is_credit_report'] = True
                    result['matched_keyword'] = ', '.join(matched_keywords)
                    result['confidence'] = min(credit_report_score / len(self.keywords), 1.0)
            
            return result
        
        except Exception as e:
            result['error'] = f'处理 PDF 出错: {str(e)}'
            return result
    
    def batch_detect(self, pdf_paths: List[str]) -> List[dict]:
        """
        批量检测多个 PDF 文件
        :param pdf_paths: PDF 文件路径列表
        :return: 检测结果列表
        """
        results = []
        for pdf_path in pdf_paths:
            result = self.is_credit_report(pdf_path)
            result['file_path'] = pdf_path
            results.append(result)
        return results


# 命令行接口
if __name__ == '__main__':
    import sys
    import json
    
    detector = CreditReportDetector()
    
    # 必须提供PDF文件路径作为参数
    if len(sys.argv) > 1:
        pdf_path = sys.argv[1]
        result = detector.is_credit_report(pdf_path)
        
        # 输出JSON格式，避免编码问题
        print(json.dumps(result, ensure_ascii=True))
    else:
        error_result = {
            'is_credit_report': False,
            'error': 'Missing PDF file path parameter'
        }
        print(json.dumps(error_result, ensure_ascii=True))
        sys.exit(1)
