"""
查册文件检测器 对抗性测试套件
测试 python/company_particulars_detector.py 的各种场景
"""
import json
import sys
import os
import io
import tempfile

# Add project python dir to path
sys.path.insert(0, os.path.join(os.path.dirname(os.path.dirname(__file__)), 'python'))
from company_particulars_detector import CompanyParticularsDetector

import fitz
from PIL import Image

PASS = 0
FAIL = 0
detector = CompanyParticularsDetector()

def test(name, condition_fn):
    """Run a test case, print PASS/FAIL"""
    global PASS, FAIL
    try:
        condition_fn()
        print(f'  [PASS] {name}')
        PASS += 1
    except AssertionError as e:
        print(f'  [FAIL] {name}: {e}')
        FAIL += 1
    except Exception as e:
        print(f'  [ERROR] {name}: {type(e).__name__}: {e}')
        FAIL += 1


# CJK font path for Chinese text tests (PyMuPDF default font doesn't support CJK)
_CJK_FONT_PATH = None
for _fp in ['C:/Windows/Fonts/simsun.ttc', 'C:/Windows/Fonts/msyh.ttc']:
    if os.path.exists(_fp):
        _CJK_FONT_PATH = _fp
        break

def make_text_pdf(pages_content, use_cjk_font=False):
    """
    Create a temp PDF. pages_content is a list where each element is either:
    - str: text to insert on the page
    - list: list of (x, y, text, fontsize) tuples
    Returns temp file path.
    """
    fd, path = tempfile.mkstemp(suffix='.pdf')
    os.close(fd)
    doc = fitz.open()
    font_kwargs = {}
    if use_cjk_font and _CJK_FONT_PATH:
        font_kwargs = {'fontname': 'CJK', 'fontfile': _CJK_FONT_PATH}
    for content in pages_content:
        page = doc.new_page()
        if isinstance(content, str):
            page.insert_text((50, 50), content, fontsize=12, **font_kwargs)
        elif isinstance(content, list):
            for item in content:
                x, y, text = item[0], item[1], item[2]
                fs = item[3] if len(item) > 3 else 12
                page.insert_text((x, y), text, fontsize=fs, **font_kwargs)
        else:
            # Empty page
            pass
    doc.save(path)
    doc.close()
    return path


def make_image_page():
    """Create a page with only an embedded image, no text"""
    img = Image.new('RGB', (100, 100), color='white')
    buf = io.BytesIO()
    img.save(buf, format='PNG')
    return buf.getvalue()


def make_image_pdf(num_pages=2):
    """Create a PDF with only image pages (no text)"""
    fd, path = tempfile.mkstemp(suffix='.pdf')
    os.close(fd)
    doc = fitz.open()
    for _ in range(num_pages):
        page = doc.new_page()
        img_data = make_image_page()
        rect = fitz.Rect(0, 0, 100, 100)
        page.insert_image(rect, stream=img_data)
    doc.save(path)
    doc.close()
    return path


def cleanup(*paths):
    for p in paths:
        try:
            if os.path.exists(p):
                os.unlink(p)
        except Exception:
            pass


# ============================================================
print('=' * 60)
print('查册文件检测器 对抗性测试套件')
print('=' * 60)

# -------------------------------------------------------
# Test 1: User's scanned PDF
# -------------------------------------------------------
def t1():
    r = detector.is_company_particulars('查册文件-CompanyParticulars.pdf')
    assert r['is_scanned'] == True, f'Expected scanned=True, got {r["is_scanned"]}'
    assert r['text_length'] < 50, f'Expected text_length<50, got {r["text_length"]}'
    assert r.get('image_only_pages', 0) > 0, f'Expected image_only_pages>0, got {r.get("image_only_pages")}'
    assert 'image_only_pages' in r, 'Missing image_only_pages key in result'
test('扫描版查册文件 (user PDF): is_scanned=true, low text, has image pages', t1)

# -------------------------------------------------------
# Test 2: Non-existent file
# -------------------------------------------------------
def t2():
    r = detector.is_company_particulars('__nonexistent_file__.pdf')
    assert r['is_company_particulars'] == False
    assert '不存在' in r.get('error', '')
test('不存在的文件: error contains 文件不存在', t2)

# -------------------------------------------------------
# Test 3: Text-based 查册文件 with keywords
# -------------------------------------------------------
def t3():
    path = make_text_pdf([[
        (50, 50, 'Company Particulars Report', 14),
        (50, 100, 'Company Number: 12345678', 12),
        (50, 140, 'Registered Office Address: Room 123, Building A, HK', 12),
        (50, 180, 'Date of Incorporation: 2020-01-01', 12),
        (50, 220, 'Directors: John Doe, Jane Smith', 12),
        (50, 260, 'Share Capital: HKD 10,000', 12),
        (50, 300, 'Company Secretary: ABC Secretarial Ltd', 12),
        (50, 350, 'Padding text to ensure >50 chars total for proper detection flow', 10),
    ]])
    try:
        r = detector.is_company_particulars(path)
        assert r['is_company_particulars'] == True, f'Expected True, got {r}'
        assert r['confidence'] >= 0.6, f'Confidence too low: {r["confidence"]}'
        assert len(r['matched_keywords']) >= 2, f'Too few matched: {r["matched_keywords"]}'
    finally:
        cleanup(path)
test('文本型查册文件 (含 Company Particulars 等关键词): detected=true', t3)

# -------------------------------------------------------
# Test 4: Non-查册文件 (regular text PDF)
# -------------------------------------------------------
def t4():
    path = make_text_pdf([[
        (50, 50, 'Invoice Report', 14),
        (50, 100, 'This is a regular invoice document for payment', 12),
        (50, 140, 'Total amount: $1000.00 due in 30 days', 12),
        (50, 180, 'Customer: ABC Corp, Address: 123 Main St', 12),
        (50, 220, 'Extra padding to make sure we have more than 50 chars', 10),
    ]])
    try:
        r = detector.is_company_particulars(path)
        assert r['is_company_particulars'] == False, f'Expected False, got {r}'
        assert r['is_scanned'] == False, f'Should not be scanned, got {r["is_scanned"]}'
    finally:
        cleanup(path)
test('非查册文件 (普通发票文本): detected=false, not scanned', t4)

# -------------------------------------------------------
# Test 5: Fully scanned PDF (all pages image-only)
# -------------------------------------------------------
def t5():
    path = make_image_pdf(2)
    try:
        r = detector.is_company_particulars(path)
        assert r['is_scanned'] == True, f'Expected scanned=True, got {r["is_scanned"]}'
        assert r.get('image_only_pages', 0) == 2, f'Expected 2 image-only pages, got {r.get("image_only_pages")}'
        assert r['text_length'] == 0, f'Expected 0 text, got {r["text_length"]}'
    finally:
        cleanup(path)
test('全扫描PDF (2页纯图片无文本): is_scanned=true, image_only_pages=2', t5)

# -------------------------------------------------------
# Test 6: Edge case - 49 chars of noise text
# -------------------------------------------------------
def t6():
    path = make_text_pdf(['ABCD EFGH IJKL MNOP QRST UVWX YZab cdef ghij klmn'])  # ~49 chars
    try:
        r = detector.is_company_particulars(path)
        # Should be flagged as scanned (text too short)
        assert r['is_scanned'] == True, f'49 chars should be scanned, got {r["is_scanned"]}'
    finally:
        cleanup(path)
test('边界值: 49字符噪音文本 -> is_scanned=true', t6)

# -------------------------------------------------------
# Test 7: Edge case - 50 chars of text, no keywords
# -------------------------------------------------------
def t7():
    # exactly 50 chars of noise
    path = make_text_pdf(['ABCD EFGH IJKL MNOP QRST UVWX YZab cdef ghij klmn op'])
    try:
        r = detector.is_company_particulars(path)
        assert r['is_scanned'] == False, f'50 chars should NOT be scanned, got is_scanned={r["is_scanned"]}'
        assert r['is_company_particulars'] == False, 'No keywords, should be false'
    finally:
        cleanup(path)
test('边界值: 50字符噪音文本 -> not scanned, not company particulars', t7)

# -------------------------------------------------------
# Test 8: Mixed PDF (page1 image-only, page2 text with keywords)
# -------------------------------------------------------
def t8():
    fd, path = tempfile.mkstemp(suffix='.pdf')
    os.close(fd)
    doc = fitz.open()
    # Page 1: image only (封面)
    page1 = doc.new_page()
    img_data = make_image_page()
    rect = fitz.Rect(0, 0, 100, 100)
    page1.insert_image(rect, stream=img_data)
    # Page 2: text with 查册文件 keywords
    page2 = doc.new_page()
    page2.insert_text((50, 50), 'Company Particulars', fontsize=14)
    page2.insert_text((50, 100), 'Company Number: 99999999', fontsize=12)
    page2.insert_text((50, 150), 'Registered Office: Kowloon Tower, Flat B', fontsize=12)
    page2.insert_text((50, 200), 'Padding text to exceed fifty characters total across pages', fontsize=10)
    doc.save(path)
    doc.close()
    try:
        r = detector.is_company_particulars(path)
        # IMPORTANT: should NOT be flagged as scanned despite page1 being image-only
        # because page2 has extractable text with keywords
        assert r['is_scanned'] == False, f'FAIL: mixed PDF should NOT be scanned (page2 has text). Got is_scanned={r["is_scanned"]}'
        assert r['is_company_particulars'] == True, f'FAIL: should detect keywords on page2. Got {r["is_company_particulars"]}'
        assert 'Company Particulars' in r.get('matched_keywords', []), f'Expected Company Particulars in matched: {r.get("matched_keywords")}'
    finally:
        cleanup(path)
test('混合PDF (封面图+文本正文): NOT scanned, keywords detected on page2', t8)

# -------------------------------------------------------
# Test 9: Exclude keywords (CR证书)
# -------------------------------------------------------
def t9():
    path = make_text_pdf([[
        (50, 50, 'Certificate of Incorporation', 14),
        (50, 100, 'Business Registration Certificate', 12),
        (50, 150, 'Company Number: 11111111', 12),
        (50, 200, 'This is a certificate of incorporation document for a Hong Kong company', 10),
        (50, 250, 'Extra text padding to exceed fifty characters to avoid the scanned detection', 10),
    ]])
    try:
        r = detector.is_company_particulars(path)
        assert r['is_company_particulars'] == False, f'CR should be excluded, got {r["is_company_particulars"]}'
        assert r.get('error'), 'Should have exclusion error message'
    finally:
        cleanup(path)
test('排除关键词: CR+BR 证书被正确排除', t9)

# -------------------------------------------------------
# Test 10: CLI JSON output format
# -------------------------------------------------------
def t10():
    import subprocess
    result = subprocess.run(
        ['python', 'python/company_particulars_detector.py', '查册文件-CompanyParticulars.pdf'],
        capture_output=True, text=True, cwd=os.path.dirname(os.path.dirname(__file__))
    )
    output = json.loads(result.stdout)
    required_keys = [
        'is_company_particulars', 'matched_keywords', 'confidence',
        'error', 'text_length', 'is_scanned', 'language', 'image_only_pages'
    ]
    for k in required_keys:
        assert k in output, f'Missing key "{k}" in CLI output'
    # Verify the critical field for PHP fallback
    assert output['is_scanned'] == True, 'CLI: scanned PDF should have is_scanned=true'
    assert output['image_only_pages'] >= 1, 'CLI: should have image_only_pages >= 1'
test(f'CLI模式JSON输出: 所有 {8} 个必要字段存在, is_scanned=true', t10)

# -------------------------------------------------------
# Test 11: Single keyword "公司资料" (Chinese)
# -------------------------------------------------------
def t11():
    path = make_text_pdf([[
        (50, 50, '公司资料', 14),
        (50, 100, '公司编号: 88888888', 12),
        (50, 150, '注册办事处地址: 香港中环皇后大道1号', 12),
        (50, 200, 'Additional text to ensure the total character count exceeds fifty for proper non-scanned detection', 10),
    ]], use_cjk_font=True)
    try:
        r = detector.is_company_particulars(path)
        assert r['is_company_particulars'] == True, f'Chinese 公司资料 should match, got {r["is_company_particulars"]}'
        assert r['confidence'] >= 0.9, f'Confidence should be high for primary keyword: {r["confidence"]}'
    finally:
        cleanup(path)
test('中文查册文件 (含"公司资料"主关键词): detected=true, high confidence', t11)

# -------------------------------------------------------
# Test 12: 2 secondary keywords but no primary (>=3 needed for secondary-only)
# -------------------------------------------------------
def t12():
    path = make_text_pdf([[
        (50, 50, 'Company Number: 77777777', 12),
        (50, 100, 'Date of Incorporation: 2019-06-15', 12),
        (50, 150, 'Some unrelated business document text for filler content to exceed the fifty character minimum threshold', 10),
    ]])
    try:
        r = detector.is_company_particulars(path)
        # total_score = 2 (Company Number + Date of Incorporation)
        # has_primary_keyword = False
        # total_score < 3, so not detected (3-keyword threshold not met)
        # But total_score >= 2 and exclude_score == 0 → confidence 0.6
        assert r['is_company_particulars'] == True, f'2 keywords + no exclude -> should be true (0.6 conf). Got {r}'
        assert r['confidence'] == 0.6, f'Expected 0.6 confidence, got {r["confidence"]}'
    finally:
        cleanup(path)
test('2个次要关键词 (无主关键词, 无排除词): detected=true, confidence=0.6', t12)

# -------------------------------------------------------
# Test 13: Empty PDF (no pages)
# -------------------------------------------------------
def t13():
    fd, path = tempfile.mkstemp(suffix='.pdf')
    os.close(fd)
    # Create a single empty page (PyMuPDF requires at least 1 page to save)
    doc = fitz.open()
    doc.new_page()
    doc.save(path)
    doc.close()
    try:
        r = detector.is_company_particulars(path)
        assert r['is_company_particulars'] == False
        # Empty page has 0 text -> should be flagged as scanned
        assert r['is_scanned'] == True, f'Empty page should be scanned: {r}'
    finally:
        cleanup(path)
test('空PDF (单空白页): is_scanned=true (no extractable text)', t13)

# -------------------------------------------------------
# Test 14: Defense-in-depth: PHP fallback fields always present
# -------------------------------------------------------
def t14():
    """Verify all fields that PHP uses for OCR fallback are always present"""
    # Test with various inputs
    cases = [
        ('查册文件-CompanyParticulars.pdf', 'scanned user PDF'),
    ]
    for pdf_path, desc in cases:
        r = detector.is_company_particulars(pdf_path)
        for field in ['is_company_particulars', 'is_scanned', 'text_length', 'image_only_pages']:
            assert field in r, f'{desc}: missing field "{field}" for PHP fallback'
test('PHP fallback 字段完整性: 所有必要字段始终存在', t14)

# ============================================================
print('\n' + '=' * 60)
print(f'RESULTS: {PASS} passed, {FAIL} failed, {PASS + FAIL} total')
if FAIL > 0:
    print('*** SOME TESTS FAILED - DO NOT DEPLOY ***')
    sys.exit(1)
else:
    print('All tests passed - safe to deploy')
    sys.exit(0)
