"""营业执照关键字段解析：从 OCR 文字行里启发式提取（自动识别中文/欧洲国家）。"""
from __future__ import annotations

import re

# ===== 中文执照规则 =====
_CN_MARKERS = ("统一社会信用代码", "法定代表人", "经营范围", "注册资本", "营业执照")
# 统一社会信用代码：18 位（数字 + 大写字母，不含 I/O/Z/S/V）
_CREDIT_CODE = re.compile(r"\b[0-9A-HJ-NPQRTUWXY]{18}\b")
# 中文地址行：省/市/县/区 … 路/街/道/村/号（兜底识别，处理「住所」被 OCR 拆行的情况）
_CN_ADDRESS_LINE = re.compile(r"(省|市|县|区).{0,30}(路|街|道|村|号)")


# ===== 欧洲各国规则（country_code -> 规则集）=====
# 每个规则集四个正则：reg_num(注册号) / suffix(公司名后缀) / address(地址) / person(法人)
_COUNTRY_RULES: dict[str, dict] = {
    "de": {
        "reg_num": re.compile(r"\b(HRA|HRB)\s*\.?\s*\d[\w/\.\-]*", re.IGNORECASE),
        "suffix": re.compile(r"\b(UG|GmbH|AG|KG|OHG|SE)\b", re.IGNORECASE),
        "address": re.compile(
            r"(Gesch[aä]ftsanschrift|Anschrift|Sitz)\s*[:：]|\b\d{5}\s+[A-Za-zÄÖÜäöüß]",
            re.IGNORECASE,
        ),
        "person": re.compile(r"(Gesch[aä]ftsf[uü]hrer|Managing\s*Director)\s*[:：]", re.IGNORECASE),
    },
    "at": {
        "reg_num": re.compile(r"\b(FN)\s*\.?\s*\d+\w?\b", re.IGNORECASE),
        "suffix": re.compile(r"\b(GmbH|AG|KG|OG|SE)\b", re.IGNORECASE),
        "address": re.compile(
            r"(Firmensitz|Sitz|Gesch[aä]ftsanschrift|Anschrift)\s*[:：]|\b\d{4}\s+[A-Za-zÄÖÜäöüß]",
            re.IGNORECASE,
        ),
        "person": re.compile(r"(Gesch[aä]ftsf[uü]hrer|Vorstand)\s*[:：]", re.IGNORECASE),
    },
    "nl": {
        "reg_num": re.compile(r"\b(KvK|KVK)\s*[:#]?\s*\d{7,8}\b", re.IGNORECASE),
        "suffix": re.compile(r"\b(BV|NV|VOF|Stichting)\b", re.IGNORECASE),
        "address": re.compile(
            r"(Vestigingsadres|Bezoekadres|Adres)\s*[:：]|\b\d{4}\s?[A-Z]{2}\b",
            re.IGNORECASE,
        ),
        "person": re.compile(r"(Bestuurder|Directeur)\s*[:：]", re.IGNORECASE),
    },
    "fr": {
        "reg_num": re.compile(r"\b(SIREN|SIRET|R\.?C\.?S\.?)\s*[:#]?\s*\d{9,14}\b", re.IGNORECASE),
        "suffix": re.compile(r"\b(SAS|SARL|SA|SCI|SASU|EURL)\b", re.IGNORECASE),
        "address": re.compile(
            r"(Si[eè]ge\s*social|Adresse)\s*[:：]|\b\d{5}\s+[A-Za-zÀ-ÿ]",
            re.IGNORECASE,
        ),
        "person": re.compile(r"(G[eé]rant|Pr[eé]sident|Directeur)\s*[:：]", re.IGNORECASE),
    },
    "gb": {
        "reg_num": re.compile(r"\b(Company\s*No\.?|Registration\s*No\.?)\s*[:#]?\s*\d{6,8}\b", re.IGNORECASE),
        "suffix": re.compile(r"\b(Ltd\.?|Limited|PLC|LLP)\b", re.IGNORECASE),
        "address": re.compile(r"(Registered\s*Office|Registered\s*Address)\s*[:：]", re.IGNORECASE),
        "person": re.compile(r"(Director|Secretary)\s*[:：]", re.IGNORECASE),
    },
    "pl": {
        "reg_num": re.compile(r"\b(KRS)\s*[:#]?\s*\d{9,10}\b", re.IGNORECASE),
        "suffix": re.compile(
            r"(sp\.\s*z\.\s*o\.\s*o\.|S\.\s*A\.|sp\.\s*k\.|sp\.\s*j\.|s\.\s*c\.|"
            r"spółka\s+z\s+ograniczoną\s+odpowiedzialnością|spolka\s+z\s+ograniczona\s+odpowiedzialnoscia|"
            r"spółka\s+akcyjna|spolka\s+akcyjna)",
            re.IGNORECASE,
        ),
        "name_label_next": ("firma, pod",),
        "strip": re.compile(r"\s*(sp\.|S\.\s*A\.|s\.\s*c\.|sp[oó][lł]?k).*$", re.IGNORECASE),
        "address": re.compile(r"\bul\.|\bkod\s+\d{2}-\d{3}", re.IGNORECASE),
        "address_label_next": ("Siedziba", "adres", "Adres"),
        "person": re.compile(r"(Zarząd|Członek\s*zarządu|Prezes|Prokurent)\s*[:：]", re.IGNORECASE),
        "person_combine": (("Imiona",), ("Nazwisko",)),
    },
}

# 未识别到国家时的通用兜底
_GENERIC_RULES = {
    "reg_num": re.compile(
        r"\b(HRA|HRB)\s*\.?\s*(\d[\w/\.\-]*)|"
        r"\b(Reg\.?\s*No\.?|Registration\s*No\.?|Nr\.?)\s*[:#]?\s*([\w/\.\-]+)",
        re.IGNORECASE,
    ),
    "suffix": re.compile(
        r"\b(UG|GmbH|AG|KG|OHG|SE|Co\.|Ltd\.?|Limited|LLC|BV|NV|SAS|SARL|SA|SL|Oy|ApS)\b",
        re.IGNORECASE,
    ),
    "address": re.compile(
        r"(Gesch[aä]ftsanschrift|Anschrift|Sitz|Registered\s*Office|Address|Adresse)\s*[:：]"
        r"|\b\d{4,5}\s+[A-Za-zÄÖÜäöüß]",
        re.IGNORECASE,
    ),
    "person": re.compile(
        r"(Gesch[aä]ftsf[uü]hrer|Managing\s*Director|Legal\s*Representative|Director)\s*[:：]",
        re.IGNORECASE,
    ),
}


def _is_chinese(lines: list[str]) -> bool:
    text = "\n".join(lines)
    return any(marker in text for marker in _CN_MARKERS)


def _detect_country(lines: list[str]) -> str | None:
    """根据注册号前缀自动识别国家。"""
    text = "\n".join(lines)
    if re.search(r"\b(HRB|HRA)\b", text, re.IGNORECASE):
        return "de"
    if re.search(r"\bFN\s*\d", text):
        return "at"
    if re.search(r"\b(KvK|KVK)\b", text, re.IGNORECASE):
        return "nl"
    if re.search(r"\b(SIREN|SIRET)\b", text, re.IGNORECASE):
        return "fr"
    if re.search(r"(Company\s*No\.?|Registered\s*Office)", text, re.IGNORECASE):
        return "gb"
    if re.search(r"\bKRS\b", text, re.IGNORECASE):
        return "pl"
    return None


def _value_after_label(lines: list[str], labels: tuple[str, ...]) -> str | None:
    """取标签后的值：支持「标签: 值」与「标签\n值」两种排版。"""
    for i, line in enumerate(lines):
        line = line.strip()
        for label in labels:
            if label in line:
                rest = line.split(label, 1)[1].strip("：: 　")
                if rest:
                    return rest
                for j in range(i + 1, len(lines)):
                    candidate = lines[j].strip()
                    if candidate:
                        return candidate
    return None


def _value_next_line(lines: list[str], labels: tuple[str, ...]) -> str | None:
    """找标签所在行（不区分大小写），返回紧邻的下一非空行。"""
    lower_labels = [label.lower() for label in labels]
    for i, line in enumerate(lines):
        lower_line = line.strip().lower()
        if any(label in lower_line for label in lower_labels):
            for j in range(i + 1, len(lines)):
                candidate = lines[j].strip()
                if candidate:
                    return candidate
    return None


def _parse_chinese(lines: list[str]) -> dict:
    reg_num = None
    for line in lines:
        m = _CREDIT_CODE.search(line)
        if m:
            reg_num = m.group(0)
            break

    # 公司名：含「公司」且足够长（区别于「有限责任公司」这类类型行）
    name = None
    for line in lines:
        line = line.strip()
        if len(line) > 8 and "公司" in line:
            name = line
            break

    address = _value_after_label(lines, ("住所", "住址", "注册地址", "地址"))
    if not address:
        for line in lines:
            line = line.strip()
            if _CN_ADDRESS_LINE.search(line):
                address = line
                break

    return {
        "reg_num": reg_num,
        "name": name,
        "address": address,
        "person": _value_after_label(lines, ("法定代表人", "法人")),
    }


def _parse_european(lines: list[str], rules: dict) -> dict:
    reg_num = address = person = None
    for line in lines:
        line = line.strip()
        if not line:
            continue
        if not reg_num and rules["reg_num"].search(line):
            reg_num = line
        if not address and rules["address"].search(line):
            address = line
        if not person and rules["person"].search(line):
            person = line

    # 地址/法人：正则匹配不到时，用「标签下一行」兜底
    if not address and rules.get("address_label_next"):
        address = _value_next_line(lines, rules["address_label_next"])
    if not person and rules.get("person_label_next"):
        person = _value_next_line(lines, rules["person_label_next"])
    # 组合式法人（如波兰：名 Imiona + 姓 Nazwisko）
    if rules.get("person_combine"):
        first_labels, surname_labels = rules["person_combine"]
        first = _value_next_line(lines, first_labels)
        surname = _value_next_line(lines, surname_labels)
        person = " ".join(x for x in (first, surname) if x) or None

    # 公司名：标签优先（同行的用 name_label，下一行的用 name_label_next），后缀兜底
    name = None
    if rules.get("name_label"):
        name = _value_after_label(lines, rules["name_label"])
    if not name and rules.get("name_label_next"):
        name = _value_next_line(lines, rules["name_label_next"])
    if not name:
        for line in lines:
            line = line.strip()
            if line and rules["suffix"].search(line):
                name = line
                break

    # 去掉法律形式后缀，得到具体商号（去空则保留原名）
    if name and rules.get("strip"):
        stripped = rules["strip"].sub("", name).strip(" ,.;-")
        if stripped:
            name = stripped

    return {"reg_num": reg_num, "name": name, "address": address, "person": person}


def parse_biz_license(lines: list[str], country_code: str | None = None) -> dict:
    """从营业执照 OCR 文字行解析关键字段（自动识别中文/欧洲国家）。

    返回 {"reg_num", "name", "address", "person"}，无匹配为 None。
    country_code 可显式指定（de/at/nl/fr/gb），否则按注册号前缀自动识别。
    """
    if _is_chinese(lines):
        return _parse_chinese(lines)
    code = country_code or _detect_country(lines)
    rules = _COUNTRY_RULES.get(code) if code else _GENERIC_RULES
    return _parse_european(lines, rules)
