"""Text, title and brand normalisation. Pure functions, no I/O."""
import logging
import re
import unicodedata
from typing import Iterable

logger = logging.getLogger(__name__)

# Zero-width space, joiners and BOM: invisible, so \s never catches them.
_ZERO_WIDTH = re.compile("[\u200b\u200c\u200d\ufeff]")
# Non-breaking, thin, ideographic etc. spaces: \s doesn't match \xa0 or \u3000.
_ODD_SPACES = re.compile("[\xa0\u2000-\u200a\u202f\u3000]")
_WHITESPACE = re.compile(r"\s+")
_SKU_SUFFIX = re.compile(
    r"\s*[\(\[]\s*(sku|item|ref|art)[-.: ]*[\w-]+\s*[\)\]]\s*\Z", re.I)
_PLACEHOLDER_BRAND = re.compile(
    r"\A(n/?a|none|unknown|default|generic|no brand|-)\Z", re.I)

# A prefix rule, not fuzzy similarity: a similarity score cannot separate
# "Galaxiq Store" from "Galaxy" — any threshold loose enough to catch the
# former also nulls out a genuine, unrelated brand.


def normalise_text(s: str | None) -> str | None:
    """NFKC first: merchants paste from Word, Excel and PDFs, so ligatures and
    full-width characters arrive routinely and would never match otherwise."""
    if s is None or not isinstance(s, str):
        return None
    out = unicodedata.normalize("NFKC", s)
    out = _ZERO_WIDTH.sub("", out)
    out = _ODD_SPACES.sub(" ", out)
    out = _WHITESPACE.sub(" ", out).strip()
    return out or None


def _title_case(s: str) -> str:
    return " ".join(w.capitalize() for w in s.split(" "))


def _is_shop_name(vendor: str, shop_name: str) -> bool:
    v, s = vendor.lower(), shop_name.lower()
    if v == s:
        return True
    if v.startswith(s) and not v[len(s):len(s) + 1].isalnum():
        return True  # "Galaxiq Store" vendor against shop "Galaxiq"
    if s.startswith(v) and not s[len(v):len(v) + 1].isalnum():
        return True  # vendor "Galaxiq" against shop "Galaxiq Store"
    return False


def clean_title(raw: str | None, brand: str | None = None) -> str | None:
    t = normalise_text(raw)
    if not t:
        return None

    if brand:
        stripped = re.sub(
            rf"\A{re.escape(brand)}\s*[-–—|:]\s*", "", t, flags=re.I)
        # Guard: a title that is only the brand needs it more than the separation.
        if len(stripped.strip()) > 1:
            t = stripped

    t = _SKU_SUFFIX.sub("", t)

    if len(t) > 8 and t == t.upper() and re.search(r"[A-Z]{4,}", t):
        t = _title_case(t)

    return t.strip() or None


def normalise_brand(
    vendor: str | None,
    shop_name: str | None,
    known_brands: Iterable[str] = (),
) -> str | None:
    """Shopify defaults vendor to the shop name when a merchant leaves it blank,
    so without the shop-name check a whole catalog shares one brand and any
    brand-match signal becomes noise."""
    b = normalise_text(vendor)
    if not b:
        return None

    if _PLACEHOLDER_BRAND.match(b):
        return None

    if shop_name and _is_shop_name(b, shop_name):
        return None

    for known in known_brands:
        if known.lower() == b.lower():
            return known

    return b
