"""Canonicalises attribute keys and values from options, tags and metafields."""
import logging
import re

from app.services.catalog.text import normalise_text

logger = logging.getLogger(__name__)

# "title" is Shopify's single-variant placeholder option ("Default Title");
# "title_tag"/"description_tag" are SEO metafield copy, not product data.
# Both inflate the attribute count that quality_score depends on.
DROPPED_KEYS = {"title", "title_tag", "description_tag"}

KEY_ALIASES = {
    "colour": "color", "color": "color", "shade": "color", "colourway": "color",
    "size": "size", "sizes": "size", "size (uk)": "size", "fit": "size",
    "material": "material", "fabric": "material", "composition": "material",
    "gender": "gender", "sex": "gender", "for": "gender", "department": "gender",
    "style": "style", "design": "style", "pattern": "pattern",
    "capacity": "capacity", "volume": "capacity",
    "weight": "weight", "length": "length", "width": "width",
    "flavour": "flavor", "flavor": "flavor",
    "scent": "scent", "fragrance": "scent",
}

COLOR_BASE = {
    "navy": "blue", "midnight blue": "blue", "cobalt": "blue", "teal": "blue",
    "charcoal": "grey", "slate": "grey", "gunmetal": "grey", "graphite": "grey",
    "ivory": "white", "cream": "white", "ecru": "white", "off white": "white",
    "burgundy": "red", "maroon": "red", "wine": "red", "crimson": "red",
    "tan": "brown", "camel": "brown", "khaki": "brown", "beige": "brown",
}

_ALPHA_SIZE = {
    "xs": "xs", "extrasmall": "xs", "s": "s", "small": "s", "m": "m",
    "medium": "m", "l": "l", "large": "l", "xl": "xl", "extralarge": "xl",
    "xxl": "xxl", "2xl": "xxl",
}

GENDER = {
    "mens": "male", "men": "male", "man": "male", "male": "male", "boys": "male",
    "womens": "female", "women": "female", "woman": "female", "female": "female",
    "ladies": "female", "girls": "female",
    "unisex": "unisex", "all": "unisex", "adult": "unisex", "kids": "kids",
}

# Sizes arrive as ml/L, g/kg, mm/cm/m across metafields and tags. Converting
# to a shared base unit lets identical capacities match regardless of which
# unit the merchant typed (500ml and 0.5L must compare equal).
_TO_BASE = {"ml": 1.0, "l": 1000.0, "g": 1.0, "kg": 1000.0,
            "mm": 1.0, "cm": 10.0, "m": 1000.0}
_BASE_UNIT = {"ml": "ml", "l": "ml", "g": "g", "kg": "g",
              "mm": "mm", "cm": "mm", "m": "mm"}

_KV_TAG = re.compile(r"\A([a-z][a-z0-9 _-]*?)\s*[:=]\s*(.+)\Z", re.I)
_REGIONAL = re.compile(r"\A(uk|us|eu|jp)?[- ]?(\d+(?:\.\d+)?)\Z")
_MEASURED = re.compile(r"\A(\d+(?:\.\d+)?)\s*(ml|l|g|kg|oz|lb|cm|mm|m|in)\Z")

# Structured data wins: a metafield/option is an explicit merchant statement,
# a key:value tag is a weaker keyword guess. Bare-tag inference needs a known-
# value vocabulary and is deliberately out of scope for this module.
SOURCE_CONFIDENCE = {"metafield": 1.0, "option": 1.0, "tag_kv": 0.9}


def normalise_key(k):
    c = normalise_text(k)
    if not c:
        return None
    c = c.lower()
    if c in DROPPED_KEYS:
        return None
    if c in KEY_ALIASES:
        return KEY_ALIASES[c]
    return re.sub(r"[^a-z0-9]+", "_", c).strip("_") or None


def _normalise_colour(raw):
    v = normalise_text(raw)
    if not v:
        return None
    v = v.lower()
    first = re.split(r"[/,|+]| and ", v)[0].strip()
    if first in COLOR_BASE:
        return COLOR_BASE[first]
    for k, base in COLOR_BASE.items():
        if k in first:
            return base
    return first or None


def _normalise_size(raw):
    v = normalise_text(raw)
    if not v:
        return None, None
    flat = v.lower().replace(" ", "")

    if flat in _ALPHA_SIZE:
        return _ALPHA_SIZE[flat], "alpha"

    m = _MEASURED.match(flat)
    if m and m.group(2) in _TO_BASE:
        amount = float(m.group(1)) * _TO_BASE[m.group(2)]
        return f"{amount:g}", _BASE_UNIT[m.group(2)]

    m = _REGIONAL.match(flat)
    if m:
        return m.group(2), m.group(1) or "numeric"

    return flat, None


def normalise_value(key, raw):
    if key == "color":
        return _normalise_colour(raw), None
    if key == "size":
        return _normalise_size(raw)
    if key == "gender":
        v = normalise_text(raw)
        return (GENDER.get(v.lower().replace("'", "")) if v else None), None
    v = normalise_text(raw)
    return (v.lower() if v else None), None


def _attr(key, raw_value, source):
    value, unit = normalise_value(key, raw_value)
    if value is None:
        return None
    # An unrecognised provenance must degrade the sync, not abort it — and
    # must never outrank a source whose trustworthiness we do know, so it
    # falls back to the lowest confidence in the table rather than raising.
    confidence = SOURCE_CONFIDENCE.get(source, 0.5)
    return {"key": key, "raw_value": normalise_text(raw_value), "value": value,
            "unit": unit, "source": source, "confidence": confidence}


def build_attributes(source_product: dict) -> list:
    """Collect attributes from every source, keeping the most trusted per key."""
    found = []

    for a in source_product.get("attributes_raw") or []:
        key = normalise_key(a.get("key"))
        if key:
            attr = _attr(key, a.get("value"), a.get("source", "metafield"))
            if attr:
                found.append(attr)

    for variant in source_product.get("variants") or []:
        for name, value in (variant.get("options") or {}).items():
            key = normalise_key(name)
            if key:
                attr = _attr(key, value, "option")
                if attr:
                    found.append(attr)

    for tag in source_product.get("tags") or []:
        t = normalise_text(tag)
        if not t:
            continue
        m = _KV_TAG.match(t)
        if m:
            key = normalise_key(m.group(1))
            if key:
                attr = _attr(key, m.group(2), "tag_kv")
                if attr:
                    found.append(attr)

    # No separate (key, value) dedupe pass here: this loop already keeps the
    # highest-confidence entry per key across all of `found`, which subsumes
    # deduping by (key, value) -- an extra stage for that could never change
    # the outcome.
    by_key = {}
    for a in found:
        seen = by_key.get(a["key"])
        if seen is None or a["confidence"] > seen["confidence"]:
            by_key[a["key"]] = a

    # Deterministic order: a later task hashes this list to decide whether to
    # re-embed a product, so an unstable order would trigger nightly re-embeds.
    return sorted(by_key.values(), key=lambda a: (a["key"], a["value"]))
