"""Resolves a product's category from whichever source actually carries one."""
import logging
import re

from app.services.catalog.text import normalise_text

logger = logging.getLogger(__name__)

MAX_DEPTH = 4

# Nearly every store has a collection holding products from unrelated categories.
# Resolving to one makes every discounted product a category sibling of every
# other, which is exactly the wrong grouping. "Automated Collection" covers 8 of
# 17 products in the live test store and carries no meaning at all.
MERCH_COLLECTIONS = re.compile(
    r"\A(sale|clearance|new|new in|featured|best ?sellers?|trending|all|"
    r"home ?page|staff picks|automated collection)\Z", re.I)


def _truncate(path: list) -> list:
    # A 7-level path fragments the catalog into single-product leaves, which
    # makes category affinity useless because nothing shares a leaf.
    return path[:MAX_DEPTH]


def resolve_taxonomy(source_product: dict) -> dict:
    raw = normalise_text(source_product.get("raw_category"))
    if raw:
        path = [normalise_text(p) for p in raw.split(" > ")]
        return {"path": _truncate([p for p in path if p]),
                "source": "native", "raw": raw}

    ptype = normalise_text(source_product.get("product_type"))
    if ptype:
        return {"path": [ptype], "source": "product_type", "raw": ptype}

    # Sorted: taxonomy_path feeds content_hash, so picking the first entry of
    # an unsorted list would make the hash depend on source ordering and
    # re-embed the catalog on every sync whenever a source reorders its tags.
    for title in sorted(source_product.get("collections") or []):
        c = normalise_text(title)
        if c and not MERCH_COLLECTIONS.match(c):
            return {"path": [c], "source": "collection", "raw": c}

    for tag in sorted(source_product.get("tags") or []):
        t = normalise_text(tag)
        if t:
            return {"path": [t], "source": "tag", "raw": t}

    return {"path": [], "source": "none", "raw": None}
