"""Turns a SourceProduct into the row shape strategist_products stores.

Deterministic and pure: no network, no database, no LLM. Every rule here is
reproducible from a fixture, which is what makes dictionary changes safe to make.
"""
import hashlib
import json
import logging

from app.services.catalog.attributes import build_attributes
from app.services.catalog.commercial import (
    clean_image_url, is_in_stock, price_range, resolve_product_url,
)
from app.services.catalog.taxonomy import resolve_taxonomy
from app.services.catalog.text import clean_title, normalise_brand, normalise_text
from app.services.catalog.fx import to_reference_cents

logger = logging.getLogger(__name__)

MIN_TITLE_LENGTH = 3

# Of everything that can land in missing_fields, only these make a product
# unservable. A missing product_url means a recommendation cannot be clicked;
# a missing price_reference only makes cross-currency comparison harder, and
# every consumer of price_reference_cents already falls back to price_cents.
# Treating every gap as fatal made currency effectively mandatory on a source
# where it is documented as optional.
BLOCKING_FIELDS = ("product_url",)


def normalise(source_product: dict, context: dict) -> dict | None:
    external_id = source_product.get("external_id")
    if external_id in (None, ""):
        return None

    variants = source_product.get("variants") or []
    if not variants:
        return None

    brand = normalise_brand(source_product.get("brand"),
                             context.get("shop_name"),
                             context.get("known_brands") or ())

    name = clean_title(source_product.get("title"), brand)
    if not name or len(name) < MIN_TITLE_LENGTH:
        return None

    lo, hi, compare_at, on_sale = price_range(variants)
    if lo is not None and lo < 0:
        return None

    url = resolve_product_url(source_product,
                               context.get("primary_domain"),
                               context.get("url_template"))

    # A URL is genuinely needed, but rejecting the product hides the problem:
    # the merchant would see an empty catalog with no explanation. Keep it,
    # flag it, and let it be excluded from serving instead.
    # Not every gap is equal. A product with no URL cannot be linked to, so it
    # must never be served. A product with no converted reference price is only
    # harder to compare against another currency -- every consumer of
    # price_reference_cents already falls back to price_cents -- so flagging it
    # is information, not a reason to hide the product.
    missing_fields = [] if url else ["product_url"]

    # A crawled page carries a price only when the merchant publishes
    # schema.org markup. Dropping the product would make a site without markup
    # look like a site with no products, so it is kept and flagged -- the same
    # rule as the URL above. It still shows in the catalogue and still pairs on
    # similarity; only the rules that need a number (upsell, the price-ratio
    # guard) skip it. A guessed price would be worse: wrong is invisible where
    # missing is not.
    if lo is None:
        missing_fields.append("price")

    # A product may carry its own currency -- a CSV has a Currency column, so
    # one upload can hold two currencies. Shopify and the HTTP sources have one
    # currency for the whole connection and set it on the context instead.
    currency = source_product.get("currency") or context.get("currency")

    price_reference_cents, fx_rate_used = to_reference_cents(
        lo, currency, context.get("reference_currency"))
    if price_reference_cents is None:
        # Comparing an unconverted price is worse than not recommending the
        # product: the failure is invisible and looks like a ranking decision
        # rather than the missing exchange rate it actually is.
        missing_fields.append("price_reference")

    taxonomy = resolve_taxonomy(source_product)
    attributes = build_attributes(source_product)
    tags = sorted({t for t in (normalise_text(x) for x in
                                source_product.get("tags") or []) if t})

    return {
        # Includes source_ref: /catalog/sync loops over every source of a
        # tenant, and two sources of the same kind (two Shopify stores, two
        # HTTP APIs) would otherwise collide on {source_kind}:{external_id}
        # and overwrite each other's rows.
        "product_key": f"{source_product['source_kind']}:{source_product['source_ref']}:{external_id}",
        "source_kind": source_product["source_kind"],
        "source_ref": source_product["source_ref"],
        "external_id": str(external_id),
        "name": name,
        "description": normalise_text(source_product.get("description")),
        "brand": brand,
        "image_url": clean_image_url(source_product.get("image_url")),
        "product_url": url,
        "taxonomy_path": taxonomy["path"],
        "taxonomy_source": taxonomy["source"],
        "raw_category": taxonomy["raw"],
        "category": taxonomy["path"][-1] if taxonomy["path"] else None,
        "tags": tags,
        "attributes": attributes,
        "price_cents": lo,
        "price_max_cents": hi,
        "compare_at_cents": compare_at,
        "currency": currency,
        "price_reference_cents": price_reference_cents,
        "fx_rate_used": fx_rate_used,
        "on_sale": on_sale,
        "in_stock": is_in_stock(source_product),
        "status": normalise_text(source_product.get("status")),
        "missing_fields": missing_fields,
        # What a shopper presses on the card. A source that supplies none
        # leaves this empty and the card falls back to the product URL.
        "ctas": source_product.get("ctas") or [],
        # What a shopper presses on the card. Sources that supply none leave
        # this empty, and the card falls back to the product URL.
        "ctas": source_product.get("ctas") or [],
        "tenant_relations": source_product.get("tenant_relations") or {},
        "rating": source_product.get("rating"),
        "review_count": source_product.get("review_count"),
        "featured_rank": source_product.get("featured_rank"),
    }


def quality_score(product: dict) -> float:
    s = 0.0
    name = product.get("name") or ""
    s += 0.25 if len(name) >= 10 else 0.0
    s += 0.25 if len(product.get("taxonomy_path") or []) >= 2 else 0.0
    s += 0.15 if product.get("brand") else 0.0
    s += min(len(product.get("attributes") or []), 4) / 4 * 0.20
    s += 0.10 if product.get("image_url") else 0.0
    s += 0.05 if product.get("description") else 0.0
    return round(s, 2)


def content_hash(product: dict) -> str:
    # Covers only fields that affect matching, so a price change updates the
    # row without forcing a re-embed. Arrays are sorted: tags and attributes
    # arrive in arbitrary order, and hashing them unsorted would change the
    # hash on every sync and re-embed the whole catalog nightly.
    parts = [
        product.get("name") or "",
        product.get("brand") or "",
        ">".join(product.get("taxonomy_path") or []),
        ",".join(sorted(product.get("tags") or [])),
        ",".join(sorted(f"{a['key']}={a['value']}"
                         for a in product.get("attributes") or [])),
        # Description is deliberately absent: the matching specification's
        # embedding input excludes it, so hashing it would re-embed a catalog
        # after an edit that cannot change any embedding.
    ]
    return hashlib.sha256("|".join(parts).encode()).hexdigest()


def record_hash(product: dict) -> str:
    return hashlib.sha256(
        json.dumps(product, sort_keys=True, default=str).encode()).hexdigest()
