"""The three pairing rules.

Deterministic and dependency-free on purpose: these rules are the part of pairing
worth testing, and a rule that needs a database or a model to exercise stops
being tested honestly.

score and confidence stay separate throughout. A pair can be a strong
relationship derived by a weak method; the approval queue gates on the
derivation, not on the relationship.
"""
import logging

logger = logging.getLogger(__name__)

# A fourth type, "bundle" -- an anchor plus complements from distinct categories
# -- was built and removed. Unlike the three below it described no relationship
# between two products; it assembled a set, and the content signal available here
# is not strong enough to assemble one. On live data it produced "Dress Pea with
# Yellow Peeler", where the embedding had matched "Pea" inside "Peeler".
# Reviving it needs order history, not another threshold.
PAIR_TYPES = ("similar", "complement", "upsell")

SIMILAR_MIN_COSINE = 0.55
SIMILAR_MAX_PRICE_RATIO = 3.0
ACCESSORY_COMPLEMENT_MIN_COSINE = 0.22
GENDER_NEUTRAL = (None, "unisex", "kids")
ACCESSORY_FOR_KEY = "accessory_for"
WEAK_COMPLEMENT_MIN_COSINE = 0.35
UPSELL_MIN_PRICE_RATIO = 1.15
UPSELL_MAX_RATING_DROP = 0.3

# Compatibility is a rule over already-extracted attributes, never a model call:
# the attributes were paid for once in Phase 1 and must not be paid for again.
NEUTRAL_COLORS = {"black", "white", "grey", "gray", "navy", "beige", "cream"}
CLASHING = {frozenset({"orange", "red"}), frozenset({"orange", "pink"}),
            frozenset({"red", "pink"}), frozenset({"brown", "black"})}


def _pairs(product: dict) -> set:
    return {(a.get("key"), str(a.get("value")).lower())
            for a in (product.get("attributes") or [])
            if a.get("key") and a.get("value") is not None}


def attribute_jaccard(a: dict, b: dict) -> float:
    left, right = _pairs(a), _pairs(b)
    if not left or not right:
        return 0.0
    return len(left & right) / len(left | right)


def _same_category(a: dict, b: dict) -> bool:
    return a.get("category") is not None and a.get("category") == b.get("category")


def _price_ratio(a: dict, b: dict):
    low, high = a.get("price_cents"), b.get("price_cents")
    if not low or not high:
        return None
    return max(low, high) / min(low, high)


def _attr(product: dict, key: str):
    for a in product.get("attributes") or []:
        if a.get("key") == key and a.get("value") is not None:
            return str(a["value"]).lower()
    return None


def _colour_compatibility(a: dict, b: dict) -> float:
    """1.0 matching, 0.7 unremarkable, 0.5 unknown, 0.0 clashing.

    0.5 rather than 0.0 for unknown: only 52 of 218 live products carry a colour,
    and treating absent data as a clash would suppress most valid pairs. Unknown
    sits below unremarkable so a pair we can actually see is preferred to one we
    are guessing about.
    """
    left, right = _attr(a, "color"), _attr(b, "color")
    if not left or not right:
        return 0.5
    # Both neutral (or identical) is a genuine match; one neutral side alone
    # must not blank out a clash on the other -- "white" pairs with anything,
    # but that is not the same as every colour pairing perfectly.
    if left == right or (left in NEUTRAL_COLORS and right in NEUTRAL_COLORS):
        return 1.0
    return 0.0 if frozenset({left, right}) in CLASHING else 0.7


def score_similar(a: dict, b: dict, cosine: float):
    if not _same_category(a, b) or cosine < SIMILAR_MIN_COSINE:
        return None
    ratio = _price_ratio(a, b)
    if ratio is not None and ratio > SIMILAR_MAX_PRICE_RATIO:
        return None

    overlap = attribute_jaccard(a, b)
    score = 0.7 * cosine + 0.3 * overlap
    reasons = [f"same category ({a.get('category')})",
               f"{cosine:.2f} text similarity",
               f"{overlap:.2f} attribute overlap"]
    return {"score": round(score, 4), "confidence": round(cosine, 4),
            "source": "embedding", "reasons": reasons}


def _genders_clash(a: dict, b: dict) -> bool:
    """A men's shirt must not suggest a women's handbag.

    Only a definite disagreement blocks: unisex, kids and unknown all stay
    compatible, so nothing legitimate is lost. Phase 1 has extracted gender all
    along and nothing read it.
    """
    left, right = _attr(a, "gender"), _attr(b, "gender")
    if left in GENDER_NEUTRAL or right in GENDER_NEUTRAL:
        return False
    return left != right


def _accessory_targets(product: dict) -> list:
    return [a.get("value") for a in (product.get("attributes") or [])
            if a.get("key") == ACCESSORY_FOR_KEY and a.get("value")]


def _accessorises(candidate: dict, anchor: dict) -> str:
    """Is the candidate an accessory FOR this anchor, not merely an accessory?

    Returns "declared" when the accessory names this anchor's type, "unknown"
    when it names nothing, and "no" when it names something else.

    is_accessory is a boolean, so without this a basketball rim and a pair of
    earbuds both "complement" a shirt. Matching is loose on purpose: the model
    writes "phones" where the category is "smartphones", and "shirts" where it
    is "mens-shirts".
    """
    targets = _accessory_targets(candidate)
    if not targets:
        # Nothing extracted: fall back to the old behaviour rather than
        # silently dropping every accessory on a catalogue not yet re-enriched.
        return "unknown"

    haystack = " ".join(str(v).lower() for v in
                        (anchor.get("category"), anchor.get("product_type")) if v)
    for target in targets:
        word = str(target).lower().strip().rstrip("s")
        if word and (word in haystack or haystack.rstrip("s").endswith(word)):
            return "declared"
    return "no"


def score_complement(a: dict, b: dict, cosine: float):
    """Directional: a suggests b. Never the reverse unless scored separately."""
    if _same_category(a, b):
        return None
    if a.get("is_accessory"):
        return None
    if _genders_clash(a, b):
        return None

    compatibility = _colour_compatibility(a, b)
    if b.get("is_accessory"):
        targeting = _accessorises(b, a)
        if targeting == "no":
            return None
        # The cosine floor was a weak stand-in for "is this accessory actually
        # for this thing". Where the accessory says so outright, that IS the
        # evidence -- and demanding text similarity as well loses real pairs: a
        # handbag and a dress share almost no vocabulary.
        if targeting == "unknown" and cosine < ACCESSORY_COMPLEMENT_MIN_COSINE:
            return None
        # is_accessory says a thing is an accessory, not what it is an accessory
        # FOR -- so a phone would otherwise complement a spatula. Measured on the
        # live catalog, a phone's similarity to mobile-accessories runs to a 0.29
        # median against 0.18 for kitchen-accessories, so text does separate them
        # where the boolean cannot. Cosine therefore outweighs colour here:
        # two black things are not related, they are coincidental.
        if cosine < ACCESSORY_COMPLEMENT_MIN_COSINE:
            return None
        score = 0.5 + 0.1 * compatibility + 0.4 * cosine
        confidence = 0.75
        reasons = [f"{b.get('category')} is an accessory",
                   f"different category from {a.get('category')}",
                   f"{cosine:.2f} text similarity"]
    else:
        # Neither side is an accessory, so there is no structural signal that
        # these belong together -- only the attributes and the text. Colour
        # agreement alone is not evidence: without the cosine floor a black
        # laptop "complements" black bread, because they share a colour and sit
        # in different categories. Both must hold.
        if compatibility < 0.7 or cosine < WEAK_COMPLEMENT_MIN_COSINE:
            return None
        score = 0.3 + 0.3 * compatibility + 0.2 * cosine
        confidence = 0.4
        reasons = [f"different categories ({a.get('category')} / {b.get('category')})",
                   "compatible attributes",
                   f"{cosine:.2f} text similarity"]

    if compatibility < 1.0 and _attr(a, "color") and _attr(b, "color"):
        reasons.append(f"colour {_attr(a, 'color')} with {_attr(b, 'color')}")
    return {"score": round(min(score, 1.0), 4), "confidence": confidence,
            "source": "attribute", "reasons": reasons}


def score_upsell(a: dict, b: dict):
    """Directional: b is the trade-up from a. Arithmetic, never a judgement."""
    if not _same_category(a, b):
        return None
    low, high = a.get("price_cents"), b.get("price_cents")
    if not low or not high or high / low < UPSELL_MIN_PRICE_RATIO:
        return None

    a_rating, b_rating = a.get("rating"), b.get("rating")
    if a_rating is not None and b_rating is not None:
        if b_rating < a_rating - UPSELL_MAX_RATING_DROP:
            return None

    lift = min((high / low - 1) / 2, 1.0)
    reasons = [f"{high / low:.1f}x the price", f"same category ({a.get('category')})"]
    if b_rating is not None:
        reasons.append(f"rated {b_rating}")
    # Arithmetic over two columns leaves nothing to be uncertain about, so this
    # never reaches the approval queue. That is the threshold rule applying
    # normally, not an exemption.
    return {"score": round(0.5 + 0.5 * lift, 4), "confidence": 0.95,
            "source": "arithmetic", "reasons": reasons}


