"""Merges inferred attributes into existing ones without overwriting.

A merchant's own value always wins. Extraction fills gaps only, and everything it
produces is marked inferred with a confidence below any structured source, so the
matching layer can weight a real value above a guess.
"""
import logging

logger = logging.getLogger(__name__)

INFERRED_CONFIDENCE = 0.5

# These get their own columns because they are filtered on. Storing them as
# attributes as well would inflate the attribute count quality_score depends on.
SCALAR_FIELDS = ("is_accessory", "price_tier")

# Extracted as lists, stored as one attribute row each so the pairing rules can
# read them without parsing a joined string.
_LIST_FIELDS = {"key_features": "feature", "accessory_for": "accessory_for"}


def merge_attributes(existing: list, inferred: dict) -> tuple:
    merged = list(existing or [])
    present = {a["key"] for a in merged}
    conflicts = 0

    for key, value in sorted((inferred or {}).items()):
        if key in SCALAR_FIELDS:
            continue

        if key in _LIST_FIELDS:
            attribute_key = _LIST_FIELDS[key]
            for item in value:
                if not any(a["key"] == attribute_key and a["value"] == item
                           for a in merged):
                    merged.append(_inferred(attribute_key, item))
            continue

        if key in present:
            prior = next(a for a in merged if a["key"] == key)
            if prior.get("value") != value:
                conflicts += 1
            continue

        merged.append(_inferred(key, value))

    return sorted(merged, key=lambda a: (a["key"], str(a["value"]))), conflicts


def _inferred(key: str, value) -> dict:
    return {"key": key, "raw_value": value, "value": value, "unit": None,
            "source": "inferred", "confidence": INFERRED_CONFIDENCE}
