"""Turns a Shopify-format product CSV into source products.

Pure: no database, no network, no request objects, so every grouping and
validation rule is testable without a fixture. The output is the same shape the
Shopify and HTTP connectors emit, so normalise() takes it unchanged.
"""
import csv
import io
import logging

logger = logging.getLogger(__name__)

REQUIRED_COLUMNS = ("Handle", "Title")
_OUT_OF_STOCK_QTY = 0

# Shopify's own export headers are historical artefacts -- "Body (HTML)" is just
# the description, and a merchant filling this in by hand reasonably assumes it
# wants HTML. Both spellings are accepted so a raw Shopify export imports
# unchanged while the hand-filled template reads like English.
COLUMN_ALIASES = {
    "description": "Body (HTML)",
    "brand": "Vendor",
    "category": "Type",
    "sku": "Variant SKU",
    "quantity": "Variant Inventory Qty",
    "stock": "Variant Inventory Qty",
    "price": "Variant Price",
    "compare at price": "Variant Compare At Price",
    "image url": "Image Src",
    "image": "Image Src",
    "url handle": "Handle",
    # Shopify's own export has no product-URL column -- it derives one from the
    # store domain. A CSV has no domain, and a product with no URL cannot be
    # linked to, so the column is offered here instead.
    "url": "Product URL",
    "product url": "Product URL",
    "link": "Product URL",
    "option1": "Option1 Value",
    "option2": "Option2 Value",
}


def canonical_header(name: str) -> str:
    """A CSV header under the name the rest of this module uses."""
    cleaned = _clean(name)
    return COLUMN_ALIASES.get(cleaned.lower(), cleaned)


class CsvFormatError(Exception):
    """The file cannot be read at all. Row problems are reported, not raised."""


def _clean(value):
    return (value or "").strip()


def _options(row: dict) -> dict:
    options = {}
    for index in (1, 2, 3):
        name = _clean(row.get(f"Option{index} Name"))
        value = _clean(row.get(f"Option{index} Value"))
        if name and value:
            options[name] = value
    return options


def _cta(row: dict, product_url: str) -> list:
    """The button a shopper presses, if the merchant supplied one.

    Same shape the crawler produces -- label, url, type -- so a card renders
    identically whether the product came from a spreadsheet or a page. The type
    is inferred from the label the way it is for a scraped button, rather than
    asking a merchant to learn a vocabulary.

    A label with no URL falls back to the product's own page: a button that
    goes nowhere is worse than one that goes to the product.
    """
    label = _clean(row.get("CTA Text"))
    url = _clean(row.get("CTA URL")) or product_url
    if not label or not url:
        return []

    # Imported here, not at module scope: the classifier lives beside the page
    # scraper, which brings BeautifulSoup and the model client with it, and this
    # module is deliberately free of both. One set of rules is still worth the
    # deferred import -- a duplicate list would drift.
    from app.services.content.product_extraction import _classify_cta

    return [{"label": label, "url": url, "type": _classify_cta(url, label)}]


def _parse_price(raw: str):
    """Returns (price, reason). reason is None on success."""
    value = _clean(raw)
    if not value:
        return None, "Variant Price is required"
    try:
        price = float(value)
    except ValueError:
        return None, f"Variant Price {value!r} is not a number"
    if price < 0:
        return None, f"Variant Price {value!r} cannot be negative"
    return price, None


def _parse_quantity(raw: str):
    """Returns (quantity, reason). Blank means untracked, treated as 0."""
    value = _clean(raw)
    if not value:
        return 0, None
    try:
        quantity = int(float(value))
    except ValueError:
        return None, f"Variant Inventory Qty {value!r} is not a number"
    if quantity < 0:
        return None, f"Variant Inventory Qty {value!r} cannot be negative"
    return quantity, None


def _compare_at_price(raw: str):
    value = _clean(raw)
    if not value:
        return None
    try:
        return float(value)
    except ValueError:
        return None


def parse_products(text: str, source_ref: str,
                   currency: str = None) -> tuple[list[dict], list[dict]]:
    # Currency is optional, as it is when connecting a product API. It is still
    # never guessed: without one the products carry no converted reference
    # price and are flagged, which affects comparing prices across sources and
    # nothing else.
    currency = _clean(currency) or None
    if not text or not text.strip():
        raise CsvFormatError("file is empty")

    reader = csv.DictReader(io.StringIO(text))
    fieldnames = [canonical_header(f) for f in (reader.fieldnames or [])]
    missing = [c for c in REQUIRED_COLUMNS if c not in fieldnames]
    if missing:
        raise CsvFormatError(f"missing required column(s): {', '.join(missing)}")

    rows = [(number, {canonical_header(k): v for k, v in row.items()})
            for number, row in enumerate(reader, start=2)]
    if not rows:
        raise CsvFormatError("file has no data rows")

    products: dict[str, dict] = {}
    errors: list[dict] = []

    for row_number, row in rows:
        handle = _clean(row.get("Handle"))
        if not handle:
            errors.append({"row": row_number, "reason": "Handle is required"})
            continue

        price, price_error = _parse_price(row.get("Variant Price"))
        if price_error:
            errors.append({"row": row_number, "reason": price_error})
            continue

        quantity, quantity_error = _parse_quantity(row.get("Variant Inventory Qty"))
        if quantity_error:
            errors.append({"row": row_number, "reason": quantity_error})
            continue

        product = products.get(handle)
        if product is None:
            title = _clean(row.get("Title"))
            if not title:
                errors.append({"row": row_number, "reason": "Title is required on a handle's first row"})
                continue

            tags_raw = _clean(row.get("Tags"))
            tags = [t.strip() for t in tags_raw.split(",") if t.strip()] if tags_raw else []

            product = {
                "external_id": handle,
                "title": title,
                "description": _clean(row.get("Body (HTML)")) or None,
                "brand": _clean(row.get("Vendor")) or None,
                "handle": handle,
                "product_url": _clean(row.get("Product URL")) or None,
                "image_url": _clean(row.get("Image Src")) or None,
                "raw_category": None,
                "product_type": _clean(row.get("Type")) or None,
                "tags": tags,
                "collections": [],
                "status": _clean(row.get("Status")) or None,
                "tracks_inventory": True,
                "variants": [],
                "attributes_raw": [],
                "ctas": _cta(row, _clean(row.get("Product URL"))),
                "source_kind": "csv",
                "source_ref": source_ref,
                "tenant_relations": {},
                "rating": None,
                "review_count": None,
                "featured_rank": None,
                # The row wins over the upload-wide value: it is the more
                # specific statement, and it lets one file hold two currencies.
                "currency": _clean(row.get("Currency")).upper() or currency,
            }
            products[handle] = product

        product["variants"].append({
            "price": price,
            "compare_at_price": _compare_at_price(row.get("Variant Compare At Price")),
            "available": quantity != _OUT_OF_STOCK_QTY,
            "options": _options(row),
        })

    if errors:
        logger.warning("csv import %s: %d row error(s)", source_ref, len(errors))

    return list(products.values()), errors
