"""Turns a crawled storefront into source products.

The third way into the catalogue, alongside a CSV upload and a product API: the
merchant gives us a page on their own site and we read it. The pages themselves
are fetched and parsed by the website crawler that already serves chat
ingestion; this module's job is to hand what it finds to normalise() in exactly
the shape the Shopify and HTTP connectors produce, so a crawled product is a
first-class catalogue row rather than a name and a picture.

Price, brand and stock come from the page's own schema.org markup where it has
any. That is exact data the merchant already publishes for Google, so reading it
costs nothing and cannot hallucinate. A page without it yields a product with no
price, which normalise() flags -- the same treatment a CSV row with no currency
gets, and for the same reason: a guessed price is worse than a missing one.
"""
import asyncio
import json
import logging
import xml.etree.ElementTree as ET
from urllib.parse import urlparse

import httpx

from app.core.config import settings
from app.core.llm_client import client as openai_client
from app.core.prompts import CURRENCY_INFERENCE_PROMPT
from app.services.content.product_extraction import backfill_categories, extract_page_products
from app.services.content.website import _looks_like_product_url, extract_text_from_url
from app.services.infra.database import insert_llm_usage
from app.services.infra.ssrf import assert_safe_url

logger = logging.getLogger(__name__)

_MAX_CURRENCY_INFERENCE_CHARS = 4000
_SITEMAP_NS = "{http://www.sitemaps.org/schemas/sitemap/0.9}"


async def _discover_via_sitemap(url: str) -> list:
    """Product URLs straight from /sitemap.xml, cheaper than discovering them
    by crawling. Never raises -- any failure just falls back to the normal
    link-following crawl."""
    try:
        base = urlparse(url)
        async with httpx.AsyncClient(timeout=10.0) as client:
            resp = await client.get(f"{base.scheme}://{base.netloc}/sitemap.xml")
            if resp.status_code != 200:
                return []
            root = ET.fromstring(resp.content)

            if root.tag == f"{_SITEMAP_NS}sitemapindex":
                children = [el.text for el in root.iter(f"{_SITEMAP_NS}loc") if el.text]
                products_child = next((c for c in children if "product" in c.lower()), None)
                if not products_child:
                    return []
                resp = await client.get(products_child)
                if resp.status_code != 200:
                    return []
                root = ET.fromstring(resp.content)

            urls = [el.text for el in root.iter(f"{_SITEMAP_NS}loc") if el.text]
    except Exception as ex:
        logger.info("Sitemap discovery skipped for %s: %s", url, ex)
        return []

    seen = set()
    product_urls = []
    for u in urls:
        if _looks_like_product_url(u) and u not in seen:
            seen.add(u)
            product_urls.append(u)
    return product_urls

# schema.org spells availability as a URL. Only an explicit out-of-stock hides a
# product: a page that says nothing is treated as available, because most small
# storefronts omit the field entirely and hiding their whole catalogue over a
# missing annotation would be worse than occasionally showing a sold-out item.
_OUT_OF_STOCK = ("outofstock", "soldout", "discontinued")


class CrawlSourceError(Exception):
    """The crawl could not complete. Row-level problems are skipped, not raised."""


def _text(value) -> str:
    """schema.org lets almost any field be a string, a list, or a nested node."""
    if isinstance(value, list):
        value = value[0] if value else None
    if isinstance(value, dict):
        value = value.get("name") or value.get("@value")
    return str(value).strip() if value not in (None, "") else ""


def _offer(block: dict) -> dict:
    """The offer to read price/availability from -- a plain Product's own
    offers, or (Shopify's ProductGroup schema) the first variant's offers
    when the group itself carries none. One variant's price stands in for
    the whole product here, matching to_source_product's single-variant
    model of a crawled page."""
    offers = block.get("offers")
    if not offers:
        variants = block.get("hasVariant")
        if isinstance(variants, list) and variants:
            offers = variants[0].get("offers")
    if isinstance(offers, list):
        offers = offers[0] if offers else None
    return offers if isinstance(offers, dict) else {}


def _price(offer: dict):
    raw = offer.get("price") or offer.get("lowPrice")
    if raw in (None, ""):
        return None
    try:
        price = float(str(raw).replace(",", ""))
    except (TypeError, ValueError):
        return None
    return price if price >= 0 else None


def _in_stock(offer: dict) -> bool:
    availability = _text(offer.get("availability")).lower()
    return not any(term in availability for term in _OUT_OF_STOCK)


def _external_id(product: dict) -> str:
    """A stable id for this product within its site.

    The page URL, not the name: a merchant renaming a product must update the
    row rather than orphan it and create a second one.
    """
    raw = product.get("raw") or {}
    sku = raw.get("sku")
    return str(sku) if sku else (product.get("product_url") or "").rstrip("/")


def to_source_product(product: dict, source_ref: str, currency: str = None) -> dict:
    """One crawled product in the shape normalise() consumes.

    price and brand prefer the page's own structured markup; product.get(...)
    is only ever consulted when that markup had nothing (extract_with_llm's
    price/brand are themselves None on any page that had JSON-LD or Open
    Graph data, since extract_page_products never reaches the model then).
    """
    block = (product.get("raw") or {}).get("jsonld") or {}
    offer = _offer(block)
    price = _price(offer) or product.get("price")

    external_id = _external_id(product)
    if not external_id or not product.get("name"):
        return None

    return {
        "external_id": external_id,
        "title": product.get("name"),
        "description": product.get("description") or None,
        "brand": _text(block.get("brand")) or product.get("brand") or None,
        "handle": None,
        "product_url": product.get("product_url") or None,
        "image_url": product.get("image_url") or None,
        "raw_category": None,
        "product_type": product.get("category") or None,
        "tags": [],
        "collections": [],
        "status": None,
        "tracks_inventory": True,
        # extract_page_products already read the page's own buttons; not
        # carrying them through would silently discard real markup for a
        # source that, unlike Shopify or an API, actually has some.
        "ctas": product.get("ctas") or [],
        # A crawled page shows one buyable thing. Its size and colour choices
        # are options on that one product, not separate variants with their own
        # prices, so there is exactly one variant here.
        "variants": [{
            "price": price,
            "compare_at_price": _price({"price": offer.get("highPrice")}),
            "available": _in_stock(offer),
            "options": {},
        }],
        "attributes_raw": [],
        "source_kind": "crawl",
        "source_ref": source_ref,
        "tenant_relations": {},
        "rating": None,
        "review_count": None,
        "featured_rank": None,
        # currency is the fallback the caller resolved (source config, or an
        # LLM inference over page context) -- a site does not reliably declare
        # its own currency in markup, and reading the wrong one off a page
        # would mis-price the catalogue invisibly.
        "currency": _text(offer.get("priceCurrency")).upper() or currency,
    }


async def _infer_currency(pages: list, tenant_id: str) -> str:
    """One LLM call, fired at most once per crawl, only when neither the
    source config nor any page's own markup states a currency. Never raises
    and never guesses from a bare symbol alone -- see CURRENCY_INFERENCE_PROMPT.
    Returns None on any failure or low-confidence answer, same as today's
    behaviour before this fallback existed.
    """
    text = "\n\n".join((p.get("text") or "")[:_MAX_CURRENCY_INFERENCE_CHARS // max(1, len(pages))]
                       for p in pages[:3]).strip()
    if not text:
        return None

    prompt = CURRENCY_INFERENCE_PROMPT.format(url=pages[0].get("url") or "", text=text)

    def _call():
        return openai_client.chat.completions.create(
            model=settings.LLM_MODEL,
            messages=[{"role": "user", "content": prompt}],
            max_completion_tokens=100,
            timeout=30.0,
            response_format={"type": "json_object"},
        )

    try:
        response = await asyncio.to_thread(_call)
        payload = json.loads(response.choices[0].message.content)
    except Exception as ex:
        logger.warning(f"Currency inference failed: {ex}")
        return None

    try:
        usage = getattr(response, "usage", None)
        if usage:
            insert_llm_usage(tenant_id, "Currency Inference", settings.LLM_MODEL,
                             usage.prompt_tokens, usage.completion_tokens, usage.total_tokens)
    except Exception as ex:
        logger.warning(f"Could not log currency inference usage: {ex}")

    currency = payload.get("currency") if isinstance(payload, dict) else None
    if not isinstance(currency, str) or len(currency) != 3:
        return None
    return currency.upper()


async def fetch_products(config: dict, source_ref: str, tenant_id: str, progress=None) -> list:
    """Crawl the merchant's site and return every product found, as source products.

    Raises rather than returning a short list when the crawl itself fails: the
    caller deletes products it did not see this run, so a half-finished crawl
    must never be mistaken for a shrunken catalogue.

    progress, if given, is called as progress(percent, step) as pages are
    visited -- best-effort, see extract_text_from_url.
    """
    url = config.get("url") or config.get("base_url")
    if not url:
        raise CrawlSourceError("source config must declare a url")

    # Re-checked on every run, not only at connect time: a stored URL is as
    # user-supplied as a freshly submitted one, and DNS can be repointed at an
    # internal address long after the source was saved.
    assert_safe_url(url)

    max_pages = int(config.get("max_pages") or settings.CRAWL_MAX_PAGES)
    seed_urls = await _discover_via_sitemap(url)
    if seed_urls:
        logger.info("Sitemap discovery found %d product url(s) for %s",
                    len(seed_urls), url)
    try:
        if seed_urls:
            result = await extract_text_from_url(
                url, max_depth=0, max_pages=max_pages, tenant_id=tenant_id,
                seed_urls=seed_urls[:max_pages], progress=progress)
        else:
            result = await extract_text_from_url(
                url, max_depth=20, max_pages=max_pages, tenant_id=tenant_id,
                progress=progress)
    except Exception as ex:
        raise CrawlSourceError(f"crawl of {url} failed: {type(ex).__name__}") from ex

    pages = result.get("pages") or []
    if not pages:
        raise CrawlSourceError(f"crawl of {url} returned no pages")

    pairs = []
    for page in pages:
        for product in extract_page_products(page, page.get("html", ""), tenant_id):
            pairs.append((product, page))

    # Batched across the whole crawl (see backfill_categories), not per page --
    # a crawl can be dozens of single-product pages, each with at most one
    # product missing a category, so batching here is what actually collapses
    # that into a handful of model calls instead of one per page.
    backfill_categories(pairs, tenant_id)
    found = [product for product, _ in pairs]

    currency = config.get("currency")
    if not currency and not any(_offer((p.get("raw") or {}).get("jsonld") or {}).get("priceCurrency")
                                for p in found):
        currency = await _infer_currency(pages, tenant_id)

    source_products = [sp for sp in
                       (to_source_product(p, source_ref, currency) for p in found)
                       if sp is not None]

    logger.info("Crawl of %s visited %d page(s) and found %d product(s)",
                url, len(pages), len(source_products))
    return source_products
