"""Turn a crawled page into product records.

Pure parsing: no database, no network. Two sources, in order of trust --
the page's own JSON-LD, then its Open Graph meta tags. A third source, a
language model reading the page text, lives in Task 4 and is only reached
when neither of these produces anything.

CTAs are never guessed. They are read straight off the page's own markup
via extract_ctas() -- exact, since a real button's URL is right there in
the HTML, rather than a language model's approximation of it.
"""
import hashlib
import json
import logging
import re
from functools import partial
from urllib.parse import parse_qsl, urlencode, urljoin, urlparse, urlunparse

from bs4 import BeautifulSoup

from app.core.config import settings
from app.core.llm_client import client as openai_client
from app.core.prompts import CATEGORY_BACKFILL_PROMPT, PRODUCT_EXTRACTION_PROMPT
from app.services.infra.database import insert_llm_usage

logger = logging.getLogger(__name__)

# What the frontend styles against. A CTA whose type is not in this set is
# dropped rather than passed through, so the contract cannot drift.
# "visit" is a member, not an escape hatch: an unrecognised CTA is kept
# under this type rather than dropped, since a real button we can't name is
# still a working link -- and "visit" tells the frontend what to actually
# do with it (render a plain, clickable link) rather than reading as an
# admission of defeat the way "other" would. It is distinct from "view",
# which specifically means the product's own detail page; "visit" is any
# other real link on the page that isn't one of the specific action types.
CTA_TYPES = {"add_to_cart", "buy_now", "view", "enquire", "book_demo",
             "donate", "subscribe", "visit"}

_SKIPPED_SCHEMES = ("javascript:", "mailto:", "tel:")

_MAX_CTAS = 6
# A label that needs more than this to say its piece is not a button
# label. Kept deliberately short so a bad extraction (a whole card's text
# flattened into one string) is obviously wrong rather than plausible.
_MAX_LABEL_CHARS = 60

_HEADING_TAGS = ("h1", "h2", "h3", "h4", "h5")

# Ordered (type, needles) rules matched against "href label" lowercased.
# First match wins, so more specific phrases are listed before generic ones.
# The enquire needles deliberately exclude the bare word "contact": an
# ordinary "Contact Us" nav link is not an enquiry about the product on the
# page, and a bare substring match also misfires on unrelated product names.
_CTA_RULES = [
    ("add_to_cart", ("add-to-cart", "/cart/add", "?add_to_cart", "add to cart",
                      "add to bag", "add to basket")),
    ("buy_now", ("buy now", "buy-now", "purchase now")),
    ("book_demo", ("book a demo", "book demo", "schedule a demo", "request a demo")),
    ("donate", ("donate",)),
    ("subscribe", ("subscribe",)),
    ("enquire", ("contact us", "get in touch", "enquire", "enquiry",
                 "request a quote", "get started", "start free", "talk to us",
                 "talk to sales")),
    ("view", ("learn more", "view details", "see more", "read more", "explore")),
]

# A card whose only links are all unclassified is noise: six near-identical
# "visit" buttons tell a visitor nothing that one or two didn't already.
_MAX_VISIT_ONLY_CTAS = 2

# Tags that are page chrome rather than content -- the same set
# app/services/website.py decomposes before extracting page text, for the
# same reason: nav/header/footer/aside links are not what the page is
# about, and letting them fill the CTA cap would crowd out real buttons.
_CHROME_TAGS = ("nav", "header", "footer", "aside")

# Rank bands for the cap: real product actions first, then the bare "view"
# fallback type, then anything unclassified. Document order is preserved
# within each band, so a late add_to_cart still beats an early visit.
_CTA_RANK = {
    "add_to_cart": 0, "buy_now": 0, "book_demo": 0, "donate": 0,
    "subscribe": 0, "enquire": 0,
    "view": 1,
    "visit": 2,
}

# Runs of pipes (with surrounding whitespace) collapse to one space, so
# taxonomy terms mashed together by markup ("Erin Recommends|Clothing")
# read as words rather than a raw delimiter.
_SEPARATOR_RUN_RE = re.compile(r"\s*\|+\s*")

# Path keywords that mark a link as an action anywhere on the site, even
# when it is not this product's own page and not on this product's own
# host's product area -- /contact, /cart, /checkout, /enquiry, /book,
# /demo, /donate, /subscribe. This is what keeps a genuine "Get Started"
# -> /contact CTA alive despite living on a different path than the
# product page it is attached to.
_ACTION_PATH_KEYWORDS = ("contact", "cart", "checkout", "enquiry", "enquire",
                         "book", "demo", "donate", "subscribe")

# Class/id/data-section-type substrings that mark an element as a
# cross-sell or related-products rail on any major platform -- decomposed
# before CTA/option extraction ever runs, the same way chrome is. This is
# what actually prevents the wrong-product bug (a related product's own
# add-to-cart button posting back to *this* page): scoping by where the
# element lives in the DOM, rather than trying to recognise its wording,
# is the only thing that survives a site using different phrasing
# ("Add to basket" instead of "Add to cart") or a different language.
_RELATED_CONTAINER_MARKERS = ("related", "upsell", "cross-sell", "crosssell",
                              "also-bought", "recommend", "you-may-also")

# Matched as a whole word/token, not a raw substring: a WooCommerce
# category slug like "product_cat-erin-recommendsclothing-..." (a real
# category literally named "Erin Recommends" on one crawled site) contains
# "recommend" as a substring of "recommendsclothing" without being a
# recommendations rail at all. \b prevents that false positive while still
# matching the marker as its own hyphen-delimited word.
_RELATED_CONTAINER_RE = re.compile(
    r"\b(" + "|".join(re.escape(m) for m in _RELATED_CONTAINER_MARKERS) + r")\b",
    re.IGNORECASE)


def _is_related_container(tag) -> bool:
    haystack = " ".join(tag.get("class") or [])
    haystack += " " + (tag.get("id") or "")
    haystack += " " + (tag.get("data-section-type") or "")
    return bool(_RELATED_CONTAINER_RE.search(haystack))


def _decompose_chrome(soup) -> None:
    """Removes page chrome (nav/header/footer/aside, role=navigation) and
    cross-sell/related-products rails from soup, in place. Shared by CTA
    and option extraction so both see the same notion of "the page's own
    content"."""
    for chrome in soup.find_all(_CHROME_TAGS):
        chrome.decompose()
    for chrome in soup.find_all(attrs={"role": "navigation"}):
        chrome.decompose()
    for related in soup.find_all(_is_related_container):
        related.decompose()


# Class/id/itemtype substrings that mark an element as the product's own
# detail area -- WooCommerce's "summary entry-summary", and the generic
# product-main/product-detail/product-info/product-single naming several
# other platforms use (books.toscrape.com spells the first one
# "product_main", underscore rather than hyphen -- _is_product_container
# normalises that away rather than doubling this list), plus schema.org's
# itemtype="...Product" microdata.
_PRODUCT_CONTAINER_MARKERS = ("product-main", "product-detail", "product-info",
                              "product-single", "summary", "entry-summary")

# Matched as a whole word/token (see _RELATED_CONTAINER_RE for why: a
# generated class name can contain a marker as a substring of an unrelated
# longer word, e.g. a category slug containing "...info..." with no
# separator, without the element being a product-info block at all).
_PRODUCT_CONTAINER_RE = re.compile(
    r"\b(" + "|".join(re.escape(m) for m in _PRODUCT_CONTAINER_MARKERS) + r")\b",
    re.IGNORECASE)

# Tried only when nothing above matches anywhere on the page. A generic
# landmark is still a far tighter scope than the whole document, and it is
# what covers a plain-HTML site that marks no product-specific container
# at all.
_PRODUCT_CONTAINER_FALLBACK_TAGS = ("main", "article")


def _is_product_container(tag) -> bool:
    if "Product" in (tag.get("itemtype") or ""):
        return True
    haystack = " ".join(tag.get("class") or [])
    haystack += " " + (tag.get("id") or "")
    # "product_main" and "product-main" are the same convention with a
    # different separator -- normalise before matching so one rule covers
    # both rather than listing every spelling.
    haystack = haystack.replace("_", "-")
    return bool(_PRODUCT_CONTAINER_RE.search(haystack))


def _find_product_container(soup):
    """The product's own detail area, if the page marks one -- CTAs and
    options are read from inside it when found, which is what keeps
    category/navigation links (books.toscrape.com's "Books", "Poetry" ...)
    out without needing to know what a category link looks like.

    Tried in order: an element matching _is_product_container, then a
    generic <main>/<article> landmark, then None (the whole page). find()
    returns the first match in document order, which for nested markup is
    the outermost (BeautifulSoup visits parents before children).
    """
    container = soup.find(_is_product_container)
    if container is not None:
        return container
    return soup.find(_PRODUCT_CONTAINER_FALLBACK_TAGS)


def product_key_for(url: str, sku: str = None) -> str:
    """A stable id so re-ingestion updates a product instead of duplicating it.

    Prefers the SKU, which survives a site restructure. Falls back to a hash of
    the URL, which does not, but is better than nothing.
    """
    if sku:
        return f"sku:{str(sku).strip()}"
    digest = hashlib.sha1((url or "").strip().lower().encode()).hexdigest()[:16]
    return f"url:{digest}"


def _slugify(name: str) -> str:
    """A stable, readable fragment of a name for use as part of a key."""
    slug = re.sub(r"[^a-z0-9]+", "-", (name or "").strip().lower()).strip("-")
    return slug or "unnamed"


def product_key_for_name(name: str) -> str:
    """A stable id for a product identified only by its name.

    The model reads prose, not a per-product page or a SKU, so the same
    product mentioned on several pages (a homepage teaser and its own page,
    say) must collapse onto one key rather than minting a new one per page
    it happens to be mentioned on. The products table is per-tenant (its
    own schema), so a name collision across tenants is not a concern here.
    """
    return f"name:{_slugify(name)}"


def _first(value):
    """JSON-LD lets almost any field be a value or a list of them."""
    if isinstance(value, list):
        return value[0] if value else None
    return value


def _text_of(value) -> str:
    """A field that may be a string, a list, or a nested {'@value': ...}."""
    value = _first(value)
    if isinstance(value, dict):
        value = value.get("@value") or value.get("name") or value.get("url")
    return str(value).strip() if value else ""


def _classify_cta(href: str, label: str) -> str:
    haystack = f"{href} {label}".lower()
    for cta_type, needles in _CTA_RULES:
        if any(needle in haystack for needle in needles):
            return cta_type
    return "visit"


def _href_of(tag):
    """The URL an element points at, from whichever attribute carries it."""
    return tag.get("href") or tag.get("formaction") or tag.get("data-href")


# Hidden form fields that identify the product an add-to-cart form submits.
# Carried over onto the resolved URL so it is actually actionable rather
# than a bare page link -- WooCommerce's own add-to-cart form uses exactly
# this pair.
_FORM_PRODUCT_FIELDS = ("add-to-cart", "product_id")

# The WooCommerce class its own add-to-cart button always carries -- a
# second, class-based way to recognise product identity for forms that
# submit the SKU some other way and carry neither hidden field above.
_ADD_TO_CART_BUTTON_CLASS = "single_add_to_cart_button"

# A plain <button type="submit"> whose visible text is exactly one of
# these, verbatim (case-insensitive, not a substring), is site furniture --
# a search box, a newsletter signup, a generic form -- unless the form it
# submits carries product identity. Matching the full trimmed text rather
# than a substring keeps a real "Send enquiry" button safe from "send".
_GENERIC_FORM_VERBS = {"submit", "search", "go", "send",
                       "subscribe to newsletter", "sign up for updates"}


def _form_has_product_identity(tag, form) -> bool:
    """Whether this button or its enclosing form carries evidence it is a
    real add-to-cart action: a hidden add-to-cart/product_id input, or
    WooCommerce's own single_add_to_cart_button class on the button."""
    if _ADD_TO_CART_BUTTON_CLASS in (tag.get("class") or []):
        return True
    return any(form.find("input", attrs={"name": field_name})
              for field_name in _FORM_PRODUCT_FIELDS)


def _form_action_href(tag, page_url: str) -> str:
    """The URL a plain `<button type="submit">` resolves to when it has no
    href/formaction/data-href of its own -- its enclosing `<form>`'s
    action, falling back to the page URL when the form has none, which is
    what a browser does. The product-identifying hidden inputs the form
    carries (add-to-cart, product_id) are folded into the query string, so
    the result is an actionable URL and not just a link back to the page.

    This is the standard e-commerce add-to-cart pattern: a variable
    product's "Add to cart" button is a form submit, not a link, so
    without this it is invisible to CTA extraction entirely. But the same
    pattern also covers a page's search box or newsletter signup, which
    are not product actions -- so a generic verb with no product identity
    (see _form_has_product_identity) returns None here rather than a URL,
    keeping it out of CTA extraction altogether rather than relying on
    filter_ctas_for_product to catch it later.
    """
    if tag.name != "button":
        return None
    button_type = (tag.get("type") or "submit").strip().lower()
    if button_type != "submit":
        return None
    form = tag.find_parent("form")
    if form is None:
        return None

    if not _form_has_product_identity(tag, form):
        label_text = tag.get_text(strip=True).strip().lower()
        if label_text in _GENERIC_FORM_VERBS:
            return None

    action = (form.get("action") or "").strip()
    target = urljoin(page_url or "", action) if action else (page_url or "")
    if not target:
        return None

    extra_params = []
    for field_name in _FORM_PRODUCT_FIELDS:
        field = form.find("input", attrs={"name": field_name})
        value = (field.get("value") or "").strip() if field else ""
        if value:
            extra_params.append((field_name, value))
    if not extra_params:
        return target

    parts = urlparse(target)
    query = parse_qsl(parts.query, keep_blank_values=True)
    query.extend(extra_params)
    return urlunparse(parts._replace(query=urlencode(query)))


def _label_of(tag) -> str:
    """The element's own words for a button, in order of how deliberately
    the author chose them for exactly this purpose.

    A plain "Add to cart" button's flattened text is fine. But a
    card-shaped anchor -- a whole product card wrapped in one <a>,
    heading and blurb included -- flattens into a run-on sentence that
    reads as a bug when rendered on a button. aria-label and title exist
    precisely to give an element accessible/tooltip text distinct from its
    visual content, so they win when present. A nested heading is the next
    best proxy for "the name of the thing this points to." Only after all
    of those are absent do we fall back to the raw flattened text.

    Every candidate is still the site's own words -- this only chooses
    which of them to surface.
    """
    aria_label = (tag.get("aria-label") or "").strip()
    if aria_label:
        label = aria_label
    else:
        title = (tag.get("title") or "").strip()
        if title:
            label = title
        else:
            heading = tag.find(_HEADING_TAGS)
            heading_text = heading.get_text(separator=" ", strip=True) if heading else ""
            if heading_text:
                label = heading_text
            else:
                label = tag.get_text(separator=" ", strip=True)

    label = " ".join(label.split())
    return label[:_MAX_LABEL_CHARS]


def _clean_label(label: str) -> str:
    """Strip structural noise a raw label can carry that a visitor should
    never see: pipe-separated taxonomy terms mashed together by markup
    ("Erin Recommends|Clothing") collapse to a single space so they read
    as words rather than a raw delimiter.

    Cross-product carousel buttons are no longer handled by inspecting the
    label at all -- see _decompose_chrome / _RELATED_CONTAINER_MARKERS,
    which removes the whole related-products rail structurally before a
    candidate is even collected, regardless of what a given site's button
    happens to say ("Add to cart", "Add to basket", or anything else).
    """
    label = label or ""
    label = _SEPARATOR_RUN_RE.sub(" ", label)
    return label.strip()


def _cta_candidates(html: str, page_url: str) -> list:
    """<a>/<button> elements on the page, turned into unranked, uncapped CTA
    candidates -- the raw material both extract_ctas and _ctas_for rank and
    cap from.

    Looks at <a> tags with an href, and <button> tags carrying a formaction
    or data-href -- or, failing that, a plain `<button type="submit">`
    inside a `<form>`, resolved via _form_action_href to the form's own
    action. That last case is the standard e-commerce add-to-cart pattern:
    without it, a variable product's add-to-cart button (a form submit,
    not a link) is invisible here.

    Scoped to the page's own content: chrome (nav/header/footer/aside,
    role=navigation) and cross-sell/related-products rails are decomposed
    first -- see _decompose_chrome -- and when the page marks its own
    product detail area (see _find_product_container), only elements
    inside it are considered at all, so a books.toscrape.com-style
    category link elsewhere on the page is never a candidate in the first
    place.

    Labels are kept raw (not yet cleaned). Never raises: malformed input
    just yields [].
    """
    try:
        soup = BeautifulSoup(html or "", "html.parser")
        _decompose_chrome(soup)
        scope = _find_product_container(soup) or soup
        elements = scope.find_all(["a", "button"])
    except Exception as ex:
        logger.warning(f"CTA extraction failed for {page_url}: {ex}")
        return []

    candidates = []
    seen_urls = set()
    for tag in elements:
        raw_href = _href_of(tag) or _form_action_href(tag, page_url)
        if not raw_href:
            continue
        raw_href = raw_href.strip()
        if not raw_href or raw_href == "#" or raw_href.startswith("#"):
            continue
        if raw_href.lower().startswith(_SKIPPED_SCHEMES):
            continue

        label = _label_of(tag)
        if not label:
            continue

        try:
            url = urljoin(page_url or "", raw_href)
        except Exception:
            continue
        if not url or url in seen_urls:
            continue

        cta_type = _classify_cta(raw_href, label)
        candidates.append({"type": cta_type, "label": label, "url": url})
        seen_urls.add(url)

    return candidates


def _rank_and_cap(candidates: list) -> list:
    """Real product actions first, then "view", then "visit", before the cap
    is applied, so a genuine add_to_cart late in the document still beats a
    generic link found earlier. Document order is preserved within each
    band. Labels are cleaned here -- once a candidate has survived to be
    returned, whatever structural noise it carried for classification
    purposes has done its job and should not reach a visitor.
    """
    ranked = sorted(enumerate(candidates),
                     key=lambda pair: (_CTA_RANK[pair[1]["type"]], pair[0]))
    capped = []
    for _, cta in ranked[:_MAX_CTAS]:
        capped.append({**cta, "label": _clean_label(cta["label"])})
    return capped


def extract_ctas(html: str, page_url: str) -> list:
    """CTAs read straight off the page's own markup.

    Exact, since the URL and label come verbatim from the site rather than
    being guessed. Never raises: malformed input just yields [].

    This is the page-only view: no product context, so it cannot tell a
    sibling product's link from this product's own (see
    filter_ctas_for_product, applied by _ctas_for once a product_url is
    known).
    """
    return _rank_and_cap(_cta_candidates(html, page_url))


def _looks_like_action_path(url: str) -> bool:
    """Whether url's path reads like an action anywhere on the site --
    /contact, /cart, /checkout, /enquiry, /book, /demo, /donate,
    /subscribe -- rather than a listing or another product's page."""
    segments = [s for s in urlparse(url or "").path.lower().split("/") if s]
    return any(keyword in segment for segment in segments
              for keyword in _ACTION_PATH_KEYWORDS)


def filter_ctas_for_product(ctas: list, product_url: str, product_name: str = "") -> list:
    """Keep a CTA only if it is an action on this product, or an action
    elsewhere -- same-site navigation (sibling products, category pages,
    listings) is dropped.

    This is a second line of defence, not the primary mechanism: the
    related-products rail that used to slip through here is now removed
    structurally, before a CTA is even collected (see _decompose_chrome /
    _RELATED_CONTAINER_MARKERS in _cta_candidates), which is what actually
    survives a site wording its button differently ("Add to basket") or in
    another language -- matching that wording here never did. What is left
    is path-based housekeeping for whatever a page's markup does not mark
    as a related-products container but is still just navigation.

    Rules, in the order they are checked:

    1. Same path as the product's own URL -- keep (an action on this
       product: ?add-to-cart=, #reviews, a form post back to itself).
    2. A different host entirely -- keep (an external booking/store link
       is a real action).
    3. A path that looks like an action anywhere on the site -- keep.
    4. Anything else (a different path, same host) is navigation -- a
       sibling product, a category page, a listing. Drop.
    """
    own = urlparse(product_url or "")
    own_path = own.path.rstrip("/")

    kept = []
    for cta in ctas:
        cta_parts = urlparse(cta.get("url") or "")
        if cta_parts.path.rstrip("/") == own_path:
            kept.append(cta)
            continue
        if own.netloc and cta_parts.netloc and cta_parts.netloc != own.netloc:
            kept.append(cta)
            continue
        if _looks_like_action_path(cta.get("url") or ""):
            kept.append(cta)
            continue
        # A different path on the same host: navigation. Drop.

    return kept


def _ctas_for(html: str, page_url: str, product_url: str, product_name: str = "") -> list:
    """CTAs from the page, filtered to genuine actions on this product or
    elsewhere on the site, or a bare "view" fallback so a product always
    has at least one way to reach it.

    Filtering happens on the uncapped candidate list, before ranking and
    the six-CTA cap -- a page whose related-products rail contributes many
    "add to cart" candidates must not crowd this product's own button out
    of the cap before filter_ctas_for_product ever gets to see it.

    When every surviving CTA is unclassified ("visit"), only the first
    _MAX_VISIT_ONLY_CTAS survive -- a card whose only actions are
    indistinguishable links is better served by a couple of them than by
    all six.
    """
    candidates = filter_ctas_for_product(
        _cta_candidates(html, page_url), product_url, product_name)
    ctas = _rank_and_cap(candidates)
    if not ctas:
        return [{"type": "view", "label": "View details", "url": product_url}]
    if all(cta["type"] == "visit" for cta in ctas):
        return ctas[:_MAX_VISIT_ONLY_CTAS]
    return ctas


# A page carrying more than this many distinct option groups, or an option
# offering more values than this, is not a normal size/colour picker --
# capped defensively rather than trusted to whatever a page's markup
# happens to contain.
_MAX_OPTIONS = 10
_MAX_OPTION_VALUES = 30

# A <select>'s name/id is treated as a product-attribute picker when it
# looks like WooCommerce's `attribute_*` / `pa_*` convention or the
# generic `option*` one Shopify and others use.
_OPTION_KEY_RE = re.compile(r"^(attribute_|option|pa_)", re.IGNORECASE)

# Stripped off an option key before it is title-cased into a display name:
# "attribute_pa_color" / "attribute_size" / "pa_color" -> "color" / "size".
_OPTION_PREFIX_STRIP_RE = re.compile(r"^(attribute_pa_|attribute_|pa_|option)", re.IGNORECASE)

# An <option> whose text opens with one of these is a placeholder ("Choose
# an option", "Select a size"), not a real, selectable value.
_PLACEHOLDER_OPTION_RE = re.compile(r"^(choose|select)\b", re.IGNORECASE)


def _option_name_from_key(key: str) -> str:
    """"attribute_size" -> "Size", "attribute_pa_color" -> "Color": strip
    the platform's structural prefix and title-case what is left."""
    name = _OPTION_PREFIX_STRIP_RE.sub("", key or "")
    name = name.replace("_", " ").replace("-", " ").strip()
    return name.title() if name else (key or "").strip().title()


def _is_placeholder_option(text: str, value: str) -> bool:
    """An <option> with no real value, or whose text reads as an
    instruction ("Choose an option") rather than a selectable value."""
    text = (text or "").strip()
    if not text or not (value or "").strip():
        return True
    return bool(_PLACEHOLDER_OPTION_RE.match(text))


def _options_from_selects(soup) -> list:
    """<select> elements shaped like a product attribute picker --
    WooCommerce's attribute_*/pa_* naming and Shopify's option* both match.
    Placeholder entries ("Choose an option") are dropped."""
    options = []
    seen_names = set()
    for select in soup.find_all("select"):
        key = select.get("name") or select.get("id") or ""
        if not _OPTION_KEY_RE.match(key):
            continue

        values = []
        for opt in select.find_all("option"):
            text = opt.get_text(strip=True)
            value = opt.get("value") or ""
            if _is_placeholder_option(text, value):
                continue
            if text not in values:
                values.append(text)
        if not values:
            continue

        name = _option_name_from_key(key)
        if name.lower() in seen_names:
            continue
        seen_names.add(name.lower())
        options.append({"name": name, "values": values[:_MAX_OPTION_VALUES]})

    return options


def _control_label_text(soup, el) -> str:
    """Visible label text for a form control, checked in order: an
    associated <label for=id>, a <label> the control is nested inside, an
    aria-label on the control itself, or the nearest preceding <label>
    sibling. "" when none of those exist."""
    el_id = el.get("id")
    if el_id:
        label = soup.find("label", attrs={"for": el_id})
        if label:
            text = label.get_text(strip=True)
            if text:
                return text
    parent_label = el.find_parent("label")
    if parent_label:
        text = parent_label.get_text(strip=True)
        if text:
            return text
    aria = (el.get("aria-label") or "").strip()
    if aria:
        return aria
    previous_label = el.find_previous_sibling("label")
    if previous_label:
        return previous_label.get_text(strip=True)
    return ""


def _options_from_radios(soup) -> list:
    """Radio groups sharing a name -- a common way to render size/colour
    swatches without a <select>. A single radio on its own is a yes/no
    toggle, not a set of variant choices, so a group needs at least two
    distinct values to count."""
    groups, order = {}, []
    for radio in soup.find_all("input", attrs={"type": "radio"}):
        name = (radio.get("name") or "").strip()
        if not name:
            continue
        if name not in groups:
            groups[name] = []
            order.append(name)

        value = _control_label_text(soup, radio) or (radio.get("value") or "").strip()
        if value and value not in groups[name]:
            groups[name].append(value)

    options = []
    for name in order:
        values = groups[name]
        if len(values) < 2:
            continue
        options.append({"name": _option_name_from_key(name),
                        "values": values[:_MAX_OPTION_VALUES]})
    return options


def _options_from_data_attrs(soup) -> list:
    """Elements carrying data-attribute_name -- the pattern WooCommerce
    variation-swatch plugins use for swatch grids that are neither a
    <select> nor plain radios."""
    groups, order = {}, []
    for el in soup.find_all(attrs={"data-attribute_name": True}):
        key = (el.get("data-attribute_name") or "").strip()
        if not key:
            continue
        if key not in groups:
            groups[key] = []
            order.append(key)

        value = (el.get("data-value") or el.get_text(strip=True) or "").strip()
        if value and value not in groups[key]:
            groups[key].append(value)

    return [{"name": _option_name_from_key(key), "values": groups[key][:_MAX_OPTION_VALUES]}
            for key in order if groups[key]]


# Hidden fields, or an action, that mark a <form> as the product's own
# add-to-cart form -- the same identity signal _form_has_product_identity
# uses for CTAs, plus the bare "id" field some Shopify themes submit and
# an action containing "/cart" for forms that identify the product neither
# way but still clearly post to a cart endpoint.
_PRODUCT_FORM_HIDDEN_FIELDS = ("add-to-cart", "product_id", "id")


def _is_product_form(form) -> bool:
    """Whether form is the product's own add-to-cart form: a
    single_add_to_cart_button control, a hidden add-to-cart/product_id/id
    input, or an action containing "/cart" -- the structural signature of
    an add-to-cart form on WooCommerce, Shopify and sites like them,
    regardless of what its button says or what language the site is in.
    """
    if form.find(class_=_ADD_TO_CART_BUTTON_CLASS) is not None:
        return True
    if any(form.find("input", attrs={"name": field_name})
          for field_name in _PRODUCT_FORM_HIDDEN_FIELDS):
        return True
    return "/cart" in (form.get("action") or "").lower()


def _find_product_form(soup):
    """The product's own add-to-cart form, if the page has one. Trying
    this before the name-based reader below is what makes an unfamiliar
    platform's option markup legible without knowing its naming
    convention: whatever <select>/radio group lives inside this specific
    form is a product option, whatever it happens to be called."""
    for form in soup.find_all("form"):
        if _is_product_form(form):
            return form
    return None


def _options_from_form(form, soup) -> list:
    """Every <select> and radio group inside the product's own add-to-cart
    form is a product option -- no naming convention required, because
    being inside this specific form already is the evidence. The option's
    name is the control's own label (see _control_label_text) when the
    page provides one, falling back to the name/id-key heuristic only when
    it does not."""
    options = []
    seen = set()

    for select in form.find_all("select"):
        values = []
        for opt in select.find_all("option"):
            text = opt.get_text(strip=True)
            value = opt.get("value") or ""
            if _is_placeholder_option(text, value):
                continue
            if text not in values:
                values.append(text)
        if not values:
            continue

        key = select.get("name") or select.get("id") or ""
        name = _control_label_text(soup, select) or _option_name_from_key(key)
        if name.lower() in seen:
            continue
        seen.add(name.lower())
        options.append({"name": name, "values": values[:_MAX_OPTION_VALUES]})

    groups, order = {}, []
    for radio in form.find_all("input", attrs={"type": "radio"}):
        group_key = (radio.get("name") or "").strip()
        if not group_key:
            continue
        if group_key not in groups:
            groups[group_key] = []
            order.append(group_key)
        value = _control_label_text(soup, radio) or (radio.get("value") or "").strip()
        if value and value not in groups[group_key]:
            groups[group_key].append(value)

    for group_key in order:
        values = groups[group_key]
        if len(values) < 2:
            continue
        name = _option_name_from_key(group_key)
        if name.lower() in seen:
            continue
        seen.add(name.lower())
        options.append({"name": name, "values": values[:_MAX_OPTION_VALUES]})

    return options[:_MAX_OPTIONS]


def extract_options(html: str) -> list:
    """Product options (size, colour, ...) read deterministically off the
    page's own markup.

    Tried in order:
    1. Every <select>/radio group inside the product's own add-to-cart
       form (see _find_product_form / _options_from_form) -- this is what
       reads an unfamiliar platform's options correctly without knowing
       its naming convention, because scoping by "lives inside this form"
       does not depend on what the form's fields happen to be called.
    2. A name-based fallback for pages with no such form: <select>
       elements shaped like a product attribute (attribute_*/pa_*/option*),
       radio groups sharing a name, and elements carrying
       data-attribute_name.

    Either way, values are taken verbatim, placeholder entries ("Choose an
    option") are dropped, and cross-sell/related-products rails have
    already been removed (see _decompose_chrome) so a related product's
    own option markup can never be mistaken for this one's.

    Never raises: malformed input, or a page with no option markup at all,
    yields [].
    """
    try:
        soup = BeautifulSoup(html or "", "html.parser")
        _decompose_chrome(soup)
    except Exception as ex:
        logger.warning(f"Option extraction failed: {ex}")
        return []

    form = _find_product_form(soup)
    if form is not None:
        options = _options_from_form(form, soup)
        if options:
            return options

    options = _options_from_selects(soup)
    seen = {o["name"].lower() for o in options}

    for extra in (_options_from_radios(soup) + _options_from_data_attrs(soup)):
        if extra["name"].lower() in seen:
            continue
        seen.add(extra["name"].lower())
        options.append(extra)

    return options[:_MAX_OPTIONS]


# Query-parameter keys that are just an opaque identifier rather than a
# meaningful attribute name -- a variant group keyed on one of these is
# named "Options" rather than, say, "Variant".
_GENERIC_VARIANT_PARAM_KEYS = {"variant", "variant_id", "id", "v", "vid",
                               "sku", "product_id"}


def _option_name_from_query_key(key: str) -> str:
    if (key or "").strip().lower() in _GENERIC_VARIANT_PARAM_KEYS:
        return "Options"
    return _option_name_from_key(key)


def _group_variant_ctas(ctas: list, product_url: str) -> tuple:
    """Finds a query-parameter variant picker hiding among a product's own
    CTAs -- the pattern a JavaScript storefront with no <select> and no
    add-to-cart form (Shopify Hydrogen and similar) renders size/colour
    choices as: plain links back to the product's own page that differ
    from each other only in the value of one query parameter (?variant=...).
    Those are not distinct calls to action; they are one option rendered
    as buttons.

    A group needs at least two members and exactly one varying query key
    to count -- a lone link differing from nothing is not a picker, and a
    pair that differs in more than one parameter is not this pattern.
    Values are the CTAs' own link text, taken in document order.

    Returns (options, cta_urls_to_remove) -- options is a single-entry (or
    empty) list; the caller removes the named URLs from the CTA list it
    already has, rather than this function mutating anything.
    """
    own = urlparse(product_url or "")
    own_path = own.path.rstrip("/")

    groups, order = {}, []
    for cta in ctas:
        parts = urlparse(cta.get("url") or "")
        if parts.path.rstrip("/") != own_path:
            continue
        if own.netloc and parts.netloc and parts.netloc != own.netloc:
            continue
        query_pairs = parse_qsl(parts.query, keep_blank_values=True)
        if not query_pairs:
            continue
        key_set = tuple(sorted(k for k, _ in query_pairs))
        if key_set not in groups:
            groups[key_set] = []
            order.append(key_set)
        groups[key_set].append((cta, dict(query_pairs)))

    options = []
    excluded = set()
    for key_set in order:
        members = groups[key_set]
        if len(members) < 2:
            continue

        varying_keys = [key for key in key_set
                        if len({query[key] for _, query in members}) > 1]
        if len(varying_keys) != 1:
            continue

        values = []
        for cta, _ in members:
            label = (cta.get("label") or "").strip()
            if label and label not in values:
                values.append(label)
        if len(values) < 2:
            continue

        options.append({"name": _option_name_from_query_key(varying_keys[0]),
                        "values": values[:_MAX_OPTION_VALUES]})
        excluded.update(cta["url"] for cta, _ in members)
        break  # one variant picker per page is what every platform we've seen renders

    return options[:_MAX_OPTIONS], excluded


def _ctas_and_options(html: str, page_url: str, product_url: str,
                      product_name: str = "") -> tuple:
    """The product's final (ctas, options) together, because the two are
    not always independent: when extract_options finds nothing (no
    <select>, no add-to-cart form to scope options to) and the CTAs
    themselves contain a query-parameter variant picker -- see
    _group_variant_ctas -- that group is folded into options and removed
    from the CTA list, so a visitor sees "Size: 154cm, 158cm" rather than
    two buttons labelled "154cm" and "158cm" with nowhere useful to go.

    Left alone whenever extract_options already found something (a
    <select> or add-to-cart-form reader already answered the question) or
    the CTAs contain no such group.
    """
    options = extract_options(html)
    ctas = _ctas_for(html, page_url, product_url, product_name)

    if not options:
        variant_options, excluded_urls = _group_variant_ctas(ctas, product_url)
        if variant_options:
            options = variant_options
            ctas = [cta for cta in ctas if cta["url"] not in excluded_urls]
            if not ctas:
                ctas = [{"type": "view", "label": "View details", "url": product_url}]

    return ctas, options


def _sanitize_llm_options(raw) -> list:
    """The model's options, cleaned to the same shape extract_options
    produces. Display text only -- these values are never used to build a
    URL, so a hallucinated one is cosmetic at worst rather than a broken
    link."""
    options = []
    for entry in (raw or [])[:_MAX_OPTIONS]:
        if not isinstance(entry, dict):
            continue
        name = str(entry.get("name") or "").strip()
        values_raw = entry.get("values")
        if not name or not isinstance(values_raw, list):
            continue

        values = []
        for value in values_raw[:_MAX_OPTION_VALUES]:
            text = str(value).strip()
            if text and text not in values:
                values.append(text)
        if values:
            options.append({"name": name, "values": values})

    return options


# Meta tags checked, in order, for a product image. og:image is the
# convention essentially every site with any social-sharing setup already
# populates; the others are progressively less common fallbacks.
_IMAGE_META_PROPS = ("og:image", "og:image:secure_url", "twitter:image")


def _image_for(html_or_soup, page_url: str, page: dict = None) -> str:
    """The page's own product image, read deterministically -- never guessed
    and never left to the model.

    Checked in order: og:image, og:image:secure_url, twitter:image, then the
    first entry in page["images"] -- the crawler already collected and
    filtered that list (icons and social-network chrome removed, URLs
    already resolved to absolute) while building the page, so it is reused
    here rather than re-parsed. Relative meta values are resolved to
    absolute against page_url. Returns None when nothing usable exists.

    Accepts either a BeautifulSoup already built by the caller or a raw HTML
    string, so callers that already have a soup (extract_from_meta) are not
    made to parse the page twice.
    """
    try:
        soup = (html_or_soup if isinstance(html_or_soup, BeautifulSoup)
                else BeautifulSoup(html_or_soup or "", "html.parser"))
        for prop in _IMAGE_META_PROPS:
            value = _meta_content(soup, prop)
            if value:
                return urljoin(page_url or "", value)
    except Exception as ex:
        logger.warning(f"Image extraction failed for {page_url}: {ex}")

    for image in (page or {}).get("images") or []:
        url = (image.get("url") if isinstance(image, dict) else None) or ""
        if url:
            return urljoin(page_url or "", url)

    return None


def extract_from_jsonld(page: dict, html: str = "") -> list:
    """Products from the page's structured data. Exact; never guessed.

    html defaults to "" for callers that only have JSON-LD -- an empty
    string means "no page markup available", which extract_ctas treats as
    zero CTAs, falling back to the bare "view" CTA below. Not an oversight.
    """
    products = []
    for block in page.get("jsonld") or []:
        types = block.get("@type")
        types = types if isinstance(types, list) else [types]
        type_names = [str(t) for t in types if t]
        # ProductGroup is Shopify's schema for a product with variants -- name,
        # description, brand and category sit at the same top-level keys a
        # plain Product uses, it just has no top-level offers of its own
        # (each variant in hasVariant carries its own). Treating it as
        # unrecognised silently dropped brand/category/price to the LLM tier,
        # which by design never supplies price or brand.
        if not any(t in ("Product", "ProductGroup") for t in type_names):
            continue

        name = _text_of(block.get("name"))
        if not name:
            # A product with no name is unusable in a card.
            continue

        offer = _first(block.get("offers")) or {}
        if not isinstance(offer, dict):
            offer = {}

        url = _text_of(offer.get("url")) or page.get("url") or ""
        sku = _text_of(block.get("sku")) or _text_of(block.get("mpn"))

        raw = {"jsonld": block}
        if sku:
            raw["sku"] = sku

        ctas, options = _ctas_and_options(html, page.get("url") or "", url, name)

        products.append({
            "product_key": product_key_for(url, sku),
            "name": name,
            "description": _text_of(block.get("description")),
            "image_url": (_text_of(block.get("image")) or
                          _image_for(html, page.get("url") or "", page)),
            "product_url": url,
            "category": _text_of(block.get("category")) or None,
            "ctas": ctas,
            "options": options,
            "raw": raw,
        })
    return products


def _meta_content(soup, prop: str) -> str:
    """One Open Graph value. Sites use `property` or `name` interchangeably."""
    tag = soup.find("meta", attrs={"property": prop}) or \
          soup.find("meta", attrs={"name": prop})
    return (tag.get("content") or "").strip() if tag else ""


# Meta tag prefixes worth keeping verbatim in `raw` when a page has no
# JSON-LD -- everything the Open Graph/product meta conventions carry, not
# just the handful of properties the extractors above happen to read.
_RAW_META_PREFIXES = ("og:", "product:")


def _meta_tags_raw(soup) -> dict:
    """Every og:*/product:* meta tag on the page, verbatim, keyed by its
    property (or name) attribute. This is the reserve stored in `raw` when
    there is no JSON-LD Product block to keep instead -- exact fields the
    page already carries (price, currency, brand, availability, ...) that
    today's extraction does not surface on the card, kept so that adding
    one later is a read of stored data rather than another crawl.
    """
    tags = {}
    for tag in soup.find_all("meta"):
        key = tag.get("property") or tag.get("name") or ""
        if not key.lower().startswith(_RAW_META_PREFIXES):
            continue
        content = (tag.get("content") or "").strip()
        if content:
            tags[key] = content
    return tags


def extract_from_meta(page: dict, html: str) -> list:
    """Open Graph fallback for pages with no JSON-LD but og:type=product."""
    soup = BeautifulSoup(html or "", "html.parser")
    meta = partial(_meta_content, soup)

    if meta("og:type").lower() not in ("product", "product.item"):
        return []
    name = meta("og:title") or page.get("title") or ""
    if not name:
        return []

    url = meta("og:url") or page.get("url") or ""
    ctas, options = _ctas_and_options(html, page.get("url") or "", url, name)
    return [{
        "product_key": product_key_for(url),
        "name": name,
        "description": meta("og:description"),
        "image_url": _image_for(soup, page.get("url") or "", page),
        "product_url": url,
        "category": None,
        "ctas": ctas,
        "options": options,
        "raw": {"meta": _meta_tags_raw(soup)},
    }]


# Enough of a page to identify what is on it, without paying for the whole thing.
_MAX_PAGE_CHARS = 6000


def _llm_price(value) -> float:
    """A number the model wrote down, or None if it isn't trustworthy as a
    price. The prompt already tells the model not to guess, but this is the
    backstop: a non-numeric or non-positive value is treated the same as no
    answer, never coerced into one."""
    if value is None:
        return None
    try:
        price = float(value)
    except (TypeError, ValueError):
        return None
    return price if price > 0 else None


def extract_with_llm(page: dict, tenant_id: str, user_id: str = None, html: str = "") -> list:
    """Products read out of page text. The least trustworthy source.

    price and brand are asked for, but only ever used as a fallback for a
    page whose structured markup had neither -- extract_page_products only
    reaches this tier when JSON-LD and Open Graph both came back empty.
    CTAs come from extract_ctas() on the page's own markup, not the model.
    html defaults to "" for callers with no markup -- see extract_from_jsonld.

    Never raises. Extraction failing must not fail an ingestion.
    """
    text = (page.get("text") or "")[:_MAX_PAGE_CHARS]
    if not text.strip():
        return []

    prompt = PRODUCT_EXTRACTION_PROMPT.format(
        title=page.get("title") or "", url=page.get("url") or "", content=text)

    try:
        response = openai_client.chat.completions.create(
            model=settings.LLM_MODEL,
            messages=[{"role": "user", "content": prompt}],
            max_completion_tokens=1200,
            timeout=60.0,
        )
        payload = json.loads(response.choices[0].message.content)
    except Exception as ex:
        logger.warning(f"Product extraction failed for {page.get('url')}: {ex}")
        return []

    try:
        usage = getattr(response, "usage", None)
        if usage:
            insert_llm_usage(tenant_id, "Product Extraction", settings.LLM_MODEL,
                             usage.prompt_tokens, usage.completion_tokens, usage.total_tokens)
    except Exception as ex:
        logger.warning(f"Could not log product extraction usage: {ex}")

    url = page.get("url") or ""
    # Read once, from the page, for every product this page yields. Never
    # from the model: a hallucinated image URL renders as a broken image on
    # a customer's card, so item.get("image_url") is not even consulted.
    image_url = _image_for(html, url, page)
    # Same for CTAs and options: product_url equals page_url here (the
    # model has no per-product page to key on), so every item on this page
    # shares the same answer -- computed once rather than once per item.
    # The model is only consulted per item, and only when the deterministic
    # pass (including the variant-picker fold, see _ctas_and_options) came
    # back empty -- options come in too many shapes (swatch grids, custom
    # widgets) for a deterministic reader to catch them all, but when it
    # does find something it is exact and the model adds nothing.
    page_ctas, page_options = _ctas_and_options(html, url, url)
    # No JSON-LD reaches this path (extract_page_products only falls
    # through to the model when neither earlier source found anything), so
    # the reserve is whatever og:/product: meta tags the page carries.
    raw = {"meta": _meta_tags_raw(BeautifulSoup(html or "", "html.parser"))}

    products = []
    for item in (payload.get("products") or []):
        if not isinstance(item, dict):
            continue
        name = (item.get("name") or "").strip()
        if not name:
            continue

        products.append({
            "product_key": product_key_for_name(name),
            "name": name,
            "description": (item.get("description") or "").strip() or None,
            "image_url": image_url,
            "product_url": url,
            "category": (item.get("category") or None),
            "brand": (item.get("brand") or "").strip() or None,
            "price": _llm_price(item.get("price")),
            "ctas": page_ctas,
            "options": page_options or _sanitize_llm_options(item.get("options")),
            "raw": raw,
        })
    return products


def extract_page_products(page: dict, html: str, tenant_id: str,
                          user_id: str = None) -> list:
    """Products from one page, using the most trustworthy source available.

    Stops at the first source that yields anything: exact data on the page beats
    a model reading prose, and calling the model anyway would spend the tenant's
    token quota for a worse answer.

    Never raises. A page that cannot be parsed simply contributes no products.
    """
    try:
        products = extract_from_jsonld(page, html)
        if products:
            return products

        products = extract_from_meta(page, html)
        if products:
            return products

        return extract_with_llm(page, tenant_id, user_id=user_id, html=html)
    except Exception as ex:
        logger.warning(f"Product extraction failed for {page.get('url')}: {ex}")
        return []


# category is the one field an exact source (JSON-LD/meta) can find a product
# but still leave empty -- it isn't required by Google's rich-snippet spec,
# even on sites that otherwise state it in a breadcrumb or section heading.
# Batched like enrichment/extractor.py's attribute extraction: one call per
# CATEGORY_BACKFILL_BATCH_SIZE products instead of one call per product, since
# a whole crawl can be dozens of single-product pages and category is a
# low-stakes field not worth a network round-trip each.
CATEGORY_BACKFILL_BATCH_SIZE = 20
# Breadcrumbs/section headings that state a category sit near the top of a
# product page; a full _MAX_PAGE_CHARS-sized excerpt per item would blow the
# token budget once CATEGORY_BACKFILL_BATCH_SIZE items share one prompt.
_CATEGORY_BACKFILL_PAGE_CHARS = 800


def backfill_categories(pairs: list, tenant_id: str, user_id: str = None) -> None:
    """Mutates products in place, filling in category for any (product, page)
    pair whose product has none. Products that already have a category (an
    exact source stated one) are never touched or sent to the model."""
    needing = [(p, pg) for p, pg in pairs if not p.get("category")]
    for i in range(0, len(needing), CATEGORY_BACKFILL_BATCH_SIZE):
        _backfill_one_batch(needing[i:i + CATEGORY_BACKFILL_BATCH_SIZE], tenant_id,
                            user_id=user_id)


def _backfill_one_batch(batch: list, tenant_id: str, user_id: str = None) -> None:
    items = [
        {"index": i, "name": product.get("name") or "",
         "content": (page.get("text") or "")[:_CATEGORY_BACKFILL_PAGE_CHARS]}
        for i, (product, page) in enumerate(batch)
    ]
    prompt = CATEGORY_BACKFILL_PROMPT.format(items=json.dumps(items))

    try:
        response = openai_client.chat.completions.create(
            model=settings.LLM_MODEL,
            messages=[{"role": "user", "content": prompt}],
            max_completion_tokens=100 * len(batch),
            timeout=60.0,
        )
        payload = json.loads(response.choices[0].message.content)
    except Exception as ex:
        logger.warning(f"Category backfill batch of {len(batch)} failed: {ex}")
        return

    try:
        usage = getattr(response, "usage", None)
        if usage:
            insert_llm_usage(tenant_id, "Category Backfill", settings.LLM_MODEL,
                             usage.prompt_tokens, usage.completion_tokens, usage.total_tokens)
    except Exception as ex:
        logger.warning(f"Could not log category backfill usage: {ex}")

    for entry in (payload.get("categories") or []):
        if not isinstance(entry, dict):
            continue
        index = entry.get("index")
        if not isinstance(index, int) or not (0 <= index < len(batch)):
            continue
        category = entry.get("category")
        if isinstance(category, str) and category.strip():
            batch[index][0]["category"] = category.strip()
