from app.services.content import product_extraction as product_extraction
from app.services.content.product_extraction import (
    CTA_TYPES, extract_ctas, extract_from_jsonld, extract_from_meta, extract_options,
    extract_with_llm, filter_ctas_for_product, product_key_for,
)

PAGE = {
    "url": "https://shop.example.com/products/blue-runner",
    "title": "Blue Runner — Example Shop",
    "text": "Blue Runner. A lightweight road shoe.",
    "jsonld": [{
        "@type": "Product",
        "name": "Blue Runner",
        "description": "A lightweight road shoe.",
        "sku": "BR-42",
        "image": "https://shop.example.com/img/br.jpg",
        "offers": {
            "@type": "Offer",
            "price": "129.00",
            "priceCurrency": "AUD",
            "availability": "https://schema.org/InStock",
            "url": "https://shop.example.com/products/blue-runner",
        },
    }],
}


def test_reads_a_product_from_jsonld():
    (product,) = extract_from_jsonld(PAGE)
    assert product["name"] == "Blue Runner"
    assert product["product_url"] == "https://shop.example.com/products/blue-runner"


def test_ignores_non_product_blocks():
    page = dict(PAGE, jsonld=[{"@type": "WebSite", "name": "Shop"},
                              {"@type": "BreadcrumbList"}])
    assert extract_from_jsonld(page) == []


def test_offers_url_is_still_used_for_the_product_url():
    """Offers still carry the canonical product URL even though price is gone."""
    page = dict(PAGE, jsonld=[{"@type": "Product", "name": "X",
                               "offers": [{"url": "https://shop.example.com/p/x"},
                                          {"url": "https://shop.example.com/p/x-alt"}]}])
    assert extract_from_jsonld(page)[0]["product_url"] == "https://shop.example.com/p/x"


def test_image_may_be_a_list():
    page = dict(PAGE, jsonld=[{"@type": "Product", "name": "X",
                               "image": ["https://a/1.jpg", "https://a/2.jpg"]}])
    assert extract_from_jsonld(page)[0]["image_url"] == "https://a/1.jpg"


def test_a_product_without_a_name_is_dropped():
    page = dict(PAGE, jsonld=[{"@type": "Product", "sku": "X-1"}])
    assert extract_from_jsonld(page) == []


def test_every_product_gets_a_view_cta_at_minimum():
    (product,) = extract_from_jsonld(PAGE)
    types = [c["type"] for c in product["ctas"]]
    assert "view" in types
    assert set(types) <= CTA_TYPES


def test_product_key_is_stable_and_prefers_sku():
    a = product_key_for("https://shop.example.com/p/1", "BR-42")
    b = product_key_for("https://shop.example.com/p/1?utm_source=x", "BR-42")
    assert a == b == "sku:BR-42"


def test_product_key_falls_back_to_the_url():
    key = product_key_for("https://shop.example.com/p/1")
    assert key.startswith("url:")
    assert key == product_key_for("https://shop.example.com/p/1")


META_PAGE = {
    "url": "https://shop.example.com/products/blue-runner",
    "title": "Blue Runner — Example Shop",
}

FULL_META_HTML = """
<html><head>
<meta property="og:type" content="product">
<meta property="og:title" content="Blue Runner">
<meta property="og:description" content="A lightweight road shoe.">
<meta property="og:image" content="https://shop.example.com/img/br.jpg">
<meta property="og:url" content="https://shop.example.com/products/blue-runner">
</head><body></body></html>
"""


def test_reads_a_product_from_og_meta_tags():
    (product,) = extract_from_meta(META_PAGE, FULL_META_HTML)
    assert product["name"] == "Blue Runner"
    assert product["description"] == "A lightweight road shoe."
    assert product["image_url"] == "https://shop.example.com/img/br.jpg"
    assert product["product_url"] == "https://shop.example.com/products/blue-runner"


def test_non_product_og_type_returns_nothing():
    html = '<html><head><meta property="og:type" content="website"></head></html>'
    assert extract_from_meta(META_PAGE, html) == []


def test_product_without_any_title_returns_nothing():
    html = '<html><head><meta property="og:type" content="product"></head></html>'
    page = dict(META_PAGE, title="")
    assert extract_from_meta(page, html) == []


def test_meta_product_has_the_same_keys_as_jsonld_product():
    (jsonld_product,) = extract_from_jsonld(PAGE)
    (meta_product,) = extract_from_meta(META_PAGE, FULL_META_HTML)
    assert set(meta_product.keys()) == set(jsonld_product.keys())
    assert len(meta_product) == 9
    assert set(meta_product.keys()) == {
        "product_key", "name", "description", "image_url",
        "product_url", "category", "ctas", "options", "raw",
    }


# --- extract_ctas -----------------------------------------------------

PAGE_URL = "https://shop.example.com/products/blue-runner"


def test_add_to_cart_href_is_classified_correctly():
    html = '<a href="/cart/add?sku=BR-42">Add to Cart</a>'
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["type"] == "add_to_cart"
    assert cta["url"] == "https://shop.example.com/cart/add?sku=BR-42"


def test_unrecognised_button_becomes_visit_rather_than_dropped():
    html = '<button data-href="/wishlist/add">Save for later</button>'
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["type"] == "visit"
    assert cta["url"] == "https://shop.example.com/wishlist/add"


def test_relative_hrefs_are_resolved_to_absolute():
    html = '<a href="/products/blue-runner/reviews">Reviews</a>'
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["url"] == "https://shop.example.com/products/blue-runner/reviews"


def test_javascript_and_hash_links_are_skipped():
    html = """
    <a href="#">Top</a>
    <a href="#section-2">Jump</a>
    <a href="javascript:void(0)">Click</a>
    <a href="mailto:hi@example.com">Email us</a>
    <a href="tel:+123456789">Call us</a>
    <a href="/real-link">Real Link</a>
    """
    ctas = extract_ctas(html, PAGE_URL)
    assert [c["label"] for c in ctas] == ["Real Link"]


def test_duplicate_urls_are_collapsed_keeping_the_first():
    html = """
    <a href="/buy">Buy Now</a>
    <a href="/buy">Purchase</a>
    """
    ctas = extract_ctas(html, PAGE_URL)
    assert len(ctas) == 1
    assert ctas[0]["label"] == "Buy Now"


def test_malformed_html_returns_empty_list():
    assert extract_ctas(None, PAGE_URL) == []
    assert extract_ctas(12345, PAGE_URL) == []


def test_label_is_taken_verbatim_including_unusual_wording():
    html = '<a href="/bag/add">Add to bag</a>'
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["label"] == "Add to bag"
    assert cta["type"] == "add_to_cart"


def test_at_most_six_ctas_are_returned_in_document_order():
    html = "".join(f'<a href="/link-{i}">Link {i}</a>' for i in range(10))
    ctas = extract_ctas(html, PAGE_URL)
    assert len(ctas) == 6
    assert [c["label"] for c in ctas] == [f"Link {i}" for i in range(6)]


def test_elements_with_no_text_are_skipped():
    html = '<a href="/x"><img src="icon.png"></a><a href="/y">Real</a>'
    ctas = extract_ctas(html, PAGE_URL)
    assert [c["label"] for c in ctas] == ["Real"]


def test_chrome_links_are_excluded_in_favour_of_the_product_area():
    html = """
    <nav>
        <a href="/">Home</a>
        <a href="/about">About Us</a>
        <a href="/products">Product</a>
        <a href="/services">Services</a>
        <a href="/pricing">Pricing</a>
    </nav>
    <main>
        <a href="/cart/add">Add to bag</a>
    </main>
    """
    ctas = extract_ctas(html, PAGE_URL)
    assert [c["label"] for c in ctas] == ["Add to bag"]
    assert ctas[0]["type"] == "add_to_cart"


def test_add_to_cart_survives_the_cap_even_when_it_appears_last():
    links = [f'<a href="/link-{i}">Link {i}</a>' for i in range(6)]
    links.append('<a href="/cart/add">Add to cart</a>')
    html = "".join(links)
    ctas = extract_ctas(html, PAGE_URL)
    assert len(ctas) == 6
    assert ctas[0]["type"] == "add_to_cart"
    assert ctas[0]["label"] == "Add to cart"
    # The five "visit" survivors keep document order behind it.
    assert [c["label"] for c in ctas[1:]] == [f"Link {i}" for i in range(5)]


def test_bare_contact_is_visit_but_a_real_enquiry_phrase_is_enquire():
    html = """
    <a href="/contact">Contact</a>
    <a href="/contact-sales">Contact us about this</a>
    """
    ctas = extract_ctas(html, PAGE_URL)
    by_label = {c["label"]: c["type"] for c in ctas}
    assert by_label["Contact"] == "visit"
    assert by_label["Contact us about this"] == "enquire"


def test_ordering_within_a_band_stays_document_order():
    html = """
    <a href="/donate">Donate</a>
    <a href="/subscribe">Subscribe</a>
    <a href="/random-1">Random One</a>
    <a href="/random-2">Random Two</a>
    """
    ctas = extract_ctas(html, PAGE_URL)
    assert [c["label"] for c in ctas] == ["Donate", "Subscribe", "Random One", "Random Two"]


# --- label extraction preference order ---------------------------------

def test_aria_label_wins_over_a_long_card_blurb():
    html = ('<a href="/cart/add" aria-label="Add to bag">'
            'BrandForge Creative Maintains message, tone, and visuals for a '
            'unified brand presence across every channel.</a>')
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["label"] == "Add to bag"


def test_title_wins_when_there_is_no_aria_label():
    html = '<a href="/p/1" title="View BrandForge">See more about this product</a>'
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["label"] == "View BrandForge"


def test_card_shaped_anchor_uses_the_nested_heading_not_the_run_on_text():
    html = ('<a href="/p/1"><h3>BrandForge</h3>'
            '<p>Maintains message, tone and visuals for a unified brand '
            'presence across every channel.</p></a>')
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["label"] == "BrandForge"


def test_plain_button_with_no_aria_label_title_or_heading_is_unaffected():
    html = '<a href="/cart/add">Add to bag</a>'
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["label"] == "Add to bag"


def test_label_longer_than_sixty_characters_is_capped():
    long_label = "A" * 100
    html = f'<a href="/p/1" aria-label="{long_label}">click</a>'
    (cta,) = extract_ctas(html, PAGE_URL)
    assert cta["label"] == "A" * 60
    assert len(cta["label"]) == 60


# --- filter_ctas_for_product ---------------------------------------------

AIM_WATCH_URL = "https://www.scrapingcourse.com/ecommerce/product/aim-analog-watch/"


def _cta(cta_type, label, url):
    return {"type": cta_type, "label": label, "url": url}


def test_rule1_same_path_as_products_own_url_is_kept():
    cta = _cta("add_to_cart", "Add to cart", AIM_WATCH_URL + "?add-to-cart=2740")
    kept = filter_ctas_for_product([cta], AIM_WATCH_URL, "Aim Analog Watch")
    assert kept == [cta]


def test_rule_different_path_same_host_is_dropped():
    sibling = _cta("visit", "Bolo Sport Watch",
                   "https://www.scrapingcourse.com/ecommerce/product/bolo-sport-watch/")
    category = _cta("visit", "Watches",
                    "https://www.scrapingcourse.com/ecommerce/product-category/gear/watches/")
    kept = filter_ctas_for_product([sibling, category], AIM_WATCH_URL, "Aim Analog Watch")
    assert kept == []


def test_rule_action_like_path_elsewhere_on_the_site_is_kept():
    cta = _cta("enquire", "Get Started Free", "https://galaxiq.ai/contact")
    kept = filter_ctas_for_product(
        [cta], "https://galaxiq.ai/products/brandforge", "BrandForge")
    assert kept == [cta]


def test_rule_a_different_host_entirely_is_kept():
    cta = _cta("book_demo", "Book on Calendly", "https://calendly.com/galaxiq/demo")
    kept = filter_ctas_for_product(
        [cta], "https://galaxiq.ai/products/brandforge", "BrandForge")
    assert kept == [cta]


def test_filtering_happens_before_the_cap_so_the_real_button_is_not_crowded_out():
    """Six same-host sibling-product links, all appearing before the real
    add-to-cart button in the document. filter_ctas_for_product must drop
    navigation before the six-CTA cap is applied, or the real button never
    gets a chance."""
    siblings = "".join(
        f'<a href="https://www.scrapingcourse.com/ecommerce/product/sibling-{i}/">Sibling {i}</a>'
        for i in range(1, 7)
    )
    html = siblings + f'<a href="{AIM_WATCH_URL}?add-to-cart=2740">Add to cart</a>'
    ctas = product_extraction._ctas_for(html, AIM_WATCH_URL, AIM_WATCH_URL, "Aim Analog Watch")
    assert [c["url"] for c in ctas] == [AIM_WATCH_URL + "?add-to-cart=2740"]


def test_get_started_to_contact_survives_alongside_the_rules():
    """The concrete regression this whole rule set must not reintroduce."""
    html = '<a href="/contact">Get Started Free</a>'
    ctas = product_extraction._ctas_for(
        html, "https://galaxiq.ai/products/brandforge",
        "https://galaxiq.ai/products/brandforge", "BrandForge")
    assert ctas == [_cta("enquire", "Get Started Free", "https://galaxiq.ai/contact")]


# --- related-products containers are excluded structurally -----------------
# The wrong-product add-to-cart bug (a related product's own button, posting
# back to *this* page) is prevented by removing the whole rail before a CTA
# is ever collected -- not by recognising a button's wording, which fails
# the moment a site says "Add to basket" instead of "Add to cart", or says
# either of those in a language other than English.

def test_related_products_container_is_excluded_structurally():
    html = (
        '<div class="related products">'
        f'<a href="{AIM_WATCH_URL}?add-to-cart=2746">Add to cart</a>'
        f'<a href="{AIM_WATCH_URL}?add-to-cart=2745">Add to cart</a>'
        '</div>'
        f'<a href="{AIM_WATCH_URL}?add-to-cart=2740">Add to cart</a>'
    )
    ctas = extract_ctas(html, AIM_WATCH_URL)
    assert [c["url"] for c in ctas] == [AIM_WATCH_URL + "?add-to-cart=2740"]


def test_upsell_container_by_id_is_also_excluded():
    html = ('<div id="upsell-products"><a href="/p/other">Buy Now</a></div>'
            '<a href="/p/self">Add to cart</a>')
    ctas = extract_ctas(html, "https://shop.example.com/p/self")
    assert [c["url"] for c in ctas] == ["https://shop.example.com/p/self"]


# --- label cleaning --------------------------------------------------------

def test_bare_add_to_cart_label_is_unaffected_by_cleaning():
    (cta,) = extract_ctas('<a href="/cart/add">Add to cart</a>', PAGE_URL)
    assert cta["label"] == "Add to cart"


def test_pipes_collapse_to_a_single_space():
    (cta,) = extract_ctas('<a href="/x">Erin Recommends|Clothing</a>', PAGE_URL)
    assert cta["label"] == "Erin Recommends Clothing"


# --- product container scoping ---------------------------------------------
# books.toscrape.com-style pages carry category links ("Books", "Poetry")
# elsewhere on the page; scoping to the product's own detail area, when the
# page marks one, keeps them out without needing to recognise a category
# link by name.

def test_ctas_outside_the_product_container_are_ignored_when_one_exists():
    html = (
        '<ul class="breadcrumb"><a href="/category/books">Books</a>'
        '<a href="/category/poetry">Poetry</a></ul>'
        '<div class="product_main">'
        '<a href="/add-to-basket/123">Add to basket</a>'
        '</div>'
    )
    ctas = extract_ctas(html, "https://books.toscrape.com/catalogue/a-light/index.html")
    assert [c["label"] for c in ctas] == ["Add to basket"]


def test_falls_back_to_the_whole_page_when_no_container_is_marked():
    html = '<a href="/contact">Get Started</a>'
    ctas = extract_ctas(html, "https://galaxiq.ai/products/brandforge")
    assert [c["label"] for c in ctas] == ["Get Started"]


# --- form-submit add-to-cart buttons ---------------------------------------
# The standard e-commerce pattern for a variable product: the button itself
# carries no href/formaction/data-href at all, only a plain form submit.

def test_form_submit_add_to_cart_uses_the_forms_action():
    html = ('<form class="cart" action="/product/adrienne-trek-jacket/">'
            '<input type="hidden" name="add-to-cart" value="1872">'
            '<button type="submit" class="single_add_to_cart_button">Add to cart</button>'
            '</form>')
    (cta,) = extract_ctas(html, "https://shop.example.com/product/adrienne-trek-jacket/")
    assert cta["type"] == "add_to_cart"
    assert cta["url"] == ("https://shop.example.com/product/adrienne-trek-jacket/"
                          "?add-to-cart=1872")


def test_generic_submit_button_with_no_product_identity_is_dropped():
    """A search box or newsletter form is also, technically, a form
    submit -- but it carries no hidden product field and its own label is
    a generic verb, so it must not become a CTA."""
    html = ('<form action="/search"><input type="text" name="q">'
            '<button type="submit">Submit</button></form>')
    ctas = extract_ctas(html, "https://galaxiq.ai/products/brandforge")
    assert ctas == []


def test_send_enquiry_is_not_caught_by_the_generic_send_verb():
    """The label match is exact on the trimmed text, not a substring, so a
    legitimate "Send enquiry" button is safe from the generic verb "send"."""
    html = '<form action="/contact"><button type="submit">Send enquiry</button></form>'
    (cta,) = extract_ctas(html, "https://galaxiq.ai/contact")
    assert cta["label"] == "Send enquiry"


def test_generic_verb_button_is_kept_when_the_form_carries_product_identity():
    html = ('<form action="/cart/add"><input type="hidden" name="product_id" value="42">'
            '<button type="submit">Submit</button></form>')
    (cta,) = extract_ctas(html, "https://shop.example.com/p/x")
    assert cta["url"] == "https://shop.example.com/cart/add?product_id=42"


# --- image extraction -----------------------------------------------------

NO_IMAGE_JSONLD_PAGE = {
    "url": "https://shop.example.com/products/blue-runner",
    "title": "Blue Runner",
    "jsonld": [{"@type": "Product", "name": "Blue Runner"}],
}


def test_og_image_is_picked_when_present():
    html = '<meta property="og:image" content="https://shop.example.com/img/br.jpg">'
    (product,) = extract_from_jsonld(NO_IMAGE_JSONLD_PAGE, html)
    assert product["image_url"] == "https://shop.example.com/img/br.jpg"


def test_twitter_image_is_used_when_og_image_is_absent():
    html = '<meta name="twitter:image" content="https://shop.example.com/img/tw.jpg">'
    (product,) = extract_from_jsonld(NO_IMAGE_JSONLD_PAGE, html)
    assert product["image_url"] == "https://shop.example.com/img/tw.jpg"


def test_relative_og_image_resolves_to_absolute():
    html = '<meta property="og:image" content="/img/br.jpg">'
    (product,) = extract_from_jsonld(NO_IMAGE_JSONLD_PAGE, html)
    assert product["image_url"] == "https://shop.example.com/img/br.jpg"


def test_no_image_anywhere_yields_none_not_empty_string():
    (product,) = extract_from_jsonld(NO_IMAGE_JSONLD_PAGE, "")
    assert product["image_url"] is None


# --- extract_options ---------------------------------------------------

def test_select_yields_an_option_with_placeholders_stripped():
    html = """
    <select name="attribute_size">
        <option value="">Choose an option</option>
        <option value="xs">XS</option>
        <option value="s">S</option>
        <option value="m">M</option>
    </select>
    """
    (option,) = extract_options(html)
    assert option == {"name": "Size", "values": ["XS", "S", "M"]}


def test_radio_group_yields_an_option():
    html = """
    <input type="radio" name="attribute_color" value="red" id="c-red">
    <label for="c-red">Red</label>
    <input type="radio" name="attribute_color" value="blue" id="c-blue">
    <label for="c-blue">Blue</label>
    """
    (option,) = extract_options(html)
    assert option == {"name": "Color", "values": ["Red", "Blue"]}


def test_no_option_markup_yields_empty_list():
    assert extract_options("<p>Just some text, no form controls at all.</p>") == []
    assert extract_options("") == []


def test_form_scoped_reader_wins_over_the_name_based_fallback():
    """A <select> inside the product's own add-to-cart form is read by
    name whatever it is called -- the Shopify convention, unlike
    WooCommerce's attribute_*/pa_*, would otherwise be invisible to the
    name-based fallback."""
    html = """
    <form action="/cart/add"><input type="hidden" name="id" value="99">
        <label for="opt-size">Size</label>
        <select name="options[Size]" id="opt-size">
            <option value="">Select</option>
            <option value="s">S</option>
            <option value="l">L</option>
        </select>
    </form>
    """
    (option,) = extract_options(html)
    assert option == {"name": "Size", "values": ["S", "L"]}


def test_placeholder_only_select_outside_any_form_yields_no_option():
    html = ('<select name="attribute_size"><option value="">Choose an '
            'option</option></select>')
    assert extract_options(html) == []


# --- model options fallback ---------------------------------------------

def test_model_options_used_only_when_the_deterministic_pass_is_empty():
    from unittest.mock import MagicMock, patch
    import json as jsonlib

    page = {"url": "https://shop.example.com/p/jacket", "title": "Jacket",
           "text": "A trek jacket.", "jsonld": []}
    payload = {"products": [{"name": "Jacket", "description": "A trek jacket.",
                             "options": [{"name": "Size", "values": ["S", "M"]}]}]}
    response = MagicMock()
    response.choices = [MagicMock()]
    response.choices[0].message.content = jsonlib.dumps(payload)

    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = response
        (no_markup,) = extract_with_llm(page, "org_x", html="")
        assert no_markup["options"] == [{"name": "Size", "values": ["S", "M"]}]

    deterministic_html = ('<select name="attribute_size"><option value="">Choose'
                          '</option><option value="xl">XL</option>'
                          '<option value="xxl">XXL</option></select>')
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = response
        (with_markup,) = extract_with_llm(page, "org_x", html=deterministic_html)
        assert with_markup["options"] == [{"name": "Size", "values": ["XL", "XXL"]}]


# --- raw ------------------------------------------------------------------

def test_raw_carries_the_jsonld_block_verbatim():
    (product,) = extract_from_jsonld(PAGE)
    assert product["raw"]["jsonld"] == PAGE["jsonld"][0]
    assert product["raw"]["sku"] == "BR-42"


def test_raw_carries_meta_tags_when_there_is_no_jsonld():
    html = """
    <meta property="og:title" content="Blue Runner">
    <meta property="product:price:amount" content="129.00">
    <meta property="product:price:currency" content="AUD">
    """
    (product,) = extract_from_meta(META_PAGE, html + FULL_META_HTML)
    assert product["raw"]["meta"]["product:price:amount"] == "129.00"
    assert product["raw"]["meta"]["product:price:currency"] == "AUD"
    assert "jsonld" not in product["raw"]


# --- query-parameter variant CTA grouping ----------------------------------

HYDROGEN_URL = "https://hydrogen.shop/products/the-full-stack"


def test_two_links_differing_in_one_query_param_become_one_option():
    html = (f'<a href="{HYDROGEN_URL}?variant=154cm">154cm</a>'
           f'<a href="{HYDROGEN_URL}?variant=158cm">158cm</a>')
    ctas, options = product_extraction._ctas_and_options(html, HYDROGEN_URL, HYDROGEN_URL)
    assert options == [{"name": "Options", "values": ["154cm", "158cm"]}]
    assert ctas == [{"type": "view", "label": "View details", "url": HYDROGEN_URL}]


def test_a_meaningful_query_key_names_the_option():
    html = (f'<a href="{HYDROGEN_URL}?size=154cm">154cm</a>'
           f'<a href="{HYDROGEN_URL}?size=158cm">158cm</a>')
    ctas, options = product_extraction._ctas_and_options(html, HYDROGEN_URL, HYDROGEN_URL)
    assert options == [{"name": "Size", "values": ["154cm", "158cm"]}]


def test_a_single_link_is_not_a_picker_and_stays_a_cta():
    html = f'<a href="{HYDROGEN_URL}?variant=154cm">154cm</a>'
    ctas, options = product_extraction._ctas_and_options(html, HYDROGEN_URL, HYDROGEN_URL)
    assert options == []
    assert [c["url"] for c in ctas] == [HYDROGEN_URL + "?variant=154cm"]


def test_links_differing_in_path_are_not_grouped():
    html = ('<a href="https://hydrogen.shop/products/a?variant=1">A</a>'
           '<a href="https://hydrogen.shop/products/b?variant=2">B</a>')
    ctas, options = product_extraction._ctas_and_options(
        html, "https://hydrogen.shop/products/a", "https://hydrogen.shop/products/a")
    assert options == []
    # Only the same-path link survives filter_ctas_for_product; the other
    # is a different product's page and is dropped as navigation.
    assert [c["url"] for c in ctas] == ["https://hydrogen.shop/products/a?variant=1"]


def test_variant_grouping_does_not_run_when_deterministic_options_exist():
    """A page with a real <select> must not have its unrelated CTAs
    reinterpreted as a variant group even if they happen to share a query
    key -- the fold only runs when extract_options found nothing."""
    html = ("""
    <select name="attribute_size">
        <option value="">Choose</option>
        <option value="s">S</option>
        <option value="m">M</option>
    </select>
    """ + f'<a href="{HYDROGEN_URL}?ref=footer">Footer link 1</a>'
         f'<a href="{HYDROGEN_URL}?ref=header">Footer link 2</a>')
    ctas, options = product_extraction._ctas_and_options(html, HYDROGEN_URL, HYDROGEN_URL)
    assert options == [{"name": "Size", "values": ["S", "M"]}]
