import json
from unittest.mock import MagicMock, patch

from app.services.content import product_extraction as product_extraction

PAGE = {
    "url": "https://galaxiq.ai/products/brandforge",
    "title": "BrandForge — GalaxiQ",
    "text": "BrandForge is our brand strategy platform. Book a demo today. From $129/month.",
    "jsonld": [],
}


def _llm_returning(payload):
    response = MagicMock()
    response.choices = [MagicMock()]
    response.choices[0].message.content = json.dumps(payload)
    return response


def test_a_genuine_price_the_model_read_off_the_page_is_kept():
    """price is a narrow, validated fallback for pages with no structured
    markup at all -- extract_page_products only reaches this tier then, so
    a page WITH JSON-LD/meta data never has its price second-guessed here."""
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform",
                             "price": "129.00"}]}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x")

    assert product["price"] == 129.00


def test_a_non_numeric_price_is_rejected_not_coerced():
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform",
                             "price": "starting from $129"}]}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x")

    assert product["price"] is None


def test_a_zero_or_negative_price_is_rejected():
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform",
                             "price": 0}]}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x")

    assert product["price"] is None


def test_no_price_stated_yields_none_not_a_guess():
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform"}]}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x")

    assert product["price"] is None
    assert "currency" not in product
    assert "source" not in product


def test_a_stated_brand_is_kept():
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform",
                             "brand": "GalaxiQ"}]}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x")

    assert product["brand"] == "GalaxiQ"


def test_no_brand_stated_yields_none():
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform"}]}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x")

    assert product["brand"] is None


def test_ctas_come_from_the_page_html_not_the_model():
    """The model no longer supplies CTAs at all; even if it did, they'd be
    ignored. CTAs come from extract_ctas() on the page's own markup."""
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform"}]}
    html = '<a href="/demo">Book a demo</a>'
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x", html=html)

    assert product["ctas"] == [{
        "type": "book_demo", "label": "Book a demo",
        "url": "https://galaxiq.ai/demo",
    }]


def test_no_ctas_on_the_page_falls_back_to_a_view_cta():
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform"}]}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x", html="")

    assert product["ctas"] == [{
        "type": "view", "label": "View details",
        "url": "https://galaxiq.ai/products/brandforge",
    }]


def test_model_supplied_image_url_is_never_used():
    """A hallucinated image URL would render as a broken image on a
    customer's card, so the model's payload must never even be consulted --
    the image comes from the page's own og:image/twitter:image, or is None."""
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform",
                             "image_url": "https://evil.example.com/fake.jpg"}]}
    html = '<meta property="og:image" content="https://galaxiq.ai/og/brandforge.jpg">'
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product,) = product_extraction.extract_with_llm(PAGE, "org_x", html=html)

    assert product["image_url"] == "https://galaxiq.ai/og/brandforge.jpg"

    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (product_no_html,) = product_extraction.extract_with_llm(PAGE, "org_x", html="")

    assert product_no_html["image_url"] is None


def test_a_page_with_no_products_returns_nothing():
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning({"products": []})
        assert product_extraction.extract_with_llm(PAGE, "org_x") == []


def test_a_model_failure_returns_nothing_rather_than_raising():
    """Extraction must never fail an ingestion."""
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.side_effect = RuntimeError("azure down")
        assert product_extraction.extract_with_llm(PAGE, "org_x") == []


def test_malformed_model_output_returns_nothing():
    response = MagicMock()
    response.choices = [MagicMock()]
    response.choices[0].message.content = "not json at all"
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = response
        assert product_extraction.extract_with_llm(PAGE, "org_x") == []


def test_two_pages_describing_the_same_product_yield_one_key():
    """The model has no SKU and no dedicated product URL to key on, so two
    pages that both mention the same product (a homepage teaser and the
    product's own page, say) must collapse onto one row instead of minting
    a new key per page."""
    homepage = {**PAGE, "url": "https://galaxiq.ai/", "title": "GalaxiQ"}
    product_page = {**PAGE, "url": "https://galaxiq.ai/products/brandforge"}
    payload = {"products": [{"name": "BrandForge", "description": "Brand platform"}]}

    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        (from_homepage,) = product_extraction.extract_with_llm(homepage, "org_x")
        (from_product_page,) = product_extraction.extract_with_llm(product_page, "org_x")

    assert from_homepage["product_key"] == from_product_page["product_key"]


def test_two_different_products_get_different_keys():
    payload = {"products": [
        {"name": "BrandForge", "description": "Brand platform"},
        {"name": "InsightIQ", "description": "Market intelligence"},
    ]}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _llm_returning(payload)
        brandforge, insightiq = product_extraction.extract_with_llm(PAGE, "org_x")

    assert brandforge["product_key"] != insightiq["product_key"]


# --- category backfill: extract_page_products stays pure -----------------------

JSONLD_PAGE_NO_CATEGORY = {
    "url": "https://shop.example.com/products/blue-runner",
    "title": "Blue Runner",
    "text": "Home > Shoes > Running. Blue Runner. A lightweight road shoe. $129.00",
    "jsonld": [{
        "@type": "Product",
        "name": "Blue Runner",
        "description": "A lightweight road shoe.",
        "sku": "BR-42",
        "offers": {"@type": "Offer", "price": "129.00", "priceCurrency": "AUD"},
    }],
}


def test_extract_page_products_never_calls_the_model_for_an_exact_source():
    """Category backfill is a separate, batched step (backfill_categories) --
    extract_page_products itself must stay a pure reader of JSON-LD/meta with
    no model call, even when the exact source leaves category empty."""
    with patch.object(product_extraction, "openai_client") as client:
        (product,) = product_extraction.extract_page_products(
            JSONLD_PAGE_NO_CATEGORY, "", "org_x")
        client.chat.completions.create.assert_not_called()

    assert product["category"] is None


# --- category backfill: backfill_categories (batched) ---------------------------

def _category_batch_response(entries):
    return _llm_returning({"categories": entries})


def test_a_category_the_page_states_is_backfilled():
    product = {"name": "Blue Runner", "description": "A lightweight road shoe.",
              "category": None}
    page = JSONLD_PAGE_NO_CATEGORY

    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _category_batch_response(
            [{"index": 0, "category": "Running"}])
        product_extraction.backfill_categories([(product, page)], "org_x")

    assert product["category"] == "Running"
    # Every other field stays exactly what it already was -- the backfill
    # must never touch anything but the missing field.
    assert product["description"] == "A lightweight road shoe."


def test_a_product_that_already_has_a_category_is_never_sent_to_the_model():
    product = {"name": "Blue Runner", "category": "Running Shoes"}
    with patch.object(product_extraction, "openai_client") as client:
        product_extraction.backfill_categories([(product, JSONLD_PAGE_NO_CATEGORY)], "org_x")
        client.chat.completions.create.assert_not_called()

    assert product["category"] == "Running Shoes"


def test_no_category_stated_on_the_page_yields_none_not_a_guess():
    product = {"name": "Blue Runner", "category": None}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _category_batch_response(
            [{"index": 0, "category": None}])
        product_extraction.backfill_categories([(product, JSONLD_PAGE_NO_CATEGORY)], "org_x")

    assert product["category"] is None


def test_a_category_backfill_failure_does_not_lose_the_product():
    """The backfill is a narrow addition on top of an exact source's result --
    it must never cost the product itself if the model call fails."""
    product = {"name": "Blue Runner", "category": None}
    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.side_effect = RuntimeError("azure down")
        product_extraction.backfill_categories([(product, JSONLD_PAGE_NO_CATEGORY)], "org_x")

    assert product["name"] == "Blue Runner"
    assert product["category"] is None


def test_many_products_needing_category_are_batched_into_few_calls():
    """The whole point of batching: CATEGORY_BACKFILL_BATCH_SIZE products
    needing a category share one model call, not one call each."""
    batch_size = product_extraction.CATEGORY_BACKFILL_BATCH_SIZE
    products = [{"name": f"Product {i}", "category": None} for i in range(batch_size + 1)]
    pairs = [(p, JSONLD_PAGE_NO_CATEGORY) for p in products]

    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.side_effect = [
            _category_batch_response([{"index": i, "category": f"Cat {i}"}
                                      for i in range(batch_size)]),
            _category_batch_response([{"index": 0, "category": "Cat last"}]),
        ]
        product_extraction.backfill_categories(pairs, "org_x")

    assert client.chat.completions.create.call_count == 2
    assert products[0]["category"] == "Cat 0"
    assert products[-1]["category"] == "Cat last"


def test_a_product_with_a_category_does_not_count_toward_batch_size():
    """Only products actually needing a category fill a batch slot -- one
    that already has a category is filtered out before batching, not just
    skipped when applying the response."""
    already_categorised = {"name": "Has One", "category": "Existing"}
    needs_one = {"name": "Blue Runner", "category": None}

    with patch.object(product_extraction, "openai_client") as client:
        client.chat.completions.create.return_value = _category_batch_response(
            [{"index": 0, "category": "Running"}])
        product_extraction.backfill_categories(
            [(already_categorised, JSONLD_PAGE_NO_CATEGORY),
             (needs_one, JSONLD_PAGE_NO_CATEGORY)], "org_x")

    client.chat.completions.create.assert_called_once()
    assert already_categorised["category"] == "Existing"
    assert needs_one["category"] == "Running"


def test_no_products_needing_category_makes_no_call_at_all():
    product = {"name": "Blue Runner", "category": "Running"}
    with patch.object(product_extraction, "openai_client") as client:
        product_extraction.backfill_categories([(product, JSONLD_PAGE_NO_CATEGORY)], "org_x")
        client.chat.completions.create.assert_not_called()


def test_product_key_for_still_prefers_a_real_sku_over_a_name():
    """The name-based key is only for the model's output, which never has a
    SKU. JSON-LD's SKU path is untouched."""
    assert product_extraction.product_key_for("https://s.com/p/1", "BR-42") == "sku:BR-42"
