"""extract_text_from_url must visit product-shaped URLs before plain nav
links, and must fall back to one LLM classification call -- and only one
-- when the regex heuristic finds nothing after a threshold of pages."""
import json
from unittest.mock import AsyncMock, MagicMock, patch

import pytest

from app.services.content import website


def _rendered(url, links, title="Page"):
    """(content, links, meta) tuple shaped like get_rendered_content's
    return value. links is a list of (href, text) tuples."""
    return (
        f"<html><title>{title}</title></html>",
        [{"href": href, "text": text} for href, text in links],
        {"title": title},
    )


@pytest.mark.asyncio
async def test_product_shaped_links_are_visited_before_nav_links():
    # Homepage links to three nav/collection pages and one product page,
    # in that order. Even though the product link is discovered last, it
    # must be visited before the nav pages that were discovered first.
    home = "https://shop.example.com/"
    order = []

    async def fake_render(url, *a, **kw):
        order.append(url)
        if url == home:
            return _rendered(home, [
                ("https://shop.example.com/collections/all", "Shop All"),
                ("https://shop.example.com/collections/shirts", "Shirts"),
                ("https://shop.example.com/pages/about", "About"),
                ("https://shop.example.com/products/oxford-shirt", "Oxford Shirt"),
            ])
        return _rendered(url, [])

    fake_browser = AsyncMock()
    fake_browser.get_rendered_content = fake_render

    with patch.object(website, "get_browser", AsyncMock(return_value=fake_browser)):
        await website.extract_text_from_url(home, max_depth=2, max_pages=3)

    # First visit is always the seed URL. Of the remaining budget (2 more
    # pages), the product page must be one of them despite being discovered
    # last -- the nav pages that would fill a plain-BFS budget must not
    # crowd it out.
    assert order[0] == home
    assert "https://shop.example.com/products/oxford-shirt" in order[1:3]


@pytest.mark.asyncio
async def test_llm_fallback_fires_once_after_threshold_with_no_matches():
    home = "https://shop.example.com/"
    visited_urls = [home] + [f"https://shop.example.com/browse/{i}" for i in range(10)]

    async def fake_render(url, *a, **kw):
        idx = visited_urls.index(url) if url in visited_urls else len(visited_urls)
        # Every page links to the next unvisited /browse/N page -- none of
        # these match the regex heuristic, forcing the fallback to fire.
        next_links = []
        if idx + 1 < len(visited_urls):
            next_links = [(visited_urls[idx + 1], "Next")]
        return _rendered(url, next_links)

    fake_browser = AsyncMock()
    fake_browser.get_rendered_content = fake_render

    classify_calls = []

    async def fake_classify(candidates, tenant_id):
        classify_calls.append(candidates)
        return set()

    with patch.object(website, "get_browser", AsyncMock(return_value=fake_browser)), \
         patch.object(website, "_classify_product_links", fake_classify), \
         patch.object(website.settings, "CRAWL_PRODUCT_FALLBACK_AFTER_PAGES", 3):
        await website.extract_text_from_url(home, max_depth=10, max_pages=10, tenant_id="org_x")

    assert len(classify_calls) == 1


@pytest.mark.asyncio
async def test_llm_fallback_verdict_reprioritizes_the_frontier():
    home = "https://shop.example.com/"

    async def fake_render(url, *a, **kw):
        if url == home:
            return _rendered(home, [
                ("https://shop.example.com/nav/1", "Nav 1"),
                ("https://shop.example.com/nav/2", "Nav 2"),
                ("https://shop.example.com/nav/3", "Nav 3"),
                ("https://shop.example.com/sku/12345", "Oxford Shirt"),
            ])
        return _rendered(url, [])

    fake_browser = AsyncMock()
    fake_browser.get_rendered_content = fake_render

    async def fake_classify(candidates, tenant_id):
        return {"https://shop.example.com/sku/12345"}

    order = []
    real_render = fake_render

    async def tracking_render(url, *a, **kw):
        order.append(url)
        return await real_render(url, *a, **kw)

    fake_browser.get_rendered_content = tracking_render

    with patch.object(website, "get_browser", AsyncMock(return_value=fake_browser)), \
         patch.object(website, "_classify_product_links", fake_classify), \
         patch.object(website.settings, "CRAWL_PRODUCT_FALLBACK_AFTER_PAGES", 1):
        await website.extract_text_from_url(home, max_depth=2, max_pages=3, tenant_id="org_x")

    assert order[0] == home
    assert "https://shop.example.com/sku/12345" in order[1:3]


@pytest.mark.asyncio
async def test_llm_identified_urls_are_actually_visited_when_homepage_exceeds_page_budget():
    # Regression test for the Critical whole-branch-review bug: a homepage
    # that yields more links than max_pages must not let a single traversal
    # batch consume the entire remaining budget before the LLM fallback's
    # verdict ever gets acted on. Before the CRAWL_BATCH_SIZE fix, the first
    # post-seed batch would greedily fill the rest of the page budget with
    # plain nav links (discovered before the non-standard product links),
    # the fallback would fire and correctly identify the product URLs, but
    # the crawl would end (len(visited) == max_pages) before any of them
    # were ever visited.
    home = "https://shop.example.com/"
    nav_links = [(f"https://shop.example.com/nav/{i}", f"Nav {i}") for i in range(90)]
    # Non-standard product URLs -- won't match _looks_like_product_url --
    # discovered last, so they sort to the back of a plain insertion-order
    # frontier.
    goods_links = [
        ("https://shop.example.com/goods/alpha", "Alpha Widget"),
        ("https://shop.example.com/goods/beta", "Beta Widget"),
        ("https://shop.example.com/goods/gamma", "Gamma Widget"),
    ]
    goods_urls = {u for u, _ in goods_links}

    async def fake_render(url, *a, **kw):
        if url == home:
            return _rendered(home, nav_links + goods_links)
        return _rendered(url, [])

    fake_browser = AsyncMock()
    fake_browser.get_rendered_content = fake_render

    async def fake_classify(candidates, tenant_id):
        candidate_urls = {u for u, _ in candidates}
        return goods_urls & candidate_urls

    with patch.object(website, "get_browser", AsyncMock(return_value=fake_browser)), \
         patch.object(website, "_classify_product_links", fake_classify), \
         patch.object(website.settings, "CRAWL_PRODUCT_FALLBACK_AFTER_PAGES", 1):
        result = await website.extract_text_from_url(home, max_depth=2, max_pages=60, tenant_id="org_x")

    visited_urls = set(result["urls_visited"])
    assert goods_urls & visited_urls, (
        "None of the LLM-identified product URLs were visited -- the fallback "
        "verdict fired too late to matter."
    )


@pytest.mark.asyncio
async def test_llm_fallback_failure_falls_back_to_insertion_order_without_raising():
    home = "https://shop.example.com/"

    async def fake_render(url, *a, **kw):
        if url == home:
            return _rendered(home, [(f"https://shop.example.com/browse/{i}", "x") for i in range(5)])
        return _rendered(url, [])

    fake_browser = AsyncMock()
    fake_browser.get_rendered_content = fake_render

    async def raising_classify(candidates, tenant_id):
        raise RuntimeError("boom")

    with patch.object(website, "get_browser", AsyncMock(return_value=fake_browser)), \
         patch.object(website, "_classify_product_links", raising_classify), \
         patch.object(website.settings, "CRAWL_PRODUCT_FALLBACK_AFTER_PAGES", 1):
        result = await website.extract_text_from_url(home, max_depth=2, max_pages=4, tenant_id="org_x")

    assert len(result["urls_visited"]) == 4
