"""Cross-tenant isolation, proven against the real development database.

Mocks can only prove the code does what the mock says. These tests go through
the real per-tenant schemas of two live tenants -- one with a real catalogue,
one without -- and assert that tenant B's queries never surface tenant A's
data. Read-only: no rows are created or deleted in either tenant.
"""
import pytest
from dotenv import load_dotenv

load_dotenv()

from app.services.catalog.products import find_by_url, list_products, match_products, normalise_url

TENANT_A = "org_8c32bf3e-6a18-4739-9b1c-94c0cf11125f"  # the test tenant, whichever site is ingested
TENANT_B = "org_79b6738f-df0c-4ce7-9ec7-95c173c0c5a7"  # different tenant, no products


def _tenant_a_products():
    rows = list_products(TENANT_A, limit=1000)
    assert rows, "tenant A is expected to have products for this test to be meaningful"
    return rows


def test_find_by_url_for_tenant_as_product_is_unknown_to_tenant_b():
    a_products = _tenant_a_products()
    a_url = a_products[0]["product_url"]

    found_in_a = find_by_url(TENANT_A, a_url)
    assert found_in_a.get("product_url") == a_url

    found_in_b = find_by_url(TENANT_B, a_url)
    assert found_in_b == {}


def test_list_products_for_tenant_b_excludes_tenant_as_catalogue():
    a_names = {row["name"] for row in _tenant_a_products()}

    b_rows = list_products(TENANT_B, limit=1000)
    b_names = {row["name"] for row in b_rows}

    assert not (a_names & b_names)


def test_match_products_for_tenant_b_does_not_surface_tenant_as_matches():
    """Plain SQL, deliberately -- not `search_products`. That call goes through
    `get_embeddings`, and a combined test run rebinds the OpenAI client to
    whichever settings happened to be frozen in at import time (see
    `tests/integration/conftest.py`), so it can fail on a 401 from an
    unrelated, misconfigured embedding provider and prove nothing about
    tenant scoping. `match_products` is a single ILIKE over this tenant's own
    schema, so it exercises exactly the boundary this test is about.
    """
    # Derive the query from a product tenant A actually has, rather than
    # hardcoding a term. Which site is ingested into the test tenant changes
    # as people test against different catalogues, and a hardcoded term goes
    # red for that reason alone -- which teaches everyone to ignore this test,
    # the one test that guards a real leak.
    catalogue = list_products(TENANT_A, limit=1)
    if not catalogue:
        pytest.skip("tenant A has no products ingested; nothing to prove scoping against")
    query = catalogue[0]["name"].split()[0]

    a_hits = match_products(TENANT_A, query, limit=10)
    # A positive control: if `match_products` were broken and always
    # returned [], the isolation assertion below would pass for the wrong
    # reason -- exactly the failure mode that shipped once already with
    # `_table_exists` silently returning nothing for every tenant.
    assert a_hits, "expected tenant A's own catalogue to match its own query"

    b_hits = match_products(TENANT_B, query, limit=10)
    assert b_hits == []


def test_normalise_url_cannot_merge_two_different_hosts_onto_one_key():
    assert normalise_url("https://a.example.com/p/1") != normalise_url("https://b.example.com/p/1")
