"""Connect and manage product ingestion sources."""
import logging

import httpx
from fastapi import APIRouter, Body, HTTPException, Query

from app.api.catalog import start_job
from app.services.integrations.platform import get_integration, set_integration_status
from app.core.config import settings
from app.services.catalog.build import build_catalog
from app.services.catalog.sources import (
    KIND_CRAWL, KIND_HTTP_API, KIND_SHOPIFY, get_sources, mark_source_status,
    upsert_source,
)
from app.services.infra import jobs
from app.services.infra.ssrf import UnsafeUrlError, assert_safe_url

logger = logging.getLogger(__name__)

router = APIRouter(prefix="/sources", tags=["sources"])


def _needs(kind: str, config: dict) -> list:
    """What must still be supplied before this source's products price correctly.

    A source with no currency still ingests -- its products are simply flagged
    incomplete and excluded from serving, which from the API looks identical
    to an empty catalogue. This is the surfaced prompt that replaces guessing:
    the merchant can see exactly what's missing instead of silently getting
    nothing recommended.

    KIND_CRAWL is excluded here even though its config can also lack a
    currency: fetch_products() resolves it at crawl time via an LLM
    inference over page context (see crawl.py's _infer_currency) when no
    page's own markup states one, so there is nothing for the merchant to
    supply. KIND_HTTP_API has no such fallback -- an API's JSON has no
    equivalent of schema.org markup to infer from, so it still needs it.
    """
    needs = []
    if kind == KIND_HTTP_API and not config.get("currency"):
        needs.append("currency")
    return needs


@router.post("/http")
async def create_http_source(payload: dict = Body(...)):
    """Connect a custom product API.

    Only a base URL and credentials are required. The field map and
    url_template are inferred by the pipeline on first sync -- a merchant
    should not have to describe their own API's shape. Currency is not
    inferred or assumed from anything else the tenant has connected (a wrong
    guess there silently mis-prices a whole catalog); it is either supplied
    here or reported back via "needs" so the merchant knows to set it. This
    endpoint never refuses a connection for any of their absence.
    """
    tenant_id = payload.get("tenant_id")
    base_url = payload.get("base_url")
    if not tenant_id or not base_url:
        raise HTTPException(status_code=422,
                            detail="tenant_id and base_url are required")

    try:
        assert_safe_url(base_url)
    except UnsafeUrlError as ex:
        raise HTTPException(status_code=400, detail=str(ex))

    external_ref = httpx.URL(base_url).host
    config = {
        "base_url": base_url,
        "records_path": payload.get("records_path") or "$.products",
        "pagination": payload.get("pagination") or {
            "style": "offset", "limit_param": "limit",
            "offset_param": "skip", "page_size": 30, "total_path": "$.total"},
        # Empty until first sync infers it. Overrides are accepted for an API
        # that defeats inference, but none is required.
        "fields": payload.get("fields") or {},
        "url_template": payload.get("url_template"),
        "currency": payload.get("currency"),
        "auth": payload.get("auth") or {"style": "none"},
    }
    credentials = {"api_key": payload["api_key"]} if payload.get("api_key") else {}

    upsert_source(tenant_id, KIND_HTTP_API, external_ref, config, credentials)
    return {"kind": KIND_HTTP_API, "external_ref": external_ref,
            "status": "active", "mapping": "pending_first_sync",
            "needs": _needs(KIND_HTTP_API, config)}


@router.post("/website")
async def create_website_source(payload: dict = Body(...)):
    """Connect the merchant's own website, read by crawling it.

    The third way in, for a merchant with neither a spreadsheet nor an API: they
    give us a page on their site and we read the products off it. Prices come
    from the page's schema.org markup where it has any -- data the merchant
    already publishes for search engines, so it is exact rather than inferred.
    A page without it still yields a product, flagged as having no price.

    currency does not need to be supplied here: if no page states one in its
    own markup, fetch_products() infers it from page context (domain, locale,
    shipping/tax text) with a dedicated LLM call at crawl time, rather than
    asking the caller for it up front.

    Saving the connection also starts a build (sync -> enrich -> pair) for the
    whole tenant, not just this source -- pairing has to see every product to
    find cross-product relationships, so there is no such thing as pairing
    scoped to one newly added source. Every OTHER already-connected source
    still gets its normal 24h re-crawl throttle in that build; this one is
    always crawled fresh since it has never been read before. Set
    "auto_build": false to skip this and connect without triggering anything
    (e.g. a caller adding several sources in one batch that wants a single
    build at the end instead of one per source).
    """
    tenant_id = payload.get("tenant_id")
    url = payload.get("url") or payload.get("base_url")
    if not tenant_id or not url:
        raise HTTPException(status_code=422,
                            detail="tenant_id and url are required")

    try:
        assert_safe_url(url)
    except UnsafeUrlError as ex:
        raise HTTPException(status_code=400, detail=str(ex))

    external_ref = httpx.URL(url).host
    config = {
        "url": url,
        "currency": payload.get("currency"),
        "max_pages": payload.get("max_pages") or settings.CRAWL_MAX_PAGES,
        # A crawl is dozens of requests to the merchant's own server, so it is
        # not repeated on every build the way an API call is. Zero opts out.
        "min_interval_hours": payload.get("min_interval_hours"),
    }

    # No credentials: a public page needs none, and storing an empty envelope
    # would imply this source has an account behind it.
    upsert_source(tenant_id, KIND_CRAWL, external_ref, config, {})

    response = {"kind": KIND_CRAWL, "external_ref": external_ref,
                "status": "active", "products": "pending_first_crawl",
                "needs": _needs(KIND_CRAWL, config)}

    if payload.get("auto_build", True):
        def run(tenant, progress=None):
            return build_catalog(tenant, progress=progress)

        try:
            response["job_id"] = await start_job(tenant_id, "build", run)
            response["job_status"] = "queued"
        except jobs.JobAlreadyRunning as running:
            # That build already read the source list before this one
            # existed, so it will not include this source -- surfaced here
            # rather than silently reused, so the caller knows to press
            # build again once it finishes.
            response["job_id"] = running.job_id
            response["job_status"] = "already_running_without_this_source"
        except HTTPException as ex:
            logger.warning("Could not auto-start build after connecting %s: %s",
                           external_ref, ex.detail)
            response["job_status"] = "not_started"

    return response


@router.get("")
async def list_sources(tenant_id: str = Query(...)):
    """Every source for a tenant. Never includes credentials."""
    sources = [
        {"kind": s["kind"], "external_ref": s["external_ref"],
         "config": s["config"], "status": s["status"],
         "connected_at": s["connected_at"], "last_synced_at": s["last_synced_at"],
         "needs": _needs(s["kind"], s["config"])}
        for s in get_sources(tenant_id)
    ]

    integration = get_integration(tenant_id, KIND_SHOPIFY)
    return {"sources": sources,
            "platform_integration": {
                "shopify": integration["status"] if integration else None}}


@router.delete("/{kind}/{external_ref}")
async def delete_source(kind: str, external_ref: str, tenant_id: str = Query(...)):
    matches = [s for s in get_sources(tenant_id)
               if s["kind"] == kind and s["external_ref"] == external_ref]
    if not matches:
        raise HTTPException(status_code=404, detail="No such connected source")

    mark_source_status(kind, external_ref, "revoked")
    if kind == KIND_SHOPIFY:
        # A revoked product_sources row alone would leave integrations.status
        # ACTIVE, so the platform side would keep believing the shop is
        # connected. Both records describe the same connection and must
        # agree.
        set_integration_status(tenant_id, KIND_SHOPIFY, "INACTIVE")
    return {"kind": kind, "external_ref": external_ref, "status": "revoked"}
