"""The tenant's uploaded documents, as the settings page and the chat see them.

A document is an ingested file: its text lives in the knowledge base so the bot
can answer from it, and the original file sits in blob storage so the bot can
hand the actual PDF to a visitor who asks for it.

Both facts come from the same place -- the metadata on the file's knowledge base
rows -- so there is no separate documents table to keep in sync. The category is
one more key in that metadata.

Links given to visitors point at this service, not at blob storage, and never
expire -- a link left in a chat transcript still works months later. The real
storage link is a SAS token minted when the visitor clicks, valid for LINK_HOURS,
which keeps the storage account private and out of the chat payload.
"""
import logging
import uuid
from urllib.parse import quote

from psycopg2 import sql

from app.core.config import settings
from app.services.infra.database import (
    delete_vector_data_by_source, get_db_connection, get_source_ids_by_name,
    insert_summary, insert_vector_data,
)
from app.services.infra.embedding import chunk_text, generate_summary, get_embeddings
from app.services.content.extraction import extract_text_from_bytes
from app.services.infra.storage import storage_service

logger = logging.getLogger(__name__)

LINK_HOURS = 24

# Categories are whatever the settings dropdown offers. The backend stores the
# string and does not police the list, so adding a category is a frontend change.
DEFAULT_CATEGORY = "Uncategorised"


def _table_exists(cur, tenant_id: str) -> bool:
    """Some tenants have a schema but were never ingested into.

    Checking the schema alone is not enough: querying a missing table raises and
    would log a stack trace on every call for those tenants.
    """
    cur.execute(
        "SELECT EXISTS (SELECT FROM information_schema.tables "
        "WHERE table_schema = %s AND table_name = 'strategist_knowledge_base')",
        (tenant_id,),
    )
    return cur.fetchone()[0]


def ingest_documents(tenant_id: str, uploads: list, category: str = None,
                     user_id: str = None) -> dict:
    """Index the uploaded files and store the originals.

    `uploads` is a list of (file_name, content_bytes). Re-uploading a name
    replaces the previous version, chunks and blob alike, so the tenant does not
    end up with two copies answering the same question.

    Returns the generated summary and a row per stored document, so a caller can
    show what was added without a second listing request.
    """
    category = (category or "").strip() or DEFAULT_CATEGORY
    stored, all_text = [], ""

    for file_name, content in uploads:
        for old_id in get_source_ids_by_name(tenant_id, file_name, "file"):
            logger.info(f"Replacing existing file {file_name} ({old_id}) in {tenant_id}")
            delete_vector_data_by_source(tenant_id, old_id)
            try:
                storage_service.delete_file(tenant_id, f"{old_id}_{file_name}")
            except Exception as ex:
                logger.warning(f"Could not delete the old blob for {file_name}: {ex}")

        text = extract_text_from_bytes(content, file_name)
        all_text += text + "\n"

        source_id = str(uuid.uuid4())
        metadata = {"source": file_name, "source_id": source_id,
                    "ingestion_type": "file", "category": category}
        if user_id:
            metadata["user_id"] = user_id

        chunks = chunk_text(text)
        logger.info(f"Created {len(chunks)} chunks for {file_name} ({source_id})")
        for chunk in chunks:
            embedding = get_embeddings(chunk, tenant_id=tenant_id, user_id=user_id)
            insert_vector_data(tenant_id, chunk, embedding, metadata)

        storage_service.upload_file(tenant_id, f"{source_id}_{file_name}", content)
        stored.append({"source_id": source_id, "file_name": file_name,
                       "category": category})

    summary = generate_summary(all_text, tenant_id=tenant_id, user_id=user_id)
    insert_summary(tenant_id, "files", summary)
    return {"summary": summary, "documents": stored}


def delete_document(tenant_id: str, source_id: str) -> dict:
    """Remove a document's knowledge base rows and its stored file.

    The file name is looked up here rather than asked of the caller: the generic
    delete leaves the blob behind when the caller omits it, which quietly
    accumulates orphaned files nobody can reach.
    """
    match = next((d for d in list_documents(tenant_id) if d["source_id"] == source_id), None)
    if not match:
        return {}

    try:
        storage_service.delete_file(tenant_id, f"{source_id}_{match['file_name']}")
    except Exception as ex:
        # Losing the blob is not a reason to keep the document searchable.
        logger.warning(f"Could not delete the blob for {match['file_name']}: {ex}")

    deleted = delete_vector_data_by_source(tenant_id, source_id)
    logger.info(f"Deleted document {match['file_name']} ({deleted} chunks) from {tenant_id}")
    return {"source_id": source_id, "file_name": match["file_name"],
            "deleted_chunks": deleted}


def list_documents(tenant_id: str, category: str = None) -> list:
    """Every uploaded file for the tenant, newest first."""
    conn = get_db_connection()
    try:
        with conn.cursor() as cur:
            if not _table_exists(cur, tenant_id):
                return []
            query = sql.SQL("""
                SELECT metadata->>'source_id',
                       metadata->>'source',
                       metadata->>'category',
                       MIN(created_at)
                FROM {}.strategist_knowledge_base
                WHERE metadata->>'ingestion_type' = 'file'
                GROUP BY 1, 2, 3
                ORDER BY 4 DESC
            """).format(sql.Identifier(tenant_id))
            cur.execute(query)
            documents = []
            for source_id, name, cat, created in cur.fetchall():
                if not source_id or not name:
                    continue
                cat = cat or DEFAULT_CATEGORY
                if category and cat.lower() != category.lower():
                    continue
                documents.append({
                    "source_id": source_id,
                    "file_name": name,
                    "category": cat,
                    "uploaded_at": created.isoformat() if created else None,
                })
            return documents
    except Exception as ex:
        logger.error(f"Could not list documents for {tenant_id}: {ex}", exc_info=True)
        return []
    finally:
        conn.close()


def set_category(tenant_id: str, source_id: str, category: str) -> bool:
    """Re-categorise a document. Every chunk of the file carries the metadata."""
    category = (category or "").strip() or DEFAULT_CATEGORY
    conn = get_db_connection()
    try:
        with conn.cursor() as cur:
            if not _table_exists(cur, tenant_id):
                return False
            query = sql.SQL("""
                UPDATE {}.strategist_knowledge_base
                SET metadata = jsonb_set(metadata::jsonb, '{{category}}', to_jsonb(%s::text))
                WHERE metadata->>'source_id' = %s
            """).format(sql.Identifier(tenant_id))
            cur.execute(query, (category, source_id))
            changed = cur.rowcount
        conn.commit()
        logger.info(f"Set category '{category}' on {changed} chunks of {source_id} ({tenant_id})")
        return changed > 0
    except Exception as ex:
        conn.rollback()
        logger.error(f"Could not set category for {source_id}: {ex}", exc_info=True)
        return False
    finally:
        conn.close()


def find_document(tenant_id: str, source_id: str) -> dict:
    """One document by its id, or {} if this tenant has no such document."""
    if not source_id:
        return {}
    return next((d for d in list_documents(tenant_id)
                 if d["source_id"] == source_id), {})


def blob_url(tenant_id: str, document: dict) -> str:
    """A short-lived direct link to the stored file.

    Minted at the moment someone asks for it, never handed out in advance, so
    the window in which it is useful to anyone who intercepts it is small.
    """
    if not document:
        return ""
    return storage_service.download_url(
        tenant_id, f"{document['source_id']}_{document['file_name']}", hours=LINK_HOURS)


def download_link(tenant_id: str, file_name: str) -> dict:
    """Resolve a file name to a permanent link the visitor can be given.

    The chat model knows file names because search results carry them as
    `source`, so that is what it passes here.

    The returned URL points at this service, not at blob storage. It does not
    expire, so a link sitting in a chat transcript still works next month; the
    real storage link is generated when the visitor actually clicks.
    """
    wanted = (file_name or "").strip().lower()
    if not wanted:
        return {}

    documents = list_documents(tenant_id)
    match = next((d for d in documents if d["file_name"].lower() == wanted), None)

    if not match:
        # Near-miss fallback, for when the model drops the extension or adds a
        # word. Deliberately one-directional: only a name the visitor asked for
        # that is CONTAINED IN a real file name counts. The reverse would let a
        # short file name like "e.pdf" match a request for "brochure.pdf" and
        # hand over the wrong document.
        candidates = [d for d in documents if wanted in d["file_name"].lower()]
        # Ambiguity is worse than a miss: sending one of several possible files
        # looks authoritative and can be the wrong one. Say nothing instead.
        if len(candidates) == 1:
            match = candidates[0]
        elif len(candidates) > 1:
            logger.info(f"{file_name!r} matches {len(candidates)} documents for "
                        f"{tenant_id}; refusing to guess.")

    if not match:
        return {}

    return {"file_name": match["file_name"],
            "category": match["category"],
            "source_id": match["source_id"],
            "url": (f"{settings.PUBLIC_BASE_URL.rstrip('/')}/documents/download"
                    f"?tenant_id={quote(tenant_id)}&source_id={quote(match['source_id'])}")}
