import anyio
import json
import logging
from datetime import datetime
from typing import List, Dict

from openai import BadRequestError

from app.core.config import settings
from app.core.llm_client import client as openai_client
from app.core.prompts import chat_system_prompt
from app.services.chat.tools import TOOL_SCHEMAS, dispatch_tool
from app.services.chat.tools_settings import disabled_tools_for
from app.services.infra.database import (
    get_latest_summary, get_ticket_by_thread, insert_feedback, insert_llm_usage, insert_summary,
)
from app.services.integrations.firestore import get_chat_history, save_chat_message, get_takeover_status
from app.services.infra.quotas import check_quota, QuotaExceededError, estimate_text_tokens
from app.services.infra.redis import publish_event

logger = logging.getLogger(__name__)

# An agent needs room to search, refine the search, act, then answer.
MAX_TOOL_ROUNDS = 6


def _is_content_filter(err) -> bool:
    """True when a 400 came from Azure's content-management policy."""
    body = getattr(err, "body", None) or {}
    if isinstance(body, dict):
        inner = body.get("error") or body
        if inner.get("code") == "content_filter":
            return True
    return "content_filter" in str(err) or "content management policy" in str(err)


async def get_chat_response(tenant_id: str, user_query: str, thread_id: str = "default", user_name: str = None, role: str = "user", user_id: str = None, location: str = None):
    """
    Core RAG-based chat logic with AI ticketing and mandatory integrated analytics.
    """
    logger.info(f"Processing chat request for tenant {tenant_id}, user_id {user_id}, thread {thread_id}, user {user_name}")
    
    try:
        # 0. Early Quota Enforcement: AI Tokens (Per message)
        check_quota(tenant_id, "ai_tokens", user_id=user_id)
        # 1. Fetch history from Firestore
        history = get_chat_history(tenant_id, thread_id, user_id=user_id)
        is_new_thread = len(history) == 0
        
        # 2. Quota Enforcement: AI Conversations (New threads only)
        if is_new_thread:
            check_quota(tenant_id, "ai_conversations", user_id=user_id)
        
        # 2. Early Exit for Admin
        if role == "admin":
            logger.info("Admin response detected. Logging to Firestore and skipping AI processing.")
            save_chat_message(tenant_id, role, user_query, thread_id, user_name, "Admin Interaction", user_id=user_id)
            return {"text": "Message saved successfully.", "resources": {"urls": [], "images": []}}
            
        # 3. Exit if Admin already took over (Legacy or Explicit Flag)
        is_takeover_active = get_takeover_status(tenant_id, thread_id, user_id=user_id)
        if is_takeover_active and role == "user":
            logger.info(f"Thread {thread_id} is under human takeover. Logging message and skipping AI.")
            save_chat_message(tenant_id, role, user_query, thread_id, user_name, "User Interaction (Human Handled)", user_id=user_id)
            return {"text": "Human takeover active. A real person will respond shortly.", "resources": {"urls": [], "images": []}}

        # 3.5 Save User Message instantly to ensure thread document existence for signals
        save_chat_message(tenant_id, role, user_query, thread_id, user_name, "Chat Interaction", user_id=user_id)

        # 4. Retrieval is NOT forced here any more. The model decides whether and what to
        # look up by calling the search_knowledge_base tool inside the loop below.

        # 3. Construct Prompt with Session Context
        disabled_tools = disabled_tools_for(tenant_id)
        system_prompt = chat_system_prompt(disabled_tools)

        # Check for existing ticket in this session to inform the AI
        existing_ticket_initial = get_ticket_by_thread(tenant_id, thread_id)
        if existing_ticket_initial:
            session_context = f"\n\n[SESSION CONTEXT: A support ticket (ID: {existing_ticket_initial['ticket_id']}) is already active for this session. Do NOT ask for the user's name, email, or phone number again as we already have them. Only focus on the new issue details if the user reports something else.]"
            system_prompt += session_context

        if user_name:
            user_context = f"\n\n[USER INFO: The person you are talking to identification is {user_name}. Do NOT ask for their name.]"
            system_prompt += user_context
            
        # 3.6 Location-Based Language Preference
        if location:
            logger.info(f"Updating location preference for tenant {tenant_id} to: {location}")
            insert_summary(tenant_id, "location", location)
        
        current_location = get_latest_summary(tenant_id, "location")
        if current_location:
            language_context = f"\n\n[LANGUAGE: Use professional business {current_location} English. Adapt your vocabulary, tone, and spelling to align with local business standards in {current_location}. Prefer English unless context strongly suggests otherwise.]"
            system_prompt += language_context

        brand_persona = get_latest_summary(tenant_id, "brand_persona")
        if brand_persona:
            persona_context = f"\n\n[BRAND PERSONA: You must strictly follow these personality traits while answering: {brand_persona}]"
            system_prompt += persona_context
            
        messages = [{"role": "system", "content": system_prompt}]
        for msg in history:
            role_val = msg["role"]
            if role_val == "admin":
                role_val = "assistant"
            content_val = msg["content"]
            messages.append({"role": role_val, "content": content_val})
            
        display_query = user_query
            
        messages.append({"role": "user", "content": display_query})
        
        # 4. Quota Enforcement: AI Responses
        check_quota(tenant_id, "ai_responses", user_id=user_id)
        projected_chat_tokens = sum(estimate_text_tokens(msg.get("content", "")) for msg in messages) + 1000
        check_quota(tenant_id, "ai_tokens", user_id=user_id, requested_amount=projected_chat_tokens)

        # A tenant can turn emailing off. Withhold the tool entirely rather than
        # letting the model offer to email and then fail -- the model cannot
        # promise something it was never given.
        active_tools = TOOL_SCHEMAS
        if disabled_tools:
            active_tools = [t for t in TOOL_SCHEMAS
                            if t["function"]["name"] not in disabled_tools]
            logger.info(f"Tools disabled for {tenant_id}: {sorted(disabled_tools)}")

        # 4.5 Agentic tool loop (max MAX_TOOL_ROUNDS rounds)
        logger.info("Calling LLM with tool loop...")
        tool_ctx = {"tenant_id": tenant_id, "thread_id": thread_id,
                    "user_name": user_name, "user_id": user_id, "history": history}
        resources = []
        products = []
        tools_used = []
        ai_response = None
        total_prompt_tokens = 0
        total_completion_tokens = 0

        for round_index in range(MAX_TOOL_ROUNDS):
            # Per-round quota enforcement: a multi-round turn must not overshoot the cap.
            # Recompute the projection from the current messages list each round, since
            # assistant/tool messages appended by earlier rounds grow the prompt.
            if round_index > 0:
                current_projected_tokens = sum(
                    estimate_text_tokens(msg.get("content") or "") for msg in messages
                ) + 1000
                check_quota(tenant_id, "ai_tokens", user_id=user_id,
                            requested_amount=current_projected_tokens)

            # The OpenAI SDK client is synchronous; run it off the event loop.
            response = await anyio.to_thread.run_sync(
                lambda: openai_client.chat.completions.create(
                    model=settings.LLM_MODEL,
                    messages=messages,
                    tools=active_tools,
                    max_completion_tokens=1000,
                    timeout=60.0,
                )
            )
            usage = response.usage
            if usage is not None:
                total_prompt_tokens += usage.prompt_tokens
                total_completion_tokens += usage.completion_tokens

            msg = response.choices[0].message
            if not msg.tool_calls:
                ai_response = msg.content
                break

            messages.append({
                "role": "assistant",
                "content": msg.content or "",
                "tool_calls": [
                    {"id": tc.id, "type": "function",
                     "function": {"name": tc.function.name, "arguments": tc.function.arguments}}
                    for tc in msg.tool_calls
                ],
            })
            for tc in msg.tool_calls:
                try:
                    tool_args = json.loads(tc.function.arguments)
                except (json.JSONDecodeError, TypeError):
                    logger.warning(f"Malformed tool arguments from model for {tc.function.name}")
                    tool_args = {}
                if tc.function.name == "attach_resources":
                    candidate_resources = tool_args.get("resources", [])
                    if isinstance(candidate_resources, list):
                        resources.extend(candidate_resources)
                    else:
                        logger.warning(
                            f"attach_resources returned non-list 'resources' "
                            f"({type(candidate_resources).__name__}); ignoring."
                        )
                tools_used.append(tc.function.name)
                tool_result = await dispatch_tool(tc.function.name, tool_args, tool_ctx)
                # ensure_ascii=False keeps retrieved KB content readable (currency symbols,
                # accents, dashes) instead of \uXXXX escapes. Nothing here truncates the
                # result: search hits reach the model in full.
                messages.append({"role": "tool", "tool_call_id": tc.id,
                                 "content": json.dumps(tool_result, ensure_ascii=False)})
                # Unlike attach_resources, the document link is produced by the
                # executor rather than supplied by the model, so it is collected
                # from the result. The model cannot invent a download URL.
                if (tc.function.name == "send_document"
                        and isinstance(tool_result, dict)
                        and tool_result.get("status") == "sent"):
                    resources.append({"label": tool_result["file_name"],
                                      "url": tool_result["url"],
                                      "image": None})
                if (tc.function.name == "recommend_products"
                        and isinstance(tool_result, dict)
                        and tool_result.get("status") == "found"):
                    products.extend(tool_result["products"])
        else:
            logger.warning(
                f"Tool loop hit the {MAX_TOOL_ROUNDS}-round bound without a final text response "
                f"(tenant={tenant_id}, thread={thread_id}). Returning fallback text."
            )

        if not ai_response:
            ai_response = ("I've noted your request and taken the appropriate action. "
                           "Is there anything else I can help with?")

        insert_llm_usage(tenant_id, "Chat", settings.LLM_MODEL, total_prompt_tokens,
                         total_completion_tokens, total_prompt_tokens + total_completion_tokens,
                         {"thread_id": thread_id})

        # 8. Save exchange to history (Assistant Response - User query already saved)
        save_chat_message(tenant_id, "assistant", ai_response, thread_id, user_name, "Chat Interaction", user_id=user_id)
        
        # 9. Save to Postgres (strategist_feedback)
        try:
            insert_feedback(tenant_id, {
                "thread_id": thread_id,
                "user_name": user_name,
                "question": user_query,
                "answer": ai_response,
                "metadata": {"interaction_type": "chat", "user_id": user_id} 
            })
        except Exception as pg_err:
            logger.error(f"Failed to log interaction to Postgres for tenant {tenant_id}: {pg_err}", exc_info=True)
               # 9. Real-time Discovery: Notify Admin Dashboard if this is a new thread
        if is_new_thread:
            try:
                thread_payload = {
                    "thread_id": thread_id,
                    "user_id": user_id,
                    "user_name": user_name or "Anonymous",
                    "summary": "",
                    "takeover_active": False,
                    "created_at": datetime.utcnow().isoformat(),
                    "updated_at": datetime.utcnow().isoformat(),
                    "last_message_preview": user_query[:100] + "..."
                }
                payload = {
                    "type": "system",
                    "event": "new_thread_created",
                    "message": "A new conversation has started",
                    "data": thread_payload
                }
                await publish_event(f"tenant_{tenant_id}:conversation_list", payload)
            except Exception as ws_err:
                logger.warning(f"Failed to publish new thread to redis: {ws_err}")

        # "tools" is additive observability (which actions the model chose this turn);
        # existing consumers reading only text/resources are unaffected.
        return {"text": ai_response, "resources": resources, "tools": tools_used,
                "products": products}
    except QuotaExceededError as qe:
        logger.warning(f"Quota exceeded during chat: {qe}")
        raise
    except BadRequestError as bre:
        # Azure's content filter rejects the request with a 400. That is the user's
        # message being blocked by policy, not a system failure — saying "I
        # encountered an error" makes a working service look broken.
        if _is_content_filter(bre):
            logger.warning(f"Azure content filter blocked the request for thread {thread_id}.")
            return {
                "text": "I can't help with that request. Please rephrase it, or ask me "
                        "something about our products, pricing or support.",
                "resources": [],
                "tools": [],
            }
        logger.error(f"Bad request to the LLM: {bre}", exc_info=True)
        return {"text": "I encountered an error. Please try again.", "resources": [], "tools": []}
    except Exception as e:
        logger.error(f"Error in chat generation: {e}", exc_info=True)
        return {"text": "I encountered an error. Please try again.", "resources": [], "tools": []}
