import asyncio
import logging
from playwright.async_api import async_playwright
from typing import Dict, List, Optional, Tuple

logger = logging.getLogger(__name__)

class BrowserService:
    _instance = None
    _browser = None
    _playwright = None

    @classmethod
    async def get_instance(cls):
        if cls._instance is None:
            cls._instance = cls()
            await cls._instance._init_browser()
        return cls._instance

    async def _init_browser(self):
        self._playwright = await async_playwright().start()
        self._browser = await self._playwright.chromium.launch(headless=True)
        logger.info("Browser service initialized with Chromium (headless).")

    async def _new_context(self):
        return await self._browser.new_context(
            user_agent="Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36"
        )

    async def get_rendered_content(self, url: str, wait_until: str = "load", timeout: int = 30000) -> Tuple[str, List[Dict[str, str]], Dict[str, str]]:
        """
        Navigates to a URL and returns the fully rendered HTML and extracted links.
        """
        if self._browser is None or not self._browser.is_connected():
            logger.warning("Browser is disconnected -- relaunching before crawl.")
            await self._init_browser()

        try:
            context = await self._new_context()
        except Exception as e:
            # is_connected() can still report True for a beat after the
            # underlying Chromium process has actually died (observed in
            # production as `Browser.new_context: ... the handler is
            # closed`) -- the pre-check above catches most cases, but this
            # is the real safety net: relaunch on the first failure and
            # retry once instead of failing every crawl until the service
            # is restarted by hand.
            logger.warning(f"Browser context creation failed ({e}); relaunching and retrying once.")
            await self._init_browser()
            context = await self._new_context()

        page = await context.new_page()
        
        try:
            logger.info(f"Navigating to {url} with Playwright (wait_until={wait_until})...")
            try:
                await page.goto(url, wait_until=wait_until, timeout=timeout)
            except Exception as e:
                if "Timeout" in str(e):
                    logger.warning(f"Timeout reached for {url} with {wait_until}. Proceeding with current content.")
                else:
                    raise e
            
            # Additional wait to ensure dynamic content is ready if needed
            await asyncio.sleep(2) 
            
            content = await page.content()
            
            # Extract internal links directly using Playwright, with anchor
            # text so link-classification callers don't need a second pass.
            links = await page.eval_on_selector_all(
                "a[href]",
                "elements => elements.map(e => ({href: e.href, text: (e.textContent || '').trim()}))",
            )
            
            # Basic metadata extraction
            title = await page.title()
            
            return content, links, {"title": title}
        except Exception as e:
            logger.error(f"Playwright failed to fetch {url}: {e}", exc_info=True)
            return "", [], {}
        finally:
            await page.close()
            await context.close()

    async def shutdown(self):
        if self._browser:
            await self._browser.close()
        if self._playwright:
            await self._playwright.stop()
        logger.info("Browser service shut down.")

browser_service = None

async def get_browser():
    global browser_service
    if browser_service is None:
        browser_service = await BrowserService.get_instance()
    return browser_service
