import json
import re
from typing import List, Dict, Any

class DomCondenser:
    """
    Extracts, cleans, and condenses the interactive elements of a web page
    so that Gemini AI can quickly and accurately identify key selectors
    without blowing the token limit or getting bogged down by huge DOM trees.
    """

    EXTRACTOR_JS = """
    () => {
        const interactiveSelectors = [
            'button',
            'a[href]',
            'input',
            'textarea',
            '[role="button"]',
            '[role="tab"]',
            '[role="dialog"]',
            '[role="textbox"]',
            '[contenteditable="true"]',
            '[aria-label]',
            '[data-testid]',
            'form'
        ];

        const elements = document.querySelectorAll(interactiveSelectors.join(','));
        const results = [];
        let index = 0;

        for (const el of elements) {
            // Skip hidden or invisible elements
            const rect = el.getBoundingClientRect();
            const style = window.getComputedStyle(el);
            if (
                style.display === 'none' ||
                style.visibility === 'hidden' ||
                style.opacity === '0' ||
                (rect.width === 0 && rect.height === 0 && el.tagName !== 'INPUT')
            ) {
                // Keep hidden file inputs because they are often triggered programmatically
                if (!(el.tagName === 'INPUT' && el.type === 'file')) {
                    continue;
                }
            }

            const tag = el.tagName.toLowerCase();
            const type = el.getAttribute('type') || '';
            const ariaLabel = el.getAttribute('aria-label') || '';
            const placeholder = el.getAttribute('placeholder') || '';
            const role = el.getAttribute('role') || '';
            const testId = el.getAttribute('data-testid') || '';
            const name = el.getAttribute('name') || '';
            const id = el.id || '';
            
            // Text content (truncate if too long)
            let text = (el.innerText || el.textContent || '').trim().replace(/\\s+/g, ' ');
            if (text.length > 80) {
                text = text.substring(0, 80) + '...';
            }

            // Build a reliable CSS selector or locator hint
            let suggestedSelector = '';
            if (id && !id.match(/[0-9]{5,}/)) {
                suggestedSelector = `#${id}`;
            } else if (testId) {
                suggestedSelector = `[data-testid="${testId}"]`;
            } else if (ariaLabel) {
                suggestedSelector = `${tag}[aria-label="${ariaLabel}"]`;
            } else if (name) {
                suggestedSelector = `${tag}[name="${name}"]`;
            } else if (placeholder) {
                suggestedSelector = `${tag}[placeholder="${placeholder}"]`;
            } else if (tag === 'input' && type === 'file') {
                suggestedSelector = `input[type="file"]`;
            } else if (text && text.length < 30) {
                suggestedSelector = `text="${text}"`;
            }

            results.push({
                index: index++,
                tag: tag,
                type: type,
                role: role,
                aria_label: ariaLabel,
                placeholder: placeholder,
                text: text,
                id: id,
                suggested_selector: suggestedSelector,
                is_file_input: (tag === 'input' && type === 'file'),
                is_contenteditable: (el.getAttribute('contenteditable') === 'true')
            });

            if (results.length >= 120) {
                break; // Cap to keep token usage very low
            }
        }

        return {
            title: document.title,
            url: window.location.href,
            elements: results
        };
    }
    """

    @classmethod
    async def extract_from_page(cls, page) -> Dict[str, Any]:
        """Extract condensed DOM information asynchronously from Playwright page."""
        try:
            condensed = await page.evaluate(cls.EXTRACTOR_JS)
            return condensed
        except Exception as e:
            return {
                "title": "",
                "url": page.url,
                "error": str(e),
                "elements": []
            }

    @classmethod
    def extract_from_page_sync(cls, page) -> Dict[str, Any]:
        """Extract condensed DOM information synchronously from Playwright page."""
        try:
            condensed = page.evaluate(cls.EXTRACTOR_JS)
            return condensed
        except Exception as e:
            return {
                "title": "",
                "url": page.url,
                "error": str(e),
                "elements": []
            }
