"""
Clicking what the element index cannot see.

THE BUG THIS EXISTS FOR
-----------------------
Asked for the cheapest ORANGE towel, the agent opened IKEA's colour filter,
found the word "orange" in the page text, and then could not act on it. From
its own log, verbatim:

    "I can see color options are listed but I don't see a specific index for
     the orange checkbox"
    "I've tried 5+ times to click the orange color filter but the checkboxes
     aren't appearing as interactive elements"
    -> then it guessed: {"click": {"index": 13646}}

13646 is a DOM node id, not an element index — it was inventing one because
the real control was not in the list. Filter swatches like IKEA's are a
visually-hidden <input type=checkbox> with a styled <label> over it, and a
hidden input is not an interactive element by any reasonable definition, so
the extractor is right to drop it. What was missing was a way to click the
thing the PERSON sees.

WHAT THIS DOES
--------------
Finds the smallest visible element whose text or aria-label matches, scrolls it
into view, and clicks it the way a person would: a real CDP mouse press and
release at its centre. Real events rather than element.click() because a hidden
input needs the LABEL clicked (the browser forwards it), and because sites that
check isTrusted ignore synthetic clicks.
"""

from __future__ import annotations

from typing import Any

FIND_JS = r"""
(needle) => {
  const want = needle.trim().toLowerCase();
  const candidates = [];
  for (const el of document.querySelectorAll('label,button,a,span,div,li,input')) {
    const aria = (el.getAttribute('aria-label') || '').trim().toLowerCase();
    const text = (el.innerText || el.textContent || '').trim().toLowerCase();
    const hit = aria === want || aria.includes(want) || text === want || (text.includes(want) && text.length <= want.length + 25);
    if (!hit) continue;
    const r = el.getBoundingClientRect();
    const cs = getComputedStyle(el);
    // Must be something a person could actually hit.
    if (r.width < 4 || r.height < 4) continue;
    if (cs.visibility === 'hidden' || cs.display === 'none' || Number(cs.opacity) === 0) continue;
    candidates.push({ el, area: r.width * r.height });
  }
  if (!candidates.length) return null;
  // Smallest match: the swatch itself rather than the panel containing it.
  candidates.sort((a, b) => a.area - b.area);
  const el = candidates[0].el;
  el.scrollIntoView({ block: 'center', inline: 'center' });
  const r = el.getBoundingClientRect();
  return JSON.stringify({
    x: Math.round(r.left + r.width / 2),
    y: Math.round(r.top + r.height / 2),
    tag: el.tagName,
    label: (el.innerText || el.getAttribute('aria-label') || '').trim().slice(0, 40),
  });
}
"""


def register_click_text(tools: Any, browser_getter) -> None:
    """Adds a `click_text` action to the agent's toolset."""

    @tools.action(
        "Click a visible control by its TEXT or aria-label when it has no element index — "
        "colour swatches, filter checkboxes, custom widgets. Use this instead of guessing an index."
    )
    async def click_text(text: str) -> Any:
        from browser_use.agent.views import ActionResult

        browser = browser_getter()
        target_id = await browser.get_focused_target()
        cdp = await browser.cdp_client_for_target(target_id)

        found = await cdp.send.Runtime.callFunctionOn(
            params={
                "functionDeclaration": FIND_JS,
                "arguments": [{"value": text}],
                "executionContextId": None,
                "returnByValue": True,
                "awaitPromise": True,
            }
        ) if False else await cdp.send.Runtime.evaluate(
            params={
                "expression": f"({FIND_JS})({text!r})",
                "returnByValue": True,
                "awaitPromise": True,
            }
        )

        value = (found or {}).get("result", {}).get("value")
        if not value:
            return ActionResult(
                extracted_content=f"No visible control matching {text!r} — try different wording.",
                include_in_memory=True,
            )

        import json as _json

        spot = _json.loads(value)
        for kind in ("mousePressed", "mouseReleased"):
            await cdp.send.Input.dispatchMouseEvent(
                params={
                    "type": kind,
                    "x": spot["x"],
                    "y": spot["y"],
                    "button": "left",
                    "clickCount": 1,
                }
            )
        return ActionResult(
            extracted_content=f"Clicked {spot['tag']} {spot['label']!r} at ({spot['x']}, {spot['y']}).",
            include_in_memory=True,
        )


TABLE_JS = r"""
(needle) => {
  const want = (needle || '').trim().toLowerCase();
  const tables = [...document.querySelectorAll('table')];
  const scored = tables.map(t => {
    const txt = (t.innerText || '').toLowerCase();
    const rows = t.rows ? t.rows.length : 0;
    let score = rows;
    if (want && txt.includes(want)) score += 1000;
    return { t, score };
  }).sort((a, b) => b.score - a.score);
  if (!scored.length) return null;
  const t = scored[0].t;
  const out = [];
  for (const row of [...t.rows].slice(0, 120)) {
    const cells = [...row.cells].map(c => (c.innerText || '').replace(/\s+/g, ' ').trim().slice(0, 40));
    out.push(cells.join(' | '));
  }
  return JSON.stringify({ rows: t.rows.length, shown: out.length, text: out.join('\n') });
}
"""


def register_read_table(tools, browser_getter) -> None:
    """
    Read a table as a TABLE.

    Traced on a Wikipedia "largest cities" run: 30 steps and 292 seconds spent
    scrolling a long page up and down trying to line up a row (Tokyo) with a
    column header (UN 2018), then failing. Without vision the agent sees a list
    of interactive elements and a slice of text — a grid is exactly the shape it
    cannot reconstruct by dragging a viewport over it.

    So the grid is handed over as a grid: header row first, one row per line,
    which a model reads correctly in a single step.
    """

    @tools.action(
        "Read an HTML table as rows of text (header first, cells separated by |). "
        "Use this for any table, price list or grid instead of scrolling to find a cell. "
        "Pass a word that appears in the table you want, e.g. the row label."
    )
    async def read_table(contains: str = "") -> object:
        from browser_use.agent.views import ActionResult

        browser = browser_getter()
        target_id = await browser.get_focused_target()
        cdp = await browser.cdp_client_for_target(target_id)
        found = await cdp.send.Runtime.evaluate(
            params={
                "expression": f"({TABLE_JS})({contains!r})",
                "returnByValue": True,
                "awaitPromise": True,
            }
        )
        value = (found or {}).get("result", {}).get("value")
        if not value:
            return ActionResult(extracted_content="No table found on this page.", include_in_memory=True)
        import json as _json

        table = _json.loads(value)
        return ActionResult(
            extracted_content=f"Table with {table['rows']} rows (showing {table['shown']}):\n{table['text'][:6000]}",
            include_in_memory=True,
        )
