Chromium Browser 1.0.0

page_do.py

24.3 KB · raw

"""Do things on the open page: click, type, press, scroll, wait."""

from __future__ import annotations

import asyncio
import json

from aworg.tools.base import ToolContext, ToolError, ToolResult

from ._cdp import BrowserError, browser
from ._page import describe, extract, summary


NAME = "page_do"

DESCRIPTION = (
    "Act on the page that is open: click something, type into a field, press "
    "a key, scroll, or wait for something to appear. Takes a list of steps, "
    "does them in order, then reads the page back. Elements are found by "
    "ref, CSS selector or the text on them."
)

INPUT_SCHEMA = {
    "type": "object",
    "properties": {
        "steps": {
            "type": "array",
            "description": "The steps to carry out, in order.",
            "items": {
                "type": "object",
                "properties": {
                    "do": {
                        "type": "string",
                        "description": (
                            "click, type, press, hover, drag, scroll, or "
                            "wait."
                        ),
                    },
                    "ref": {
                        "type": "string",
                        "description": (
                            "A handle from the last read of the page, like "
                            "'ref_12'. Goes stale when the page navigates."
                        ),
                    },
                    "to_ref": {
                        "type": "string",
                        "description": "For drag: the handle to drag onto.",
                    },
                    "keys": {
                        "type": "string",
                        "description": (
                            "Modifiers held down for this step, joined by "
                            "'+': ctrl, shift, alt, meta. 'ctrl+shift'."
                        ),
                    },
                    "selector": {
                        "type": "string",
                        "description": (
                            "CSS selector of the element to act on, for "
                            "click, type and wait."
                        ),
                    },
                    "text": {
                        "type": "string",
                        "description": (
                            "For click: the visible text to find it by, "
                            "instead of a selector. For type: what to type. "
                            "For press: the key, such as Enter or Tab."
                        ),
                    },
                    "seconds": {
                        "type": "number",
                        "description": "For wait: how long, up to 15.",
                    },
                    "to": {
                        "type": "string",
                        "description": (
                            "For scroll: top, bottom, or a selector to bring "
                            "into view."
                        ),
                    },
                },
                "required": ["do"],
            },
        },
        "links": {
            "type": "boolean",
            "description": (
                "Include links and controls in what comes back. Defaults to "
                "true."
            ),
        },
        "follow_popup": {
            "type": "boolean",
            "description": (
                "If a step opens a new tab, follow it and report that page "
                "instead. Defaults to true; turn it off to stay where you "
                "are."
            ),
        },
    },
    "required": ["steps"],
}

MAX_STEPS = 20
MAX_WAIT = 15
#: Keys worth naming. Anything else is typed as a character, which is what a
#: single-character `text` means.
KEYS = {
    "enter": ("Enter", "Enter", 13, "\r"),
    "tab": ("Tab", "Tab", 9, "\t"),
    "escape": ("Escape", "Escape", 27, ""),
    "backspace": ("Backspace", "Backspace", 8, ""),
    "delete": ("Delete", "Delete", 46, ""),
    "arrowup": ("ArrowUp", "ArrowUp", 38, ""),
    "arrowdown": ("ArrowDown", "ArrowDown", 40, ""),
    "arrowleft": ("ArrowLeft", "ArrowLeft", 37, ""),
    "arrowright": ("ArrowRight", "ArrowRight", 39, ""),
}

#: Clicking and typing are done in the page rather than by aiming a mouse at
#: coordinates. A coordinate needs a screenshot, a screenshot costs thousands
#: of tokens, and the result is wrong the moment the layout moves. A
#: selector or the words on the button survive both.
CLICK = """
(sel, label, ref) => {
  const pick = () => {
    if (ref !== null && ref !== undefined) return (window.__aworg_refs || [])[ref];
    if (sel) return document.querySelector(sel);
    const wanted = (label || "").trim().toLowerCase();
    const candidates = [...document.querySelectorAll(
      "button, a, input[type=submit], input[type=button], [role=button], label")];
    return candidates.find((el) =>
      ((el.innerText || el.value || "").trim().toLowerCase() === wanted)) ||
      candidates.find((el) =>
        ((el.innerText || el.value || "").trim().toLowerCase().includes(wanted)));
  };
  const el = pick();
  if (!el) return "nothing matched";
  el.scrollIntoView({ block: "center" });
  // Focus as well as click. A real click does both, and without this a
  // field that was clicked is not the field a following keystroke reaches.
  if (el.focus) { try { el.focus({ preventScroll: true }); } catch (e) { el.focus(); } }
  el.click();
  return "";
}
"""

#: Setting .value directly does not tell a framework anything happened, so
#: the native setter is called and the events a real keystroke would raise
#: are dispatched. Without this, React and friends keep their own state and
#: the field reverts the moment anything re-renders.
TYPE = """
(sel, value, ref) => {
  const el = (ref !== null && ref !== undefined)
    ? (window.__aworg_refs || [])[ref]
    : (sel ? document.querySelector(sel) : document.activeElement);
  if (!el) return "nothing matched";
  el.focus();
  if (el.isContentEditable) {
    el.textContent = value;
  } else {
    const proto = el instanceof HTMLTextAreaElement
      ? HTMLTextAreaElement.prototype : HTMLInputElement.prototype;
    const setter = Object.getOwnPropertyDescriptor(proto, "value");
    if (setter && setter.set) setter.set.call(el, value);
    else el.value = value;
  }
  el.dispatchEvent(new Event("input", { bubbles: true }));
  el.dispatchEvent(new Event("change", { bubbles: true }));
  return "";
}
"""

SCROLL = """
(where) => {
  if (where === "top") { window.scrollTo(0, 0); return ""; }
  if (where === "bottom") { window.scrollTo(0, document.body.scrollHeight); return ""; }
  const el = document.querySelector(where);
  if (!el) return "nothing matched";
  el.scrollIntoView({ block: "center" });
  return "";
}
"""


async def run(
    context: ToolContext,
    steps: list | None = None,
    links: bool = True,
    follow_popup: bool = True,
) -> ToolResult:
    if not isinstance(steps, list) or not steps:
        raise ToolError(
            "page_do needs a list of steps, each with a 'do' of click, type, "
            "press, scroll or wait."
        )
    if len(steps) > MAX_STEPS:
        raise ToolError(f"page_do takes at most {MAX_STEPS} steps at a time.")

    done: list[str] = []
    try:
        page_browser = await browser(start_if_needed=False)
        before = page_browser.url
        for index, step in enumerate(steps, start=1):
            if not isinstance(step, dict):
                raise ToolError(f"Step {index} is not an object.")
            await _step(page_browser, index, step, done, context)

        # A click that opened a tab. The page under the Resident is
        # unchanged, which is exactly what makes this invisible otherwise:
        # it reads the old page back and concludes nothing happened.
        popped = await page_browser.popup()
        if popped and follow_popup:
            await page_browser.follow(popped)
            done.append(f"followed a new tab to {popped}")
        elif popped:
            done.append(f"a new tab opened at {popped}, not followed")

        page = await extract(page_browser)
        del before
    except BrowserError as exc:
        raise ToolError(
            (f"Did: {'; '.join(done)}. Then it failed: " if done else "") + str(exc)
        ) from exc

    return ToolResult(
        text="Did: " + "; ".join(done) + "\n\n" + describe(page_browser, page, links),
        payload={"did": done, "page": page},
        summary=f"{len(done)} step(s), " + summary(page_browser, page),
    )


async def _step(page_browser, index, step, done, context) -> None:
    action = str(step.get("do", "")).strip().lower()
    selector = step.get("selector")
    text = step.get("text")
    ref = _ref(step.get("ref"), index)
    modifiers = _modifiers(step.get("keys"), index)

    if context.activity is not None:
        context.activity.progress = f"step {index}: {action}"

    if action == "click":
        if ref is None and not selector and not text:
            raise ToolError(
                f"Step {index}: click needs a ref, a selector or some text."
            )
        if modifiers:
            await _mouse(page_browser, index, ref, selector, "click", modifiers)
        else:
            problem = await _call(page_browser, CLICK, selector, text, ref)
            if problem:
                raise ToolError(
                    f"Step {index}: nothing on the page matches "
                    f"{step.get('ref') or selector or text!r}. Read the page "
                    "again -- handles go stale when it navigates."
                )
        done.append(f"clicked {step.get('ref') or selector or text!r}")
        # A click is usually meant to change something, and what it changed
        # is the point of the call.
        await asyncio.sleep(0.4)

    elif action == "type":
        if text is None:
            raise ToolError(f"Step {index}: type needs text.")
        problem = await _call(page_browser, TYPE, selector, text, ref)
        if problem:
            raise ToolError(
                f"Step {index}: nothing on the page matches "
                f"{step.get('ref') or selector!r}."
            )
        done.append(
            f"typed into {step.get('ref') or selector or 'the focused field'}"
        )

    elif action == "hover":
        if ref is None and not selector:
            raise ToolError(f"Step {index}: hover needs a ref or a selector.")
        await _mouse(page_browser, index, ref, selector, "hover", modifiers)
        done.append(f"hovered {step.get('ref') or selector!r}")
        # Menus that open on hover take a moment to open.
        await asyncio.sleep(0.3)

    elif action == "drag":
        to_ref = _ref(step.get("to_ref"), index)
        if (ref is None and not selector) or (to_ref is None and not step.get("to")):
            raise ToolError(
                f"Step {index}: drag needs something to drag (ref or "
                "selector) and somewhere to drop it (to_ref or to)."
            )
        await _drag(page_browser, index, ref, selector, to_ref, step.get("to"))
        done.append(
            f"dragged {step.get('ref') or selector!r} onto "
            f"{step.get('to_ref') or step.get('to')!r}"
        )
        await asyncio.sleep(0.3)

    elif action == "press":
        key = str(text or "").strip()
        if not key:
            raise ToolError(f"Step {index}: press needs a key, such as Enter.")
        await _press(page_browser, key, modifiers)
        done.append(f"pressed {key}")
        await asyncio.sleep(0.4)

    elif action == "scroll":
        where = str(step.get("to") or selector or "bottom")
        problem = await page_browser.evaluate(
            f"({SCROLL})({json.dumps(where)})"
        )
        if problem:
            raise ToolError(f"Step {index}: nothing matches {where!r} to scroll to.")
        done.append(f"scrolled to {where}")

    elif action == "wait":
        if selector:
            await _wait_for(page_browser, selector, step.get("seconds"))
            done.append(f"waited for {selector}")
        else:
            seconds = max(0.0, min(float(step.get("seconds") or 1), MAX_WAIT))
            await asyncio.sleep(seconds)
            done.append(f"waited {seconds}s")

    else:
        raise ToolError(
            f"Step {index}: {action!r} is not something page_do can do. It "
            "can click, type, press, hover, drag, scroll and wait."
        )


async def _call(page_browser, function: str, *arguments) -> str:
    """Run one of the page functions above with JSON-safe arguments."""
    packed = ", ".join(json.dumps(argument) for argument in arguments)
    return await page_browser.evaluate(f"({function})({packed})") or ""


#: Modifier bits, as the protocol counts them.
MODIFIERS = {"alt": 1, "ctrl": 2, "control": 2, "meta": 4, "cmd": 4,
             "command": 4, "shift": 8}


def _modifiers(keys, index: int) -> int:
    if not keys:
        return 0
    total = 0
    for part in str(keys).replace(",", "+").split("+"):
        name = part.strip().lower()
        if not name:
            continue
        if name not in MODIFIERS:
            raise ToolError(
                f"Step {index}: {part!r} is not a modifier. They are "
                + ", ".join(sorted(set(MODIFIERS))) + "."
            )
        total |= MODIFIERS[name]
    return total


def _ref(value, index: int):
    """'ref_12' as the number 12, or None if there was no ref."""
    if value in (None, ""):
        return None
    text = str(value).strip()
    if text.startswith("ref_"):
        text = text[4:]
    if not text.isdigit():
        raise ToolError(
            f"Step {index}: {value!r} is not a handle. They look like "
            "'ref_12' and come from reading the page."
        )
    return int(text)


#: Bring something into view. Separate from measuring it, because a scroll
#: is not necessarily finished when the call that asked for it returns, and
#: a position read in the same breath can be the position it was leaving.
SCROLL_TO = """
(sel, ref, sel2, ref2) => {
  const pick = (s, r) => (r === null || r === undefined)
    ? (s ? document.querySelector(s) : null)
    : (window.__aworg_refs || [])[r];
  const first = pick(sel, ref);
  if (!first) return "nothing matched";
  const second = pick(sel2, ref2);
  first.scrollIntoView({ block: "center" });
  if (second) {
    const b = second.getBoundingClientRect();
    const seen = b.bottom > 0 && b.top < innerHeight &&
                 b.right > 0 && b.left < innerWidth;
    // Only if the other end is off screen: bringing it into view when it is
    // already visible would move the first one for no reason.
    if (!seen) second.scrollIntoView({ block: "center" });
  }
  return "";
}
"""

#: Where things are, measured and nothing else.
MEASURE = """
(sel, ref, sel2, ref2) => {
  const pick = (s, r) => (r === null || r === undefined)
    ? (s ? document.querySelector(s) : null)
    : (window.__aworg_refs || [])[r];
  const centre = (el) => {
    if (!el || !el.getBoundingClientRect) return null;
    const b = el.getBoundingClientRect();
    if (!b.width && !b.height) return null;
    return { x: b.left + b.width / 2, y: b.top + b.height / 2,
             seen: b.bottom > 0 && b.top < innerHeight &&
                   b.right > 0 && b.left < innerWidth };
  };
  return { from: centre(pick(sel, ref)), to: centre(pick(sel2, ref2)) };
}
"""

#: How long to let a scroll land before believing a coordinate.
SETTLE = 0.12


async def _positions(page_browser, index, ref, selector,
                     to_ref=None, to_selector=None, what="act on"):
    """Scroll, wait, then measure -- in that order and never fewer steps."""
    moved = await _call(page_browser, SCROLL_TO, selector, ref,
                        to_selector, to_ref)
    if moved == "nothing matched":
        raise ToolError(
            f"Step {index}: nothing to {what} -- "
            f"{step_name(ref, selector)} is not on the page. Read the page "
            "again for fresh handles."
        )
    await asyncio.sleep(SETTLE)
    where = await _call_json(page_browser, MEASURE, selector, ref,
                             to_selector, to_ref)
    if not where or not where.get("from"):
        raise ToolError(
            f"Step {index}: nothing to {what} -- "
            f"{step_name(ref, selector)} is not on the page, or has no size."
        )
    return where


async def _where(page_browser, index, ref, selector, what="act on"):
    return (await _positions(page_browser, index, ref, selector,
                             what=what))["from"]


async def _call_json(page_browser, function: str, *arguments):
    packed = ", ".join(json.dumps(argument) for argument in arguments)
    return await page_browser.evaluate(f"({function})({packed})")


def step_name(ref, selector) -> str:
    """Whichever way the caller named an element, for saying it back."""
    return f"ref_{ref}" if ref is not None else repr(selector)


async def _mouse(page_browser, index, ref, selector, kind, modifiers) -> None:
    """A real mouse event at the element's own position.

    Coordinates rather than el.click(), because a modifier-click and a hover
    are mouse state rather than a method call -- and the coordinates come
    from the element, so nothing here needs a screenshot to aim with.

    Aimed twice. Moving the pointer changes the page: a menu opens, a banner
    collapses, and what was being pointed at is now somewhere else. So the
    pointer moves, the target is measured again, and the press uses that.
    """
    spot = await _where(page_browser, index, ref, selector,
                        "hover" if kind == "hover" else "click")
    await page_browser.command("Input.dispatchMouseEvent", {
        "type": "mouseMoved", "x": spot["x"], "y": spot["y"],
        "modifiers": modifiers,
    })
    if kind == "hover":
        return

    spot = await _aim(page_browser, ref, selector, spot)
    base = {"x": spot["x"], "y": spot["y"], "modifiers": modifiers}
    await page_browser.command("Input.dispatchMouseEvent",
                               {"type": "mouseMoved", **base})
    for kind_of_event in ("mousePressed", "mouseReleased"):
        await page_browser.command("Input.dispatchMouseEvent", {
            "type": kind_of_event, "button": "left", "clickCount": 1, **base,
        })


#: How far something may have moved before it is worth correcting for. Two
#: pixels is sub-pixel layout and rounding; ten is a page that moved.
DRIFT = 3


async def _aim(page_browser, ref, selector, was):
    """Where the target is now, if that is not where it was.

    Falls back to the earlier position rather than failing: an element that
    has just vanished is a case the click itself will report, in better
    words than a measurement could.
    """
    now = await _call_json(page_browser, MEASURE, selector, ref, None, None)
    spot = (now or {}).get("from")
    if not spot:
        return was
    if abs(spot["x"] - was["x"]) < DRIFT and abs(spot["y"] - was["y"]) < DRIFT:
        return was
    return spot


#: HTML5 drag-and-drop, which synthetic mouse events cannot drive: the
#: browser takes over on mousedown and the mouseup never arrives. Dispatched
#: as the events the API itself defines, sharing one DataTransfer so that a
#: handler reading what was dropped finds what was dragged.
NATIVE_DRAG = """
(fromSel, fromRef, toSel, toRef) => {
  const pick = (sel, ref) => (ref === null || ref === undefined)
    ? document.querySelector(sel)
    : (window.__aworg_refs || [])[ref];
  const from = pick(fromSel, fromRef);
  const onto = pick(toSel, toRef);
  if (!from || !onto) return "nothing matched";
  if (!from.draggable) return "not native";
  const data = new DataTransfer();
  const fire = (el, type) => el.dispatchEvent(new DragEvent(type, {
    bubbles: true, cancelable: true, dataTransfer: data,
  }));
  from.scrollIntoView({ block: "center" });
  fire(from, "dragstart");
  fire(onto, "dragenter");
  fire(onto, "dragover");
  fire(onto, "drop");
  fire(from, "dragend");
  return "";
}
"""


async def _drag(page_browser, index, ref, selector, to_ref, to_selector) -> None:
    # The native API first, because an element that declares itself
    # draggable is telling us which of the two kinds of drag it is.
    native = await _call(page_browser, NATIVE_DRAG,
                         selector, ref, to_selector, to_ref)
    if native == "nothing matched":
        raise ToolError(
            f"Step {index}: nothing to drag -- "
            f"{step_name(ref, selector)} or {step_name(to_ref, to_selector)} "
            "is not on the page."
        )
    if native == "":
        return

    where = await _positions(page_browser, index, ref, selector,
                             to_ref, to_selector, "drag")
    start, end = where["from"], where.get("to")
    if not end:
        raise ToolError(
            f"Step {index}: nothing to drop onto -- "
            f"{step_name(to_ref, to_selector)} is not on the page."
        )
    if not start.get("seen") or not end.get("seen"):
        raise ToolError(
            f"Step {index}: {step_name(ref, selector)} and "
            f"{step_name(to_ref, to_selector)} cannot both be on screen at "
            "once, so there is no gesture that goes from one to the other. "
            "Scroll first, or drag to something nearer."
        )
    # Move first, then look again, then press. Moving the pointer here is
    # what closes whatever the pointer was on before, and that is usually
    # what was holding the page in a different shape.
    await page_browser.command("Input.dispatchMouseEvent", {
        "type": "mouseMoved", "x": start["x"], "y": start["y"], "modifiers": 0,
    })
    start = await _aim(page_browser, ref, selector, start)
    await page_browser.command("Input.dispatchMouseEvent", {
        "type": "mouseMoved", "x": start["x"], "y": start["y"], "modifiers": 0,
    })
    await page_browser.command("Input.dispatchMouseEvent", {
        "type": "mousePressed", "button": "left", "clickCount": 1,
        "x": start["x"], "y": start["y"], "modifiers": 0,
    })
    # In steps, because a drag handler that only listens for the drop will
    # take a single jump, and one that tracks movement will not follow at
    # all without something to track.
    for step in range(1, 6):
        await page_browser.command("Input.dispatchMouseEvent", {
            "type": "mouseMoved", "button": "left", "modifiers": 0,
            "x": start["x"] + (end["x"] - start["x"]) * step / 5,
            "y": start["y"] + (end["y"] - start["y"]) * step / 5,
        })
        await asyncio.sleep(0.03)

    # And aim again before letting go. The pointer has crossed the page to
    # get here, and anything that reacts to a pointer has had its say.
    end = await _aim(page_browser, to_ref, to_selector, end)
    await page_browser.command("Input.dispatchMouseEvent", {
        "type": "mouseMoved", "button": "left", "modifiers": 0,
        "x": end["x"], "y": end["y"],
    })
    await page_browser.command("Input.dispatchMouseEvent", {
        "type": "mouseReleased", "button": "left", "clickCount": 1,
        "x": end["x"], "y": end["y"], "modifiers": 0,
    })


async def _press(page_browser, key: str, modifiers: int = 0) -> None:
    """A real key event, because forms listen for keys rather than clicks."""
    named = KEYS.get(key.strip().lower())
    if named:
        code, dom_key, code_number, char = named
    elif len(key) == 1:
        code, dom_key, code_number, char = key, key, ord(key.upper()), key
    else:
        raise ToolError(
            f"{key!r} is not a key page_do knows. It knows "
            + ", ".join(sorted(KEYS)) + ", and any single character."
        )
    base = {"windowsVirtualKeyCode": code_number, "key": dom_key, "code": code,
            "modifiers": modifiers}
    await page_browser.command("Input.dispatchKeyEvent", {
        "type": "rawKeyDown", **base,
    })
    if char:
        await page_browser.command("Input.dispatchKeyEvent", {
            "type": "char", "text": char, **base,
        })
    await page_browser.command("Input.dispatchKeyEvent", {"type": "keyUp", **base})


async def _wait_for(page_browser, selector: str, seconds) -> None:
    limit = max(0.5, min(float(seconds or 10), MAX_WAIT))
    loop = asyncio.get_event_loop()
    deadline = loop.time() + limit
    while loop.time() < deadline:
        if await page_browser.evaluate(
            f"!!document.querySelector({json.dumps(selector)})"
        ):
            return
        await asyncio.sleep(0.2)
    raise ToolError(
        f"{selector!r} did not appear within {limit:g} seconds. "
        "Use read_page to see what the page is showing instead."
    )