"""Do things on the open page: click, type, press, scroll, wait.""" from __future__ import annotations import asyncio import json from aworg.tools.base import ToolContext, ToolError, ToolResult from ._cdp import BrowserError, browser from ._page import describe, extract, summary NAME = "page_do" DESCRIPTION = ( "Act on the page that is open: click something, type into a field, press " "a key, scroll, or wait for something to appear. Takes a list of steps, " "does them in order, then reads the page back. Elements are found by " "ref, CSS selector or the text on them." ) INPUT_SCHEMA = { "type": "object", "properties": { "steps": { "type": "array", "description": "The steps to carry out, in order.", "items": { "type": "object", "properties": { "do": { "type": "string", "description": ( "click, type, press, hover, drag, scroll, or " "wait." ), }, "ref": { "type": "string", "description": ( "A handle from the last read of the page, like " "'ref_12'. Goes stale when the page navigates." ), }, "to_ref": { "type": "string", "description": "For drag: the handle to drag onto.", }, "keys": { "type": "string", "description": ( "Modifiers held down for this step, joined by " "'+': ctrl, shift, alt, meta. 'ctrl+shift'." ), }, "selector": { "type": "string", "description": ( "CSS selector of the element to act on, for " "click, type and wait." ), }, "text": { "type": "string", "description": ( "For click: the visible text to find it by, " "instead of a selector. For type: what to type. " "For press: the key, such as Enter or Tab." ), }, "seconds": { "type": "number", "description": "For wait: how long, up to 15.", }, "to": { "type": "string", "description": ( "For scroll: top, bottom, or a selector to bring " "into view." ), }, }, "required": ["do"], }, }, "links": { "type": "boolean", "description": ( "Include links and controls in what comes back. Defaults to " "true." ), }, "follow_popup": { "type": "boolean", "description": ( "If a step opens a new tab, follow it and report that page " "instead. Defaults to true; turn it off to stay where you " "are." ), }, }, "required": ["steps"], } MAX_STEPS = 20 MAX_WAIT = 15 #: Keys worth naming. Anything else is typed as a character, which is what a #: single-character `text` means. KEYS = { "enter": ("Enter", "Enter", 13, "\r"), "tab": ("Tab", "Tab", 9, "\t"), "escape": ("Escape", "Escape", 27, ""), "backspace": ("Backspace", "Backspace", 8, ""), "delete": ("Delete", "Delete", 46, ""), "arrowup": ("ArrowUp", "ArrowUp", 38, ""), "arrowdown": ("ArrowDown", "ArrowDown", 40, ""), "arrowleft": ("ArrowLeft", "ArrowLeft", 37, ""), "arrowright": ("ArrowRight", "ArrowRight", 39, ""), } #: Clicking and typing are done in the page rather than by aiming a mouse at #: coordinates. A coordinate needs a screenshot, a screenshot costs thousands #: of tokens, and the result is wrong the moment the layout moves. A #: selector or the words on the button survive both. CLICK = """ (sel, label, ref) => { const pick = () => { if (ref !== null && ref !== undefined) return (window.__aworg_refs || [])[ref]; if (sel) return document.querySelector(sel); const wanted = (label || "").trim().toLowerCase(); const candidates = [...document.querySelectorAll( "button, a, input[type=submit], input[type=button], [role=button], label")]; return candidates.find((el) => ((el.innerText || el.value || "").trim().toLowerCase() === wanted)) || candidates.find((el) => ((el.innerText || el.value || "").trim().toLowerCase().includes(wanted))); }; const el = pick(); if (!el) return "nothing matched"; el.scrollIntoView({ block: "center" }); // Focus as well as click. A real click does both, and without this a // field that was clicked is not the field a following keystroke reaches. if (el.focus) { try { el.focus({ preventScroll: true }); } catch (e) { el.focus(); } } el.click(); return ""; } """ #: Setting .value directly does not tell a framework anything happened, so #: the native setter is called and the events a real keystroke would raise #: are dispatched. Without this, React and friends keep their own state and #: the field reverts the moment anything re-renders. TYPE = """ (sel, value, ref) => { const el = (ref !== null && ref !== undefined) ? (window.__aworg_refs || [])[ref] : (sel ? document.querySelector(sel) : document.activeElement); if (!el) return "nothing matched"; el.focus(); if (el.isContentEditable) { el.textContent = value; } else { const proto = el instanceof HTMLTextAreaElement ? HTMLTextAreaElement.prototype : HTMLInputElement.prototype; const setter = Object.getOwnPropertyDescriptor(proto, "value"); if (setter && setter.set) setter.set.call(el, value); else el.value = value; } el.dispatchEvent(new Event("input", { bubbles: true })); el.dispatchEvent(new Event("change", { bubbles: true })); return ""; } """ SCROLL = """ (where) => { if (where === "top") { window.scrollTo(0, 0); return ""; } if (where === "bottom") { window.scrollTo(0, document.body.scrollHeight); return ""; } const el = document.querySelector(where); if (!el) return "nothing matched"; el.scrollIntoView({ block: "center" }); return ""; } """ async def run( context: ToolContext, steps: list | None = None, links: bool = True, follow_popup: bool = True, ) -> ToolResult: if not isinstance(steps, list) or not steps: raise ToolError( "page_do needs a list of steps, each with a 'do' of click, type, " "press, scroll or wait." ) if len(steps) > MAX_STEPS: raise ToolError(f"page_do takes at most {MAX_STEPS} steps at a time.") done: list[str] = [] try: page_browser = await browser(start_if_needed=False) before = page_browser.url for index, step in enumerate(steps, start=1): if not isinstance(step, dict): raise ToolError(f"Step {index} is not an object.") await _step(page_browser, index, step, done, context) # A click that opened a tab. The page under the Resident is # unchanged, which is exactly what makes this invisible otherwise: # it reads the old page back and concludes nothing happened. popped = await page_browser.popup() if popped and follow_popup: await page_browser.follow(popped) done.append(f"followed a new tab to {popped}") elif popped: done.append(f"a new tab opened at {popped}, not followed") page = await extract(page_browser) del before except BrowserError as exc: raise ToolError( (f"Did: {'; '.join(done)}. Then it failed: " if done else "") + str(exc) ) from exc return ToolResult( text="Did: " + "; ".join(done) + "\n\n" + describe(page_browser, page, links), payload={"did": done, "page": page}, summary=f"{len(done)} step(s), " + summary(page_browser, page), ) async def _step(page_browser, index, step, done, context) -> None: action = str(step.get("do", "")).strip().lower() selector = step.get("selector") text = step.get("text") ref = _ref(step.get("ref"), index) modifiers = _modifiers(step.get("keys"), index) if context.activity is not None: context.activity.progress = f"step {index}: {action}" if action == "click": if ref is None and not selector and not text: raise ToolError( f"Step {index}: click needs a ref, a selector or some text." ) if modifiers: await _mouse(page_browser, index, ref, selector, "click", modifiers) else: problem = await _call(page_browser, CLICK, selector, text, ref) if problem: raise ToolError( f"Step {index}: nothing on the page matches " f"{step.get('ref') or selector or text!r}. Read the page " "again -- handles go stale when it navigates." ) done.append(f"clicked {step.get('ref') or selector or text!r}") # A click is usually meant to change something, and what it changed # is the point of the call. await asyncio.sleep(0.4) elif action == "type": if text is None: raise ToolError(f"Step {index}: type needs text.") problem = await _call(page_browser, TYPE, selector, text, ref) if problem: raise ToolError( f"Step {index}: nothing on the page matches " f"{step.get('ref') or selector!r}." ) done.append( f"typed into {step.get('ref') or selector or 'the focused field'}" ) elif action == "hover": if ref is None and not selector: raise ToolError(f"Step {index}: hover needs a ref or a selector.") await _mouse(page_browser, index, ref, selector, "hover", modifiers) done.append(f"hovered {step.get('ref') or selector!r}") # Menus that open on hover take a moment to open. await asyncio.sleep(0.3) elif action == "drag": to_ref = _ref(step.get("to_ref"), index) if (ref is None and not selector) or (to_ref is None and not step.get("to")): raise ToolError( f"Step {index}: drag needs something to drag (ref or " "selector) and somewhere to drop it (to_ref or to)." ) await _drag(page_browser, index, ref, selector, to_ref, step.get("to")) done.append( f"dragged {step.get('ref') or selector!r} onto " f"{step.get('to_ref') or step.get('to')!r}" ) await asyncio.sleep(0.3) elif action == "press": key = str(text or "").strip() if not key: raise ToolError(f"Step {index}: press needs a key, such as Enter.") await _press(page_browser, key, modifiers) done.append(f"pressed {key}") await asyncio.sleep(0.4) elif action == "scroll": where = str(step.get("to") or selector or "bottom") problem = await page_browser.evaluate( f"({SCROLL})({json.dumps(where)})" ) if problem: raise ToolError(f"Step {index}: nothing matches {where!r} to scroll to.") done.append(f"scrolled to {where}") elif action == "wait": if selector: await _wait_for(page_browser, selector, step.get("seconds")) done.append(f"waited for {selector}") else: seconds = max(0.0, min(float(step.get("seconds") or 1), MAX_WAIT)) await asyncio.sleep(seconds) done.append(f"waited {seconds}s") else: raise ToolError( f"Step {index}: {action!r} is not something page_do can do. It " "can click, type, press, hover, drag, scroll and wait." ) async def _call(page_browser, function: str, *arguments) -> str: """Run one of the page functions above with JSON-safe arguments.""" packed = ", ".join(json.dumps(argument) for argument in arguments) return await page_browser.evaluate(f"({function})({packed})") or "" #: Modifier bits, as the protocol counts them. MODIFIERS = {"alt": 1, "ctrl": 2, "control": 2, "meta": 4, "cmd": 4, "command": 4, "shift": 8} def _modifiers(keys, index: int) -> int: if not keys: return 0 total = 0 for part in str(keys).replace(",", "+").split("+"): name = part.strip().lower() if not name: continue if name not in MODIFIERS: raise ToolError( f"Step {index}: {part!r} is not a modifier. They are " + ", ".join(sorted(set(MODIFIERS))) + "." ) total |= MODIFIERS[name] return total def _ref(value, index: int): """'ref_12' as the number 12, or None if there was no ref.""" if value in (None, ""): return None text = str(value).strip() if text.startswith("ref_"): text = text[4:] if not text.isdigit(): raise ToolError( f"Step {index}: {value!r} is not a handle. They look like " "'ref_12' and come from reading the page." ) return int(text) #: Bring something into view. Separate from measuring it, because a scroll #: is not necessarily finished when the call that asked for it returns, and #: a position read in the same breath can be the position it was leaving. SCROLL_TO = """ (sel, ref, sel2, ref2) => { const pick = (s, r) => (r === null || r === undefined) ? (s ? document.querySelector(s) : null) : (window.__aworg_refs || [])[r]; const first = pick(sel, ref); if (!first) return "nothing matched"; const second = pick(sel2, ref2); first.scrollIntoView({ block: "center" }); if (second) { const b = second.getBoundingClientRect(); const seen = b.bottom > 0 && b.top < innerHeight && b.right > 0 && b.left < innerWidth; // Only if the other end is off screen: bringing it into view when it is // already visible would move the first one for no reason. if (!seen) second.scrollIntoView({ block: "center" }); } return ""; } """ #: Where things are, measured and nothing else. MEASURE = """ (sel, ref, sel2, ref2) => { const pick = (s, r) => (r === null || r === undefined) ? (s ? document.querySelector(s) : null) : (window.__aworg_refs || [])[r]; const centre = (el) => { if (!el || !el.getBoundingClientRect) return null; const b = el.getBoundingClientRect(); if (!b.width && !b.height) return null; return { x: b.left + b.width / 2, y: b.top + b.height / 2, seen: b.bottom > 0 && b.top < innerHeight && b.right > 0 && b.left < innerWidth }; }; return { from: centre(pick(sel, ref)), to: centre(pick(sel2, ref2)) }; } """ #: How long to let a scroll land before believing a coordinate. SETTLE = 0.12 async def _positions(page_browser, index, ref, selector, to_ref=None, to_selector=None, what="act on"): """Scroll, wait, then measure -- in that order and never fewer steps.""" moved = await _call(page_browser, SCROLL_TO, selector, ref, to_selector, to_ref) if moved == "nothing matched": raise ToolError( f"Step {index}: nothing to {what} -- " f"{step_name(ref, selector)} is not on the page. Read the page " "again for fresh handles." ) await asyncio.sleep(SETTLE) where = await _call_json(page_browser, MEASURE, selector, ref, to_selector, to_ref) if not where or not where.get("from"): raise ToolError( f"Step {index}: nothing to {what} -- " f"{step_name(ref, selector)} is not on the page, or has no size." ) return where async def _where(page_browser, index, ref, selector, what="act on"): return (await _positions(page_browser, index, ref, selector, what=what))["from"] async def _call_json(page_browser, function: str, *arguments): packed = ", ".join(json.dumps(argument) for argument in arguments) return await page_browser.evaluate(f"({function})({packed})") def step_name(ref, selector) -> str: """Whichever way the caller named an element, for saying it back.""" return f"ref_{ref}" if ref is not None else repr(selector) async def _mouse(page_browser, index, ref, selector, kind, modifiers) -> None: """A real mouse event at the element's own position. Coordinates rather than el.click(), because a modifier-click and a hover are mouse state rather than a method call -- and the coordinates come from the element, so nothing here needs a screenshot to aim with. Aimed twice. Moving the pointer changes the page: a menu opens, a banner collapses, and what was being pointed at is now somewhere else. So the pointer moves, the target is measured again, and the press uses that. """ spot = await _where(page_browser, index, ref, selector, "hover" if kind == "hover" else "click") await page_browser.command("Input.dispatchMouseEvent", { "type": "mouseMoved", "x": spot["x"], "y": spot["y"], "modifiers": modifiers, }) if kind == "hover": return spot = await _aim(page_browser, ref, selector, spot) base = {"x": spot["x"], "y": spot["y"], "modifiers": modifiers} await page_browser.command("Input.dispatchMouseEvent", {"type": "mouseMoved", **base}) for kind_of_event in ("mousePressed", "mouseReleased"): await page_browser.command("Input.dispatchMouseEvent", { "type": kind_of_event, "button": "left", "clickCount": 1, **base, }) #: How far something may have moved before it is worth correcting for. Two #: pixels is sub-pixel layout and rounding; ten is a page that moved. DRIFT = 3 async def _aim(page_browser, ref, selector, was): """Where the target is now, if that is not where it was. Falls back to the earlier position rather than failing: an element that has just vanished is a case the click itself will report, in better words than a measurement could. """ now = await _call_json(page_browser, MEASURE, selector, ref, None, None) spot = (now or {}).get("from") if not spot: return was if abs(spot["x"] - was["x"]) < DRIFT and abs(spot["y"] - was["y"]) < DRIFT: return was return spot #: HTML5 drag-and-drop, which synthetic mouse events cannot drive: the #: browser takes over on mousedown and the mouseup never arrives. Dispatched #: as the events the API itself defines, sharing one DataTransfer so that a #: handler reading what was dropped finds what was dragged. NATIVE_DRAG = """ (fromSel, fromRef, toSel, toRef) => { const pick = (sel, ref) => (ref === null || ref === undefined) ? document.querySelector(sel) : (window.__aworg_refs || [])[ref]; const from = pick(fromSel, fromRef); const onto = pick(toSel, toRef); if (!from || !onto) return "nothing matched"; if (!from.draggable) return "not native"; const data = new DataTransfer(); const fire = (el, type) => el.dispatchEvent(new DragEvent(type, { bubbles: true, cancelable: true, dataTransfer: data, })); from.scrollIntoView({ block: "center" }); fire(from, "dragstart"); fire(onto, "dragenter"); fire(onto, "dragover"); fire(onto, "drop"); fire(from, "dragend"); return ""; } """ async def _drag(page_browser, index, ref, selector, to_ref, to_selector) -> None: # The native API first, because an element that declares itself # draggable is telling us which of the two kinds of drag it is. native = await _call(page_browser, NATIVE_DRAG, selector, ref, to_selector, to_ref) if native == "nothing matched": raise ToolError( f"Step {index}: nothing to drag -- " f"{step_name(ref, selector)} or {step_name(to_ref, to_selector)} " "is not on the page." ) if native == "": return where = await _positions(page_browser, index, ref, selector, to_ref, to_selector, "drag") start, end = where["from"], where.get("to") if not end: raise ToolError( f"Step {index}: nothing to drop onto -- " f"{step_name(to_ref, to_selector)} is not on the page." ) if not start.get("seen") or not end.get("seen"): raise ToolError( f"Step {index}: {step_name(ref, selector)} and " f"{step_name(to_ref, to_selector)} cannot both be on screen at " "once, so there is no gesture that goes from one to the other. " "Scroll first, or drag to something nearer." ) # Move first, then look again, then press. Moving the pointer here is # what closes whatever the pointer was on before, and that is usually # what was holding the page in a different shape. await page_browser.command("Input.dispatchMouseEvent", { "type": "mouseMoved", "x": start["x"], "y": start["y"], "modifiers": 0, }) start = await _aim(page_browser, ref, selector, start) await page_browser.command("Input.dispatchMouseEvent", { "type": "mouseMoved", "x": start["x"], "y": start["y"], "modifiers": 0, }) await page_browser.command("Input.dispatchMouseEvent", { "type": "mousePressed", "button": "left", "clickCount": 1, "x": start["x"], "y": start["y"], "modifiers": 0, }) # In steps, because a drag handler that only listens for the drop will # take a single jump, and one that tracks movement will not follow at # all without something to track. for step in range(1, 6): await page_browser.command("Input.dispatchMouseEvent", { "type": "mouseMoved", "button": "left", "modifiers": 0, "x": start["x"] + (end["x"] - start["x"]) * step / 5, "y": start["y"] + (end["y"] - start["y"]) * step / 5, }) await asyncio.sleep(0.03) # And aim again before letting go. The pointer has crossed the page to # get here, and anything that reacts to a pointer has had its say. end = await _aim(page_browser, to_ref, to_selector, end) await page_browser.command("Input.dispatchMouseEvent", { "type": "mouseMoved", "button": "left", "modifiers": 0, "x": end["x"], "y": end["y"], }) await page_browser.command("Input.dispatchMouseEvent", { "type": "mouseReleased", "button": "left", "clickCount": 1, "x": end["x"], "y": end["y"], "modifiers": 0, }) async def _press(page_browser, key: str, modifiers: int = 0) -> None: """A real key event, because forms listen for keys rather than clicks.""" named = KEYS.get(key.strip().lower()) if named: code, dom_key, code_number, char = named elif len(key) == 1: code, dom_key, code_number, char = key, key, ord(key.upper()), key else: raise ToolError( f"{key!r} is not a key page_do knows. It knows " + ", ".join(sorted(KEYS)) + ", and any single character." ) base = {"windowsVirtualKeyCode": code_number, "key": dom_key, "code": code, "modifiers": modifiers} await page_browser.command("Input.dispatchKeyEvent", { "type": "rawKeyDown", **base, }) if char: await page_browser.command("Input.dispatchKeyEvent", { "type": "char", "text": char, **base, }) await page_browser.command("Input.dispatchKeyEvent", {"type": "keyUp", **base}) async def _wait_for(page_browser, selector: str, seconds) -> None: limit = max(0.5, min(float(seconds or 10), MAX_WAIT)) loop = asyncio.get_event_loop() deadline = loop.time() + limit while loop.time() < deadline: if await page_browser.evaluate( f"!!document.querySelector({json.dumps(selector)})" ): return await asyncio.sleep(0.2) raise ToolError( f"{selector!r} did not appear within {limit:g} seconds. " "Use read_page to see what the page is showing instead." )