"""Read the page that is already open, whole or in part.""" from __future__ import annotations from aworg.tools.base import ToolContext, ToolError, ToolResult from ._cdp import BrowserError, browser from ._page import ( TEXT_LIMIT, describe, describe_outline, extract, outline as page_outline, summary, ) NAME = "read_page" DESCRIPTION = ( "Read the page that is currently open, without loading it again. Give a " "CSS selector to read one part of it, or nothing for the whole page." ) INPUT_SCHEMA = { "type": "object", "properties": { "selector": { "type": "string", "description": ( "A CSS selector, to read just that part -- '#results', " "'.error', 'table'. Leave it out for the whole page." ), }, "html": { "type": "boolean", "description": ( "Return the HTML of the selected part rather than its text." ), }, "outline": { "type": "boolean", "description": ( "Return the page's structure -- landmarks, headings, lists, " "controls, nested as they are on the page -- instead of its " "text." ), }, }, } #: Markup is far bulkier than text for the same information, so it is held to #: a tighter limit. A Resident that needs more than this of the raw HTML #: wants a narrower selector, not a bigger budget. HTML_LIMIT = 4000 async def run( context: ToolContext, selector: str | None = None, html: bool = False, outline: bool = False, ) -> ToolResult: try: page_browser = await browser(start_if_needed=False) if outline: rows = await page_outline(page_browser) return ToolResult( text=f"The shape of {page_browser.url}:\n\n" + describe_outline(rows), payload=rows, summary=f"{len(rows)} element(s)", ) if selector: # Passed as an argument rather than spliced into the expression, # so that a selector containing a quote is a selector rather # than a way to write JavaScript. # A form field's contents are a property rather than text or # markup, so a field read as either comes back empty -- which is # exactly the case a Resident is checking when it reads one. found = await page_browser.evaluate( "(() => { const el = document.querySelector(" + _as_js_string(selector) + "); if (!el) return null; const value = (el.value ?? null);" " return { html: el.outerHTML," " text: (el.innerText || el.textContent || ''), value }; })()" ) if not found: raise ToolError( f"Nothing on this page matches {selector!r}. " "Use read_page without a selector to see what is there." ) body = found["html"] if html else found["text"] if found.get("value") is not None: body = f"value: {found['value']}" + ("\n" + body if body else "") limit = HTML_LIMIT if html else TEXT_LIMIT clipped = len(body) > limit text = body[:limit].rstrip() + ( f"\n... [first {limit} characters]" if clipped else "" ) return ToolResult( text=f"{selector} on {page_browser.url}\n\n{text}", payload=found, summary=f"{selector}, {len(body)} chars", ) page = await extract(page_browser) except BrowserError as exc: raise ToolError(str(exc)) from exc return ToolResult( text=describe(page_browser, page), payload=page, summary=summary(page_browser, page), ) def _as_js_string(value: str) -> str: """A Python string as a JavaScript literal, safely. JSON's string syntax is a subset of JavaScript's, which makes json.dumps the correct escaper here rather than quoting by hand. """ import json return json.dumps(value)