diff --git a/.env.example b/.env.example index cfee571..355a9b1 100644 --- a/.env.example +++ b/.env.example @@ -3,13 +3,18 @@ JEV_API_KEY= JEV_MODEL=jev-latest # JEV_URL=https://api.typesafe.ai/v1/systemone +# How many characters of page text one decision reads (retrieved, not truncated). +# JEV_EVIDENCE_CHARS=20000 + # A small text model writes strings, only when the operation is TYPE_TEXT. +# Optional: goals that never type into a field do not need it. TEXT_MODEL_API_KEY= TEXT_MODEL_BASE_URL=https://api.openai.com/v1 TEXT_MODEL=gpt-4o-mini TEXT_MODEL_REASONING=none -# Moli, via Lexmount. +# Moli, via Lexmount. Only for lexmount_session() / the CLI without --cdp; +# install with: uv sync --extra lexmount LEXMOUNT_API_KEY= LEXMOUNT_PROJECT_ID= LEXMOUNT_BASE_URL=https://api.lexmount.com diff --git a/README.md b/README.md index a352c9b..4c678dc 100644 --- a/README.md +++ b/README.md @@ -6,6 +6,8 @@ Give it one goal. [TypeSafe's Jev](https://docs.typesafe.ai/introduction) picks Built for [Moli](https://browser.lexmount.com), which keeps page structure and interaction state in memory and renders only when a picture is actually needed. Geometry there is a snapshot from the last render, so this agent reads **structure** instead: semantics for what is live, `textContent` for what it says, and element dispatch for what it does. +Because it never asks about layout, it runs unchanged on **any browser that speaks CDP** — including the ones that never lay a page out at all. + ## Why no layout Same page, same selectors. The only difference is what the extractor asks: @@ -19,6 +21,44 @@ Same page, same selectors. The only difference is what the extractor asks: Nothing in `snapshot.js` calls `getBoundingClientRect`, `checkVisibility`, `elementFromPoint` or `innerText`. Actions are dispatched on the element, never at a coordinate, so a stale layout cannot misdirect a click. +## Every browser + +The same `snapshot.js`, next to a reader that filters by position, on six browsers. Controls / characters of page text: + +| Browser | Page | Structure | Geometry | +| --- | --- | --- | --- | +| Moli | Google Flights | 155 / 34,189 | **5 / 7** | +| Chrome | Google Flights | 145 / 25,535 | 17 / 69 | +| chrome-headless-shell | Google Flights | 145 / 25,535 | 20 / 217 | +| Cloudflare Kitesurf | Google Flights | 209 / 36,569 | **error** — no `checkVisibility` | +| Lightpanda | Wikipedia: Jupiter | 2,875 / 137,723 | **18 / 221** — no layout at all | +| Obscura (no-render build) | Wikipedia: Jupiter | 2,876 / 137,745 | **250 / 6,000** — placeholder boxes, so everything counts as on screen | + +A reader that asks where things are fails differently on every engine without a real layout: too little, too much, or an exception. Asking what things are gives the same answer everywhere. + +Every Chromium, local or hosted, reads the same page identically: 2,876 controls and the same text on the Jupiter article, every time. + +### End to end + +Five goals — switch a page's language, follow a footer link, jump to another reference page, open a linked article, type a search and open the result — each run twice, success judged by the URL the agent ends on: + +| Browser | How | Result | Steps | +| --- | --- | --- | --- | +| Moli (Lexmount) | `lexmount_session()` | 10/10 | 2.2 | +| Chrome image (Lexmount) | `lexmount_session("normal")` | 10/10 | 2.3 | +| Cloudflare Kitesurf | `connect(url, headers=…)` | 10/10 | 2.2 | +| Cloudflare Browser Run, Chromium | `connect(url, headers=…)` | 7/7 ¹ | 2.1 | +| Browserbase | `connect(connectUrl)` | 10/10 | 2.2 | +| Chrome, chrome-headless-shell, Playwright Chromium | `connect("http://127.0.0.1:9222")` | 10/10 each | 2.2 | +| Lightpanda | `connect("http://127.0.0.1:9222")` | 9/10 ² | 2.2 | +| Obscura, no-render build | `connect(...)` | 9/10 ³ | 2.3 | +| browserless, Steel, chromedp, Kernel (self-hosted) | `connect("http://127.0.0.1:")` | 10/10 each | 2.2 | +| Selenium Grid | `selenium_session("http://127.0.0.1:4444")` | 5/5 | 2.2 | + +¹ Every run that got a browser; the rest were refused by the free plan's daily quota. ² The miss reached the article and kept clicking. ³ The miss was Obscura's own 30-second navigation deadline. + +Two limits of the browsers themselves, not of this layer: Lightpanda does not run enough of Google Flights' JavaScript to render the page, and Kitesurf's public playground meters CPU per page tightly enough that a very large page (the full Jupiter article) runs it out — through an authenticated Cloudflare account it reads the same page in full. + ## The action space Every observation produces a fresh element table: @@ -62,14 +102,21 @@ Done · 2 of 3 Done · 3 of 3 ``` +Links that share a name *and* a destination are one control repeated, and are offered once. + +### The page text is retrieved, not truncated + +A structure-first snapshot hands over the whole document — 50,000 to 150,000 characters on a long article. The first few thousand are navigation. So the page is split into rows (a table row stays one row, `Elevation | 8,848.86 m`) and the rows that bear on the goal are kept, in page order, up to `JEV_EVIDENCE_CHARS` (default 20,000). On four long articles, a 6,000-character prefix missed the answer every time; retrieval at 20,000 kept it every time. + ## Try it ```bash git clone https://github.com/lexmount/jev-nolayout.git cd jev-nolayout -uv sync +uv sync --extra lexmount cp .env.example .env -# Add JEV_API_KEY, TEXT_MODEL_API_KEY and your Lexmount credentials. +# Add JEV_API_KEY and your Lexmount credentials. +# TEXT_MODEL_API_KEY is only needed for goals that type into a field. uv run jev-nolayout https://en.wikipedia.org/wiki/Espresso "Open the article about Latte" ``` @@ -89,24 +136,46 @@ uv run jev-nolayout https://en.wikipedia.org/wiki/Espresso "Open the article abo https://en.wikipedia.org/wiki/Latte ``` -Pass `--browser normal` to run the same agent against standard Chrome. It works there too — reading structure is not a workaround, it is simply a better question. +`--browser normal` runs the same agent on Lexmount's standard Chrome. `--cdp` runs it on any other browser — a `ws://` URL, or the `http://host:port` a local browser serves: + +```bash +lightpanda serve --port 9222 & +uv run jev-nolayout --cdp http://127.0.0.1:9222 https://en.wikipedia.org/wiki/Espresso "Switch to the Deutsch edition" +``` ## In code ```python -from jev_nolayout import Agent, moli_session +from jev_nolayout import Agent, lexmount_session -with moli_session() as browser: +with lexmount_session() as browser: # Moli browser.navigate("https://docs.python.org/3/") for state in Agent(browser, "Go to the Standard Library reference").run(): print(state.steps[-1]) ``` +Any other browser is one line different: + +```python +from jev_nolayout import connect + +with connect("http://127.0.0.1:9222") as browser: + ... +``` + +`connect()` takes a `ws://` / `wss://` URL or an `http(s)://` address serving `/json/version`, uses the browser's page or opens one, and closes what it opened. It also handles what hosted browsers tend to need: + +- `headers=` for services that authenticate the websocket handshake (Cloudflare: `{"Authorization": "Bearer …"}`); `user:pass@` in the URL is sent as Basic auth. +- A websocket address advertised from inside a container (`ws://0.0.0.0:3000`, a container IP) is pointed back at the address you connected to. +- A 429 on connect is waited out when it asks for seconds, and reported plainly when it asks for hours. + +`selenium_session(grid)` does the same for a Selenium Grid, which hands out CDP only per session. + ## What it handles Multi-step navigation, autocomplete fields, calendar widgets built from unlabelled `
`s, and controls that share a name — all without a single layout query. -`examples/flights.py` drives a live Google Flights search and checks the result against the page itself rather than against the model's claim of success. +`examples/quickstart.py` is the shortest complete run on Moli. `examples/flights.py` drives a live Google Flights search and checks the result against the page itself rather than against the model's claim of success. ## License diff --git a/examples/quickstart.py b/examples/quickstart.py new file mode 100644 index 0000000..8cd0148 --- /dev/null +++ b/examples/quickstart.py @@ -0,0 +1,22 @@ +"""The shortest complete run: one goal on Moli, printed step by step. + + uv sync --extra lexmount + cp .env.example .env # JEV_API_KEY and Lexmount credentials + uv run python examples/quickstart.py + +For any other browser, replace `lexmount_session()` with +`connect("http://127.0.0.1:9222")` (or a ws:// URL) -- nothing else changes. +""" +from jev_nolayout import Agent, lexmount_session +from jev_nolayout.cli import load_env + +load_env() + +with lexmount_session() as browser: + browser.navigate("https://en.wikipedia.org/wiki/Espresso") + state = None + for state in Agent(browser, "Switch this page to the Deutsch language edition").run(): + step = state.steps[-1] if state.steps else None + if step: + print(f"{step.n:>2}. {step.operation:<9} {step.label[:50]}") + print(f"\n{state.status} in {len(state.steps)} steps -> {state.url}") diff --git a/jev_nolayout/__init__.py b/jev_nolayout/__init__.py index 242db4d..564236c 100644 --- a/jev_nolayout/__init__.py +++ b/jev_nolayout/__init__.py @@ -1,7 +1,7 @@ """A browser agent that never asks the layout engine anything.""" from .agent import Agent, Run, Step from .browser import Action, Browser, PageChanged, Snapshot -from .session import moli_session +from .session import connect, lexmount_session, moli_session, selenium_session -__all__ = ["Agent", "Run", "Step", "Action", "Browser", "PageChanged", - "Snapshot", "moli_session"] +__all__ = ["Agent", "Run", "Step", "Action", "Browser", "PageChanged", "Snapshot", + "connect", "lexmount_session", "moli_session", "selenium_session"] diff --git a/jev_nolayout/agent.py b/jev_nolayout/agent.py index a738d9a..ad8e411 100644 --- a/jev_nolayout/agent.py +++ b/jev_nolayout/agent.py @@ -46,13 +46,19 @@ def run(self): started = time.perf_counter() blank_reads = 0 waits = 0 + # The page as read at the end of the previous step, if nothing has + # happened since. Each step used to end with a read (to see what the + # action changed) and the next begin with another of the same page -- + # twice the work for one decision. On Kitesurf, which meters CPU per + # page, that second read is what ran the budget out. + carried = None # Controls that have been tried and changed nothing, by node id. spent: dict[int, int] = {} for n in range(1, MAX_STEPS + 1): step_started = time.perf_counter() try: - snapshot = self.browser.observe() + snapshot, carried = carried or self.browser.observe(), None except PageChanged: blank_reads += 1 if blank_reads >= 3: @@ -152,6 +158,7 @@ def run(self): yield self.run_state continue + carried = after changed = after.marker != before # Report the consequence, not just that something moved. diff --git a/jev_nolayout/browser.py b/jev_nolayout/browser.py index 1913867..fed0d38 100644 --- a/jev_nolayout/browser.py +++ b/jev_nolayout/browser.py @@ -12,6 +12,8 @@ import contextlib import json +import time +import uuid from dataclasses import dataclass, field from pathlib import Path @@ -32,6 +34,10 @@ })""" ACT_JS = """(action => { + // Mark this document. If the action replaces it, the next document will not + // carry the mark -- which is how the caller tells "the page changed" from + // "the page is still the one we clicked on and has not caught up yet". + window.__jevNoLayoutDoc = action.token; const e = window.__jevNoLayout?.nodes.get(action.node); if (!e?.isConnected || e.matches(':disabled') || e.closest('[aria-disabled="true"],[inert],[hidden]')) return null; @@ -59,14 +65,73 @@ else e.value = action.text || ''; e.dispatchEvent(new Event('input', {bubbles: true})); e.dispatchEvent(new Event('change', {bubbles: true})); + // How to find this field again if the page swaps it out; see REFILL_JS. + window.__jevNoLayoutField = {node: action.node, name: e.name || '', + aria: e.getAttribute('aria-label') || '', placeholder: e.placeholder || '', + form: e.form?.id || ''}; return 'filled'; } e.scrollIntoView({block: 'center'}); + // A link that leaves this document. Said up front, because `click()` returns + // before the browser has even started loading the next page -- measured on + // Lightpanda, the agent read the old page, saw no change, and clicked the + // same link again, every time. + const link = e.closest('a[href]'); + const href = link?.getAttribute('href') || ''; + const leaves = link && !href.startsWith('#') && !href.startsWith('javascript:') + && (link.target || '_self') === '_self' && !link.hasAttribute('download'); + // A form's submit button leaves the document too, as far as anyone can tell + // in advance -- and not waiting for it cost the same. Measured on + // Lightpanda: the search form did submit and did navigate, but the next + // observation read the page being left, and the agent spent some twenty + // steps waiting and re-clicking before it saw the result. + const form = e.form || null; + const submits = form && (e.type === 'submit' || e.type === 'image' + || (e.tagName === 'BUTTON' && !e.hasAttribute('type'))); e.click(); + if (leaves) return {navigating: link.href}; + if (submits) return {navigating: form.action || location.href}; return 'clicked'; })""" +# After typing: is the text still in the field the page will actually use? +# +# Some pages replace a field the moment it is first typed into. Wikipedia's +# search box is a plain until then, and a framework component after: +# the text went into the node being thrown away, the new one was empty, and +# the form submitted an empty search -- measured on Kitesurf, every time. So +# if the node we filled is gone, find its replacement by name, label or +# placeholder, and put the text there too. +REFILL_JS = """(text => { + const was = window.__jevNoLayoutField; + if (!was) return 'kept'; + const old = window.__jevNoLayout?.nodes.get(was.node); + if (old?.isConnected && old.value === text) return 'kept'; + const fields = [...document.querySelectorAll('input,textarea')].filter(f => + !f.disabled && !f.readOnly && f.type !== 'hidden' + && ((was.name && f.name === was.name) + || (was.aria && f.getAttribute('aria-label') === was.aria) + || (was.placeholder && f.placeholder === was.placeholder))); + if (!fields.length) return 'lost'; + const next = fields.find(f => (f.form?.id || '') === was.form) || fields[0]; + if (next.value === text) return 'kept'; + next.focus(); + const proto = next.tagName === 'TEXTAREA' + ? HTMLTextAreaElement.prototype : HTMLInputElement.prototype; + const setter = Object.getOwnPropertyDescriptor(proto, 'value'); + if (setter?.set) setter.set.call(next, text); else next.value = text; + next.dispatchEvent(new Event('input', {bubbles: true})); + next.dispatchEvent(new Event('change', {bubbles: true})); + return 'refilled'; +})""" + +# Has the document been replaced since ACT_JS marked it, and is the new one done? +LOADED_JS = """(token => ({ + replaced: window.__jevNoLayoutDoc !== token, + ready: document.readyState === 'complete' || document.readyState === 'interactive', +}))""" + SUBMIT_JS = """(node => { const e = window.__jevNoLayout?.nodes.get(node); if (!e?.isConnected) return null; @@ -97,6 +162,7 @@ class Action: depth: int = 0 closed: bool = False # sits inside a dialog that is not open current_value: str = "" + href: str = "" # where a link goes; empty for everything else extra: dict = field(default_factory=dict) @property @@ -114,10 +180,20 @@ def describe(self) -> str: class Snapshot: url: str title: str - text: str + rows: list[str] actions: list[Action] marker: str + def evidence(self, goal: str, budget: int | None = None) -> str: + """The page text that bears on `goal`, within a character budget.""" + from .evidence import BUDGET, select + return select(self.rows, goal, BUDGET if budget is None else budget) + + @property + def text(self) -> str: + """Every row, in order. Use `evidence()` for anything sent to a model.""" + return " ".join(self.rows) + def find(self, *needles: str, kind: str | None = None) -> Action | None: """First action whose label contains every needle.""" for action in self.actions: @@ -131,9 +207,10 @@ def find(self, *needles: str, kind: str | None = None) -> Action | None: class Browser: """A Moli page, observed structurally and driven by element dispatch.""" - def __init__(self, cdp, session_id: str): + def __init__(self, cdp, session_id: str, target: str | None = None): self._cdp = cdp self.session = session_id + self.target = target # -- plumbing --------------------------------------------------------- @@ -141,8 +218,14 @@ def call(self, method: str, **params): return self._cdp(method, session_id=self.session, **params) def evaluate(self, expression: str, await_promise: bool = False): - result = self.call("Runtime.evaluate", expression=expression, - returnByValue=True, awaitPromise=await_promise) + try: + result = self.call("Runtime.evaluate", expression=expression, + returnByValue=True, awaitPromise=await_promise) + except RuntimeError as error: + # Asked mid-navigation, a browser answers with a protocol error + # ("context was destroyed", "cannot find context") rather than a + # JavaScript exception. Same meaning, so the same signal. + raise PageChanged(f"Document changed during evaluation: {error}") from None if result.get("exceptionDetails"): raise PageChanged("Document changed during evaluation") return (result.get("result") or {}).get("value") @@ -159,27 +242,115 @@ def observe(self) -> Snapshot: node=a["node"], kind=a["kind"], role=a["role"], label=a["label"], value=a.get("value", ""), region=a.get("region", ""), depth=a.get("depth", 0), closed=bool(a.get("closed")), - current_value=a.get("current_value", ""), + current_value=a.get("current_value", ""), href=a.get("href", ""), extra={k: a[k] for k in ("checked", "selected", "expanded") if k in a})) - return Snapshot(url=raw["url"], title=raw["title"], text=raw["text"], + return Snapshot(url=raw["url"], title=raw["title"], rows=raw.get("rows") or [], actions=actions, marker=raw["marker"]) def act(self, action: Action, text: str = "") -> str: + token = uuid.uuid4().hex payload = {"node": action.node, "kind": action.kind, - "value": action.value, "text": text} + "value": action.value, "text": text, "token": token} outcome = self.evaluate(ACT_JS + "(" + json.dumps(payload) + ")") if outcome is None: raise PageChanged("Target is gone or not actionable. Observe again.") + if outcome == "filled": + self.settle() + with contextlib.suppress(PageChanged): + if self.evaluate(REFILL_JS + "(" + json.dumps(text) + ")") == "refilled": + outcome = "filled (the page replaced the field; typed again)" + if isinstance(outcome, dict): + href = outcome.get("navigating", "") + if self._await_new_document(token) or self._follow_new_page(href): + outcome = "clicked" + else: + outcome = "clicked (stayed)" self.settle() return outcome + def _follow_new_page(self, href: str) -> bool: + """Move to the page a link opened elsewhere, if it opened one. + + Most browsers load a followed link into the same page. Cloudflare's + Kitesurf does not: measured, `location.href` changed at once but the + attached document never did, and the new document turned up on a new, + unattached page target -- so an agent clicked the right link and never + arrived. Look for that page and re-attach to it. + """ + if not href: + return False + try: + targets = self._cdp("Target.getTargets").get("targetInfos") or [] + except RuntimeError: + return False + fresh = [t for t in targets if t.get("type") == "page" + and t.get("targetId") != self.target and not t.get("attached") + and t.get("url", "").split("#")[0] == href.split("#")[0]] + if not fresh: + return False + page = fresh[-1] + try: + attached = self._cdp("Target.attachToTarget", targetId=page["targetId"], + flatten=True) + except RuntimeError: + return False + # Let go of the page being left. Keeping its session attached leaks a + # handle per followed link, and a browser that caps sessions would + # eventually refuse the next attach. + with contextlib.suppress(RuntimeError): + self._cdp("Target.detachFromTarget", sessionId=self.session) + self.session, self.target = attached["sessionId"], page["targetId"] + deadline = time.monotonic() + 20 + while time.monotonic() < deadline: + with contextlib.suppress(PageChanged): + if self.evaluate("document.readyState") in ("interactive", "complete"): + break + time.sleep(0.15) + return True + + def _await_new_document(self, token: str, start: float = 5.0, + finish: float = 20.0) -> bool: + """Wait for a link click to replace the document; False if it never did. + + Two stages, because they fail differently. A navigation that is going + to happen starts within a few seconds; one that has not started by then + was intercepted -- a single-page app handling the click itself -- and + waiting longer only wastes the step. Once it has started, a slow page + is still worth waiting for. + """ + deadline = time.monotonic() + start + while time.monotonic() < deadline: + try: + state = self.evaluate(LOADED_JS + "(" + json.dumps(token) + ")") + except PageChanged: + state = None # mid-swap: no document to ask + if state and state.get("replaced"): + deadline = time.monotonic() + finish + while not state.get("ready") and time.monotonic() < deadline: + time.sleep(0.15) + with contextlib.suppress(PageChanged): + state = self.evaluate( + LOADED_JS + "(" + json.dumps(token) + ")") or state + return True + time.sleep(0.15) + return False + def submit(self, action: Action) -> str | None: outcome = self.evaluate(SUBMIT_JS + "(" + json.dumps(action.node) + ")") self.settle() return outcome - def navigate(self, url: str) -> None: + def navigate(self, url: str, wait: float = 20.0) -> None: + # `Page.navigate` returns once the navigation commits, well before + # there is anything to read. Measured on a form page, both extractors + # saw zero controls for three seconds after it returned. self.call("Page.navigate", url=url) + deadline = time.monotonic() + wait + while time.monotonic() < deadline: + with contextlib.suppress(PageChanged): + if self.evaluate("document.readyState") == "complete": + break + time.sleep(0.15) self.settle() def settle(self) -> None: diff --git a/jev_nolayout/cli.py b/jev_nolayout/cli.py index 95e906e..2b7c2f7 100644 --- a/jev_nolayout/cli.py +++ b/jev_nolayout/cli.py @@ -1,4 +1,4 @@ -"""jev-nolayout -- run one goal on Moli and print the trace.""" +"""jev-nolayout -- run one goal and print the trace.""" from __future__ import annotations import argparse @@ -24,17 +24,22 @@ def main() -> None: parser.add_argument("url") parser.add_argument("goal") parser.add_argument("--browser", default="light", - help="light = Moli (default), normal = standard Chrome") + help="Lexmount session type: light = Moli (default), " + "normal = standard Chrome") + parser.add_argument("--cdp", metavar="ENDPOINT", + help="use any CDP browser instead of creating a Lexmount " + "session: ws://... or http://host:port") parser.add_argument("--env", default=".env") args = parser.parse_args() load_env(args.env) - for required in ("JEV_API_KEY", "TEXT_MODEL_API_KEY"): - if not os.environ.get(required): - parser.error(f"{required} is not set (see .env.example)") + # The text model is only needed when a step types into a field, so it is + # checked there, not here. + if not os.environ.get("JEV_API_KEY"): + parser.error("JEV_API_KEY is not set (see .env.example)") from .agent import Agent - from .session import moli_session + from .session import connect, lexmount_session def show(step): head = f" {step.n:>2}. {step.operation:<10}" @@ -49,10 +54,11 @@ def show(step): if step.outcome: print(f" {step.outcome}") - print(f"\n goal {args.goal}\n from {args.url}\n" - f" browser {'Moli' if args.browser == 'light' else 'Chrome'}\n") + where = args.cdp or ("Moli" if args.browser == "light" else "Lexmount Chrome") + print(f"\n goal {args.goal}\n from {args.url}\n browser {where}\n") - with moli_session(args.browser) as browser: + session = connect(args.cdp) if args.cdp else lexmount_session(args.browser) + with session as browser: browser.navigate(args.url) agent = Agent(browser, args.goal, on_step=show) state = None diff --git a/jev_nolayout/evidence.py b/jev_nolayout/evidence.py new file mode 100644 index 0000000..13ce130 --- /dev/null +++ b/jev_nolayout/evidence.py @@ -0,0 +1,107 @@ +"""Choose which of a page's rows a decision gets to read. + +A structure-first snapshot hands over the whole document: 1,000 to 2,000 rows +and 50,000 to 150,000 characters on a long article. No decision should read +all of that, and the first few thousand characters are the wrong few thousand +-- they are navigation, and the answer is further down. So the rows are ranked +against the goal and the best ones are kept, in page order, up to a budget. + +This is deliberately plain: rare-word overlap, a bonus for numbers, and two +separate pools so that table rows and prose cannot starve each other. Every +part of it is there because a simpler version measurably lost an answer. +""" +from __future__ import annotations + +import math +import os +import re + +# How many characters of page text one decision sees. Measured on four long +# articles: at 6,000 the answer was dropped every time, at 12,000 three times +# in four came back, at 20,000 all four. +BUDGET = int(os.environ.get("JEV_EVIDENCE_CHARS", "20000")) + +# Besides grammar, the words a goal uses to say what KIND of thing it wants +# ("open the article about X linked from this page") rather than what it is +# about. Left in, they match on the wrong thing: "article" pulled a citation +# ending "... Article 5" above the actual article link, and the model took it. +_STOP_WORDS = ( + "a an the of in on at to for is are was were be by with from and or as it its this " + "that what which how many much do does did can could should would will your you my me " + "open find go click type select search report page article link linked about here" +) +_STOP = frozenset(_STOP_WORDS.split()) + + +def keywords(goal: str) -> list[str]: + return [w for w in re.findall(r"[a-z0-9]+", goal.lower()) + if w not in _STOP and len(w) > 2] + + +def _tidy(row: str) -> str: + # Re-insert the space a node boundary sometimes swallows between a word and + # a number ("to9 bars"). Letters and closing brackets only: including , or + # . once split 8,848.86 into "8, 848. 86". + row = re.sub(r"([A-Za-z)])(\d)", r"\1 \2", row) + row = re.sub(r"(\d)([A-Za-z(])", r"\1 \2", row) + return re.sub(r"\s{2,}", " ", row).strip() + + +def select(rows: list[str], goal: str, budget: int = BUDGET) -> str: + """The rows that bear on `goal`, in page order, within `budget` characters.""" + if not rows: + return "" + tidy = [_tidy(r) for r in rows] + words = keywords(goal) + if not words: + return " ".join(tidy)[:budget] + + low = [r.lower() for r in tidy] + n = len(low) + # Rare words carry the signal. Plain hit-counting let "espresso" -- on + # nearly every row of its own article -- decide the ranking on its own. + idf = {w: math.log(n / max(1, sum(1 for t in low if w in t))) + 0.1 for w in words} + + # Table rows and prose compete in separate pools. In one pool, whichever + # heuristic was winning starved the other: boosting table rows fixed an + # infobox answer and immediately lost a sentence answer on another page. + table, prose = [], [] + for i, t in enumerate(low): + score = sum(idf[w] for w in words if w in t) + if score <= 0: + continue + if re.search(r"\d", t): + score *= 1.4 # facts are usually numbers + if " | " in rows[i]: + table.append((score * 1.5, i)) + else: + prose.append((score * (1.2 if len(t) > 60 else 1.0), i)) + table.sort(reverse=True) + prose.sort(reverse=True) + + keep: set[int] = set() + used = 0 + + def take(ranked: list[tuple[float, int]], quota: float) -> None: + nonlocal used + spent = 0 + for _, i in ranked: + if spent > quota: + return + # A hit rarely stands alone: keep the row before it and a few + # after, which is where the value of a label usually is. + for j in range(max(0, i - 1), min(n, i + 3)): + if j not in keep: + keep.add(j) + spent += len(tidy[j]) + 1 + used += len(tidy[j]) + 1 + + take(table, budget * 0.35) + take(prose, budget * 0.60) + for i in range(n): # spend any slack on early prose + if used > budget: + break + if i not in keep and len(tidy[i]) > 80: + keep.add(i) + used += len(tidy[i]) + 1 + return " ".join(tidy[i] for i in sorted(keep))[:budget] diff --git a/jev_nolayout/model.py b/jev_nolayout/model.py index 60e8aca..09452fd 100644 --- a/jev_nolayout/model.py +++ b/jev_nolayout/model.py @@ -45,23 +45,43 @@ } +# Answers worth asking again. A decision request carries no side effects, so +# repeating one is always safe; giving up on the first dropped connection is +# what cost a run on a perfectly healthy browser -- measured, one run in ten on +# Moli and one in eleven on Obscura ended on "Model connection failed" or a +# gateway 500, and the same task passed on the next try. +RETRY_STATUS = {429, 500, 502, 503, 504, 529} +ATTEMPTS = 4 + + def post(url: str, key: str, body: dict) -> dict: - for attempt in range(3): + last = "" + for attempt in range(ATTEMPTS): + final = attempt == ATTEMPTS - 1 try: response = CLIENT.post(url, json=body, headers={"Authorization": f"Bearer {key}"}) - except httpx.HTTPError: - raise RuntimeError("Model connection failed; no action executed.") from None - if response.status_code in {429, 503, 529} and attempt < 2: + except httpx.HTTPError as error: + last = f"{type(error).__name__}: {error}" + if final: + raise RuntimeError(f"Model connection failed; no action executed ({last}).") \ + from None + time.sleep(0.5 * 2 ** attempt) + continue + if response.status_code in RETRY_STATUS and not final: + last = f"HTTP {response.status_code}" time.sleep(0.5 * 2 ** attempt) continue if response.is_error: # Carry the provider's own message: a 400 here is almost always a # malformed question, and the body says which one. + tried = f" after {attempt + 1} attempts" if attempt else "" raise RuntimeError( - f"Model provider returned HTTP {response.status_code}: " + f"Model provider returned HTTP {response.status_code}{tried}: " f"{response.text[:400]}") return response.json() - raise RuntimeError("Model unavailable") + # Every pass through the loop returns or raises; this keeps the function's + # contract explicit for readers and type checkers alike. + raise AssertionError("unreachable") # The decision API accepts at most 255 choices per question. @@ -113,6 +133,18 @@ def _rank(action, keywords: set[str]) -> tuple[int, int]: def action_space(actions, goal: str = ""): """Group observed actions into the operation/target shape Jev expects.""" keywords = _keywords(goal) + # One destination, one choice. A page links the same article from several + # places under the same name; offering each copy spends the choice budget + # on duplicates and leaves the model to pick between identical options. + seen: set[tuple[str, str, str]] = set() + unique = [] + for action in actions: + key = (action.kind, action.label, action.href) + if action.href and key in seen: + continue + seen.add(key) + unique.append(action) + actions = unique if len(actions) > MAX_CHOICES: ordered = sorted(enumerate(actions), key=lambda p: (_rank(p[1], keywords), p[0])) keep = {i for i, _ in ordered[:MAX_CHOICES]} @@ -156,7 +188,10 @@ def choose(snapshot, goal: str, history: list[dict]) -> dict: body = { "model": os.environ.get("JEV_MODEL", "jev-latest"), "state": { - "page": {"url": snapshot.url, "title": snapshot.title, "text": snapshot.text}, + # The rows that bear on the goal, not the first N characters: the + # top of a long page is navigation, and the answer is further down. + "page": {"url": snapshot.url, "title": snapshot.title, + "text": snapshot.evidence(goal)}, "elements": [ {"index": i, "role": a.role, "label": a.label, "value": a.current_value or a.value, **a.extra} @@ -216,7 +251,11 @@ def field_text(goal: str, action, history: list[dict]) -> str: if reasoning and reasoning != "none": body["reasoning"] = {"effort": reasoning} - result = post(f"{base}/chat/completions", os.environ["TEXT_MODEL_API_KEY"], body) + key = os.environ.get("TEXT_MODEL_API_KEY") + if not key: + raise RuntimeError("This step types into a field, which needs a text model: " + "set TEXT_MODEL_API_KEY (see .env.example).") + result = post(f"{base}/chat/completions", key, body) choice = result["choices"][0] content = (choice["message"].get("content") or "").strip() @@ -228,7 +267,7 @@ def field_text(goal: str, action, history: list[dict]) -> str: body["messages"] = body["messages"] + [ {"role": "system", "content": "Reply with the field value alone. No JSON, no quotes, no commentary."}] - content = ((post(f"{base}/chat/completions", os.environ["TEXT_MODEL_API_KEY"], body) + content = ((post(f"{base}/chat/completions", key, body) )["choices"][0]["message"].get("content") or "").strip() if not content: return "" diff --git a/jev_nolayout/session.py b/jev_nolayout/session.py index 1d8e51a..36c18c8 100644 --- a/jev_nolayout/session.py +++ b/jev_nolayout/session.py @@ -1,4 +1,13 @@ -"""Open a Moli session and hand back a Browser bound to its first page.""" +"""Get a Browser bound to a page, on any browser that speaks CDP. + +`connect()` is the whole contract: a CDP endpoint in, a Browser out. Nothing +below it cares which browser answered -- the snapshot reads the DOM, actions +are dispatched on elements, and neither asks the layout engine anything, so a +browser that never lays a page out works as well as one that always does. + +`lexmount_session()` is a convenience on top: it creates a Lexmount session +(Moli by default) and hands its endpoint to `connect()`. +""" from __future__ import annotations import contextlib @@ -6,9 +15,12 @@ import os import subprocess import sys +import time +import urllib.request from contextlib import contextmanager -# The CDP transport is a small script that owns one daemon per browser endpoint. +# Optional: route CDP through an external script instead of a websocket held +# in this process. Unset, the built-in websocket client is used. CDP_SCRIPT = os.environ.get("JEV_CDP_SCRIPT", "") @@ -27,13 +39,75 @@ def cdp(method: str, session_id: str | None = None, **params): return cdp -def _cdp_via_websocket(ws: str): +def _auth_from_url(ws: str, headers: dict[str, str]) -> tuple[str, dict[str, str]]: + """Move `user:pass@` out of the URL and into a Basic Authorization header. + + Some services (Bright Data's scraping browser, for one) put the account in + the websocket URL's userinfo. The websocket client does not turn that into + a header on its own, so the handshake went out unauthenticated. + """ + import base64 + from urllib.parse import unquote, urlsplit, urlunsplit + + parts = urlsplit(ws) + if not parts.username or "authorization" in {k.lower() for k in headers}: + return ws, headers + token = f"{unquote(parts.username)}:{unquote(parts.password or '')}" + host = parts.hostname + (f":{parts.port}" if parts.port else "") + return (urlunsplit(parts._replace(netloc=host)), + {**headers, "Authorization": "Basic " + base64.b64encode(token.encode()).decode()}) + + +MAX_RATE_WAIT = 60.0 + + +def _dial(wsclient, ws: str, headers: dict[str, str], attempts: int = 5): + """Open the websocket, waiting out a rate limit instead of failing on it. + + A cloud browser service that caps how many browsers may start per minute + answers the handshake with 429. That is "not yet", not "no": measured on + Cloudflare's free plan, six runs in ten were refused this way when tasks + started back to back, and every one would have been accepted a few seconds + later. Honour Retry-After when the service sends one. + """ + from websockets.exceptions import InvalidStatus + + for attempt in range(attempts): + try: + # No keepalive pings. A browser that is busy evaluating a long + # script cannot answer one -- Kitesurf, mid-snapshot on a long + # article, went quiet for over 20s -- and the client then drops a + # perfectly healthy connection with "keepalive ping timeout". + # Every call here waits for its own reply anyway, with its own + # timeout. + return wsclient.connect(ws, max_size=None, open_timeout=30, ping_interval=None, + additional_headers=headers or None) + except InvalidStatus as error: + response = error.response + if response.status_code != 429 or attempt == attempts - 1: + raise + retry = response.headers.get("Retry-After", "") + wait = float(retry) if retry.replace(".", "", 1).isdigit() else 5.0 * 2 ** attempt + # A per-minute limit asks for seconds; an exhausted daily quota + # asks for hours -- Cloudflare's free plan answered Retry-After: + # 57691. Waiting that out is not a retry, it is a hang, so say so. + if wait > MAX_RATE_WAIT: + raise RuntimeError( + f"The browser service is rate limiting this account and asks to retry " + f"in {wait / 3600:.1f}h (HTTP 429) -- a quota, not a transient limit.") \ + from None + time.sleep(wait) + raise RuntimeError("unreachable") + + +def _cdp_via_websocket(ws: str, headers: dict[str, str] | None = None): """Minimal CDP client, so the package has no transport dependency.""" import itertools import websockets.sync.client as wsclient - connection = wsclient.connect(ws, max_size=None, open_timeout=30) + ws, headers = _auth_from_url(ws, dict(headers or {})) + connection = _dial(wsclient, ws, headers) counter = itertools.count(1) def cdp(method: str, session_id: str | None = None, **params): @@ -53,32 +127,155 @@ def cdp(method: str, session_id: str | None = None, **params): return cdp -@contextmanager -def moli_session(browser_mode: str = "light"): - """Create a Moli session, attach to its page, clean up on exit. +def resolve(endpoint: str, headers: dict[str, str] | None = None) -> str: + """Turn whatever a browser hands out into its browser-level websocket URL. - `browser_mode="light"` is Moli. Pass `"normal"` for standard Chrome when you - want a control run -- this layer works on both. + Chrome, Lightpanda and most local browsers advertise an HTTP port and put + the websocket URL behind `/json/version`; cloud services usually hand out + the websocket URL itself, and some hand out an https one. Accept + all of these, so a caller can pass whatever they were given. """ - from lexmount import Lexmount - - client = Lexmount() - session = client.sessions.create(browser_mode=browser_mode, poll_timeout_sec=180) - ws = getattr(session, "ws", None) or getattr(session, "connect_url", None) + if endpoint.startswith(("ws://", "wss://")): + return endpoint + base = endpoint.rstrip("/") + request = urllib.request.Request(f"{base}/json/version", headers=headers or {}) + with urllib.request.urlopen(request, timeout=15) as reply: + ws = json.loads(reply.read()).get("webSocketDebuggerUrl") if not ws: - raise RuntimeError("Session came back without a websocket URL") + raise RuntimeError(f"{base}/json/version did not name a websocket URL") + return _reachable(ws, base) + + +def _reachable(ws: str, base: str) -> str: + """Point an advertised websocket URL at the address that actually answered. + + A browser in a container reports the address it sees from the inside: + `ws://0.0.0.0:3000/` (browserless, Steel), the container's own IP + (Selenium), or a host with no port at all. None of those can be dialled + from outside, and the connection is refused -- while the address the caller + just reached /json/version on is, by definition, reachable. Use it, keeping + the advertised path. Public hostnames are left alone: a cloud service may + legitimately hand its websocket to a different host. + """ + import ipaddress + from urllib.parse import urlsplit, urlunsplit + + advertised, asked = urlsplit(ws), urlsplit(base) + host = advertised.hostname or "" + try: + internal = ipaddress.ip_address(host) + unroutable = internal.is_unspecified or internal.is_private or internal.is_loopback + except ValueError: + unroutable = host in ("", "localhost") + if not unroutable or advertised.netloc == asked.netloc: + return ws + scheme = "wss" if asked.scheme == "https" else "ws" + return urlunsplit(advertised._replace(scheme=scheme, netloc=asked.netloc)) + + +@contextmanager +def connect(endpoint: str, *, headers: dict[str, str] | None = None, + wait_for_page: float = 0.0): + """Attach to a CDP browser and yield a Browser bound to one page. + + `endpoint` is a `ws://` / `wss://` URL, or an `http(s)://` address that + serves `/json/version`. `headers` go on the websocket handshake, for + services that authenticate there rather than in the URL (Cloudflare Browser + Run and Airtop want `Authorization: Bearer ...`); credentials written into + the URL as `user:pass@host` are sent as Basic auth. + + An existing page is used if there is one; otherwise a page is created, and + closed again on exit. Many browsers start with no page at all -- Lightpanda, + Obscura, headless-shell containers -- and waiting for one that never comes + cost 15s per run on each of them. Pass `wait_for_page` for a service that + opens its first tab a moment after handing out the endpoint, and prefers + you to use that tab. + """ + from .browser import Browser - cdp = _cdp_via_script(CDP_SCRIPT, ws) if CDP_SCRIPT else _cdp_via_websocket(ws) + ws = resolve(endpoint, headers) + cdp = (_cdp_via_script(CDP_SCRIPT, ws) if CDP_SCRIPT + else _cdp_via_websocket(ws, headers)) + created = None try: - targets = cdp("Target.getTargets")["targetInfos"] - if not targets: - raise RuntimeError("Session reported no browser targets") - page = next((t for t in targets if t["type"] == "page"), targets[0]) - attached = cdp("Target.attachToTarget", targetId=page["targetId"], flatten=True) - from .browser import Browser - yield Browser(cdp, attached["sessionId"]) + page = None + deadline = time.monotonic() + wait_for_page + while True: + with contextlib.suppress(RuntimeError): + targets = cdp("Target.getTargets").get("targetInfos") or [] + page = next((t for t in targets if t.get("type") == "page"), None) + if page is not None or time.monotonic() >= deadline: + break + time.sleep(0.25) + if page is None: + created = cdp("Target.createTarget", url="about:blank")["targetId"] + target = created or page["targetId"] + attached = cdp("Target.attachToTarget", targetId=target, flatten=True) + yield Browser(cdp, attached["sessionId"], target=target) finally: + if created: + with contextlib.suppress(Exception): + cdp("Target.closeTarget", targetId=created) if hasattr(cdp, "close"): cdp.close() + + +@contextmanager +def lexmount_session(browser_mode: str = "light"): + """Create a Lexmount session, connect to it, and delete it on exit. + + `browser_mode="light"` is Moli, a browser that keeps structure in memory + and lays pages out only when a picture is asked for. `"normal"` is the + standard Chrome image. Needs the `lexmount` extra and LEXMOUNT_* credentials. + """ + try: + from lexmount import Lexmount + except ImportError: + raise RuntimeError( + 'Lexmount support is an extra: pip install "jev-nolayout[lexmount]"') from None + + client = Lexmount() + session = client.sessions.create(browser_mode=browser_mode, poll_timeout_sec=180) + try: + ws = getattr(session, "ws", None) or getattr(session, "connect_url", None) + if not ws: + raise RuntimeError("Session came back without a websocket URL") + with connect(ws, wait_for_page=20.0) as browser: + yield browser + finally: with contextlib.suppress(Exception): client.sessions.delete(session_id=getattr(session, "session_id", None)) + + +@contextmanager +def selenium_session(grid: str, browser: str = "chrome"): + """Start a session on a Selenium Grid and connect to it over CDP. + + A Grid serves no /json/version; it hands out a CDP address only as the + `se:cdp` capability of a session it has just created -- and that address + names the node from inside the Grid's network, often a container IP. So + create the session, point the address at the Grid that answered, connect, + and end the session on exit. + """ + base = grid.rstrip("/") + request = urllib.request.Request( + f"{base}/session", method="POST", + data=json.dumps({"capabilities": {"alwaysMatch": {"browserName": browser}}}).encode(), + headers={"Content-Type": "application/json"}) + with urllib.request.urlopen(request, timeout=120) as reply: + value = json.loads(reply.read())["value"] + session_id = value["sessionId"] + try: + cdp_url = (value.get("capabilities") or {}).get("se:cdp") + if not cdp_url: + raise RuntimeError("The Grid did not offer a CDP address (se:cdp) for this browser") + with connect(_reachable(cdp_url, base)) as page: + yield page + finally: + with contextlib.suppress(Exception): + urllib.request.urlopen(urllib.request.Request( + f"{base}/session/{session_id}", method="DELETE"), timeout=30) + + +# The name this module was first published under. +moli_session = lexmount_session diff --git a/jev_nolayout/snapshot.js b/jev_nolayout/snapshot.js index a00329a..15fa2fc 100644 --- a/jev_nolayout/snapshot.js +++ b/jev_nolayout/snapshot.js @@ -25,20 +25,29 @@ // Liveness by semantics. A zero-sized box is not evidence of absence when // the box was measured two renders ago. - const live = e => - !e.closest('[aria-hidden="true"],[inert],[hidden]') - && !e.hasAttribute('hidden') - && e.getAttribute('aria-hidden') !== 'true' - && !e.matches(':disabled') - && !e.closest('[aria-disabled="true"]'); + // + // + // Asked once, natively, not once per element. What is dead is found with a + // single selector query -- every hidden, inert or aria-disabled element and + // everything under it -- and each element is then looked up in that set. + // Walking ancestors in script was half the snapshot's cost on a slow engine: + // Cloudflare's Kitesurf runs this inside a per-page CPU budget and ran out + // of it on long articles, while its selector matching is native. + const DEAD = ['[aria-hidden="true"]', '[inert]', '[hidden]', '[aria-disabled="true"]']; + const dead = new Set(document.querySelectorAll(DEAD.flatMap(d => [d, `${d} *`]).join(','))); + const live = e => !dead.has(e) && !e.matches(':disabled'); + const liveHere = e => !dead.has(e) && e.disabled !== true; // A dialog that has not been opened is still in the DOM and still passes the // liveness test above -- `aria-hidden` is often only set once it opens. The // viewport cull used to remove these for free. Without it, Google Flights // offers "Enter your origin" (the title button of a closed dialog) alongside // the real "Where from?" field, and a chooser cannot tell them apart. + const PANELS = 'dialog,[role="dialog"],[role="listbox"],[role="menu"],[popover]'; + const anyPanel = document.querySelector(PANELS) !== null; const shut = e => { - const panel = e.closest('dialog,[role="dialog"],[role="listbox"],[role="menu"],[popover]'); + if (!anyPanel) return false; // most pages have none: skip the walk + const panel = e.closest(PANELS); if (!panel) return false; if (panel.tagName === 'DIALOG') return !panel.open; if (panel.hasAttribute('popover')) return !panel.matches(':popover-open'); @@ -131,12 +140,10 @@ // A gridcell that merely wraps a button is not the button. if (rname === 'gridcell' && e.querySelector('button,[role="button"]')) continue; + // Where a control sits (region, depth) only matters when its name is + // shared, so disambiguate() works it out for those alone. const base = {node: identity(e), role: rname, label, closed: shut(e), - region: region(e), depth: (() => { - let d = 0, p = e; - while ((p = p.parentElement)) d++; - return d; - })()}; + href: e.tagName === 'A' ? e.href : ''}; for (const key of ['checked', 'selected', 'expanded']) { const value = e.getAttribute('aria-' + key); if (value !== null) base[key] = value; @@ -179,6 +186,21 @@ } for (const group of groups.values()) { if (group.length < 2) continue; + // Links that share a name and a destination are one control repeated, + // not several controls that need telling apart. Wikipedia links the + // same article from the lede, the body and two navboxes; numbering them + // "Berne Convention · 1 of 5" ... "· 5 of 5" made the one unnumbered + // near-match -- a citation pointing off-site -- look like the cleanest + // choice, and the model took it every time. + const hrefs = new Set(group.map(a => a.href || '')); + if (hrefs.size === 1 && !hrefs.has('')) continue; + for (const a of group) { + const e = cache.nodes.get(a.node); + a.region = e ? region(e) : ''; + let d = 0, p = e; + while (p && (p = p.parentElement)) d++; + a.depth = d; + } // Order is document order by depth, so the suffix is stable between // observations even when the page re-renders around the control. group.sort((x, y) => x.depth - y.depth); @@ -191,22 +213,61 @@ }; disambiguate(actions); - // textContent, not innerText: innerText is a rendered view, and on Moli it - // reports 25 characters where textContent reports 89,091. - const words = []; - const walker = document.createTreeWalker(document.body, NodeFilter.SHOW_TEXT); - let node, length = 0; - while ((node = walker.nextNode()) && length < 6000) { - const value = node.textContent.trim(); - const parent = node.parentElement; - if (!value || !parent) continue; - if (parent.closest('script,style,noscript,template')) continue; - if (!live(parent)) continue; - words.push(value); - length += value.length; - } - const text = words.join(' ').slice(0, 6000); + // The page's words, cut into retrievable units, and all of them. + // + // This used to be the first 6,000 characters of text. That is what a + // viewport-bound reader produces anyway -- it never sees more than a screen + // -- but for a reader that has the whole document it throws away what it + // just read. Measured on four long articles, the sentence carrying the + // answer was on the page every time and outside the first 6,000 characters + // every time. Choosing which rows a decision gets to see is a retrieval + // problem, so it happens on the Python side (evidence.py), with the goal in + // hand; this only has to hand over every row, in order. + // + // A block contributes its OWN text -- text nodes and inline descendants, + // with nested blocks left to speak for themselves. Taking each block's whole + // subtree would repeat every paragraph once per ancestor. A table row is the + // exception: its cells are joined into one row, "Elevation | 8,848.86 m", + // because a label and its value split into separate rows can no longer be + // paired. textContent throughout, never innerText: innerText is a rendered + // view, and on Moli it reports 25 characters where textContent reports 89,091. + const BLOCK = new Set(['ADDRESS', 'ARTICLE', 'ASIDE', 'BLOCKQUOTE', 'CAPTION', 'DD', + 'DIV', 'DL', 'DT', 'FIELDSET', 'FIGCAPTION', 'FIGURE', 'FOOTER', 'FORM', 'H1', 'H2', + 'H3', 'H4', 'H5', 'H6', 'HEADER', 'LI', 'MAIN', 'NAV', 'OL', 'P', 'PRE', 'SECTION', + 'TABLE', 'TBODY', 'TD', 'TH', 'THEAD', 'TFOOT', 'TR', 'UL', 'DETAILS', 'SUMMARY']); + const SKIP = new Set(['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEMPLATE', 'SVG', 'svg']); + const squash = s => s.replace(/\s+/g, ' ').trim(); + const own = block => { + let out = ''; + for (const child of block.childNodes) { + if (child.nodeType === 3) { out += child.nodeValue; continue; } + if (child.nodeType !== 1 || SKIP.has(child.tagName) || BLOCK.has(child.tagName)) continue; + if (!liveHere(child)) continue; + out += ' ' + child.textContent + ' '; + } + return squash(out); + }; + const rows = []; + const walk = parent => { + for (const child of parent.children) { + // Descending only through live elements is what lets liveHere() stand + // in for live(): an ancestor walk per element was half the snapshot's + // cost on a slow engine. + if (SKIP.has(child.tagName) || !liveHere(child)) continue; + if (child.tagName === 'TR') { + const cells = [...child.children].map(c => squash(c.textContent)).filter(Boolean); + if (cells.length) rows.push(cells.join(' | ')); + continue; + } + if (BLOCK.has(child.tagName)) { + const line = own(child); + if (line.length > 1) rows.push(line); // single characters are bullets + } + walk(child); + } + }; + if (live(document.body)) walk(document.body); const marker = actions.map(a => `${a.node}:${a.kind}:${a.label}`).join('|'); - return {url: location.href, title: document.title, text, actions, marker}; + return {url: location.href, title: document.title, rows, actions, marker}; })() diff --git a/pyproject.toml b/pyproject.toml index 20ae719..0ac6cd3 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,16 +1,18 @@ [project] name = "jev-nolayout" version = "0.1.0" -description = "A browser agent for Moli that reads structure instead of layout" +description = "A browser agent that reads structure instead of layout, on any CDP browser" readme = "README.md" requires-python = ">=3.11" license = { text = "Apache-2.0" } dependencies = [ "httpx[http2]>=0.28,<1", - "lexmount>=0.5", "websockets>=13", ] +[project.optional-dependencies] +lexmount = ["lexmount>=0.5"] + [project.scripts] jev-nolayout = "jev_nolayout.cli:main" diff --git a/uv.lock b/uv.lock index 1abec03..c725f84 100644 --- a/uv.lock +++ b/uv.lock @@ -268,16 +268,21 @@ version = "0.1.0" source = { editable = "." } dependencies = [ { name = "httpx", extra = ["http2"] }, - { name = "lexmount" }, { name = "websockets" }, ] +[package.optional-dependencies] +lexmount = [ + { name = "lexmount" }, +] + [package.metadata] requires-dist = [ { name = "httpx", extras = ["http2"], specifier = ">=0.28,<1" }, - { name = "lexmount", specifier = ">=0.5" }, + { name = "lexmount", marker = "extra == 'lexmount'", specifier = ">=0.5" }, { name = "websockets", specifier = ">=13" }, ] +provides-extras = ["lexmount"] [[package]] name = "lexmount"