From 3c206c34047cc8a70ad2b978c3ac420eea2b5437 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A0=20Arrufat?= Date: Fri, 4 Sep 2026 09:37:46 +0200 Subject: [PATCH 1/2] Retire the hand-written Python SDK reference in favor of the generated page The generated page already covers every class, method, property and exception, and the hand-written tables had to be re-synced with the package on every release. Drop reference/python.mdx, give the generated page the "Python SDK" title and sidebar slot, and move the conventions worth keeping (keyword-only snake_case actions, the selector / backend_node_id pair, Session.call as the escape hatch, ToolError on failure) into the generator's fixed intro until they live in the Session docstring upstream. Redirect /reference/python to the generated page and retarget the guide's link. --- redirects.mjs | 1 + scripts/generate-python-reference.py | 25 +++- src/content/guides/use-python.mdx | 2 +- src/content/reference/_meta.ts | 3 +- src/content/reference/python-api.mdx | 10 +- src/content/reference/python.mdx | 176 --------------------------- 6 files changed, 28 insertions(+), 189 deletions(-) delete mode 100644 src/content/reference/python.mdx diff --git a/redirects.mjs b/redirects.mjs index 0594d4e..e6e745d 100644 --- a/redirects.mjs +++ b/redirects.mjs @@ -11,6 +11,7 @@ export const basePath = '/docs' export const redirects = { '/python': '/reference/python-api', + '/reference/python': '/reference/python-api', '/quickstart/installation-and-setup': '/quickstart', '/quickstart/your-first-test': '/quickstart', '/quickstart/build-your-first-extraction-script': '/quickstart', diff --git a/scripts/generate-python-reference.py b/scripts/generate-python-reference.py index 7554688..4c78d6b 100644 --- a/scripts/generate-python-reference.py +++ b/scripts/generate-python-reference.py @@ -37,8 +37,8 @@ DEFAULT_OUT = ROOT / "src" / "content" / "reference" / "python-api.mdx" FRONTMATTER = """--- -title: Python API -description: Generated reference of every public class, method, property and exception in the lightpanda Python package, with the signatures and docstrings shipped in the code. +title: Python SDK +description: Reference of every public class, method, property and exception in the lightpanda Python package, generated from the signatures and docstrings shipped in the code. --- """ @@ -50,12 +50,24 @@ INTRO = ( "Every public class, method, property and exception of the " "[`lightpanda` package](https://pypi.org/project/lightpanda/), with the signatures and " - "docstrings shipped in the code. See [Python SDK](/reference/python) for a curated " - "overview of the same API and [Use the Python SDK](/guides/use-python) for practical " - "documentation. Every sync class has an asyncio twin with the same methods, " + "docstrings shipped in the code. See [Use the Python SDK](/guides/use-python) for a " + "practical walkthrough. Every sync class has an asyncio twin with the same methods, " "awaitable; the async sections below list only what the twin adds." ) +CONVENTIONS = ( + "Browser actions are keyword-only methods on [`Session`](#session) and " + "[`AsyncSession`](#asyncsession), named in snake_case after the browser's own action " + "names: the `waitForSelector` action is `wait_for_selector`, and its `backendNodeId` " + "argument is `backend_node_id`. Where a method accepts both `selector` and " + "`backend_node_id`, pass one of the two; `selector` is preferred for reproducibility and " + "wins when both are given, and `backend_node_id` takes the values returned by " + "[`tree`](#session-tree), [`links`](#session-links) or " + "[`find_element`](#session-find-element). [`Session.call`](#session-call) is the escape " + "hatch that takes the action and argument names exactly as the browser declares them. " + "A failed action raises [`ToolError`](#toolerror)." +) + FENCE_RE = re.compile(r"^\s*```") CODE_SPAN_RE = re.compile(r"(`+)(.+?)\1", re.DOTALL) MODULE_PREFIX_RE = re.compile(r"\blightpanda\.\w+\.") @@ -382,9 +394,10 @@ def generate() -> str: page.lines.append(FRONTMATTER.rstrip()) page.lines.append(BANNER) page.lines.append("") - page.lines.append("# Python API") + page.lines.append("# Python SDK") page.lines.append("") page.para(INTRO) + page.para(CONVENTIONS) page.para(render_docstring(module, links)) exceptions: list[pdoc.doc.Class] = [] diff --git a/src/content/guides/use-python.mdx b/src/content/guides/use-python.mdx index b48af54..352a7f6 100644 --- a/src/content/guides/use-python.mdx +++ b/src/content/guides/use-python.mdx @@ -164,7 +164,7 @@ asyncio.run(main()) Every browser action is a `Session` method, typed and documented in your IDE, with the action and its arguments in snake_case (`wait_for_selector`, `backend_node_id`). -Find every method's arguments in the [Python SDK reference](/reference/python), or browse the generated [Python API](/reference/python-api) reference. +Find every method's signature and docstring in the [Python SDK reference](/reference/python-api). ## Replay a saved script diff --git a/src/content/reference/_meta.ts b/src/content/reference/_meta.ts index 7f6d572..bfb6d1d 100644 --- a/src/content/reference/_meta.ts +++ b/src/content/reference/_meta.ts @@ -10,8 +10,7 @@ const meta: MetaRecord = { 'http-api': 'HTTP API', 'mcp-tools': 'MCP tools', pandascript: 'PandaScript', - python: 'Python SDK', - 'python-api': 'Python API', + 'python-api': 'Python SDK', } export default meta diff --git a/src/content/reference/python-api.mdx b/src/content/reference/python-api.mdx index cb3ca3f..5791e6f 100644 --- a/src/content/reference/python-api.mdx +++ b/src/content/reference/python-api.mdx @@ -1,12 +1,14 @@ --- -title: Python API -description: Generated reference of every public class, method, property and exception in the lightpanda Python package, with the signatures and docstrings shipped in the code. +title: Python SDK +description: Reference of every public class, method, property and exception in the lightpanda Python package, generated from the signatures and docstrings shipped in the code. --- {/* Generated by scripts/generate-python-reference.py from lightpanda-python main. Do not edit; rerun the script or wait for the python-reference workflow. */} -# Python API +# Python SDK -Every public class, method, property and exception of the [`lightpanda` package](https://pypi.org/project/lightpanda/), with the signatures and docstrings shipped in the code. See [Python SDK](/reference/python) for a curated overview of the same API and [Use the Python SDK](/guides/use-python) for practical documentation. Every sync class has an asyncio twin with the same methods, awaitable; the async sections below list only what the twin adds. +Every public class, method, property and exception of the [`lightpanda` package](https://pypi.org/project/lightpanda/), with the signatures and docstrings shipped in the code. See [Use the Python SDK](/guides/use-python) for a practical walkthrough. Every sync class has an asyncio twin with the same methods, awaitable; the async sections below list only what the twin adds. + +Browser actions are keyword-only methods on [`Session`](#session) and [`AsyncSession`](#asyncsession), named in snake_case after the browser's own action names: the `waitForSelector` action is `wait_for_selector`, and its `backendNodeId` argument is `backend_node_id`. Where a method accepts both `selector` and `backend_node_id`, pass one of the two; `selector` is preferred for reproducibility and wins when both are given, and `backend_node_id` takes the values returned by [`tree`](#session-tree), [`links`](#session-links) or [`find_element`](#session-find-element). [`Session.call`](#session-call) is the escape hatch that takes the action and argument names exactly as the browser declares them. A failed action raises [`ToolError`](#toolerror). Lightpanda for Python: a lightweight headless browser. diff --git a/src/content/reference/python.mdx b/src/content/reference/python.mdx deleted file mode 100644 index c60d4b3..0000000 --- a/src/content/reference/python.mdx +++ /dev/null @@ -1,176 +0,0 @@ ---- -title: Python SDK -description: Reference of the classes and methods in the Lightpanda Python package, covering every browser action and script replay. ---- - -# Python SDK - -The [`lightpanda` package](https://pypi.org/project/lightpanda/) exposes `Browser`/`AsyncBrowser`, which spawn and manage the bundled binary, and `Session`/`AsyncSession`, with one method per browser action. See [Use the Python SDK](/guides/use-python) for practical documentation, and the [Python API](/reference/python-api) page for every signature and docstring as shipped in the package. - -## Browser - -[`Browser()`](/reference/python-api#browser) spawns the bundled binary when constructed. It is not fork-inheritable: create a fresh instance in a forked child. - -| Argument | Default | Description | -|---|---|---| -| `binary` | `None` | Path to a specific lightpanda binary. When omitted, resolved from the `LIGHTPANDA_BIN` environment variable, then the binary bundled in the package, then `PATH`. | -| `env` | `None` | Extra environment variables for the spawned process. | -| `timeout` | `300.0` | Seconds to wait for a response before raising `ProtocolError`. | -| `verbose` | `False` | Print the spawned process's own logging. | -| `args` | `()` | Extra CLI flags for the spawned process, for example `["--http-cache-dir", path]`. | - -| Method | Returns | Description | -|---|---|---| -| `new_session()` | `Session` | Open a new isolated browsing context: its own page, cookies, and memory. | -| `tools` (property) | `dict[str, dict]` | Every available action, as `name → {description, schema}`, reported live by the running browser. | -| `close()` | `None` | Stop the browser process. | - -`with Browser() as b:` calls `close()` on exit. - -## AsyncBrowser - -[`AsyncBrowser`](/reference/python-api#asyncbrowser) mirrors `Browser` for asyncio: every call runs on a browser-owned thread pool, so the event loop is never blocked. - -| Argument | Default | Description | -|---|---|---| -| `binary`, `env`, `timeout`, `verbose`, `args` | same as `Browser` | Forwarded to the underlying `Browser`. | -| `max_concurrency` | `32` | Caps method calls executing concurrently across this browser's sessions. Worker threads are created lazily. | - -| Method | Returns | Description | -|---|---|---| -| `start()` | `AsyncBrowser` | Spawn the process and fetch its action list. Idempotent; called automatically on `async with` entry and by `new_session()`. | -| `new_session()` | `AsyncSession` | Start the browser if needed, then open a new session. | -| `session()` | async context manager | `async with browser.session() as page:` opens a session scoped to the block and closes it on exit. | -| `tools` (property) | `dict[str, dict]` | Same as `Browser.tools`. | -| `close()` | `None` | Stop the browser process, unless it was adopted with `wrap` (see below). | -| `AsyncBrowser.wrap(browser, max_concurrency=32)` (classmethod) | `AsyncBrowser` | Adopt an already-running `Browser` for use from asyncio. `close()` then shuts down only the async facade, leaving the wrapped browser running. | - -`async with AsyncBrowser() as b:` calls `start()` on entry and `close()` on exit. - -## Session and AsyncSession - -`Browser.new_session()` and `AsyncBrowser.new_session()` are the only way to obtain a [`Session`](/reference/python-api#session) or [`AsyncSession`](/reference/python-api#asyncsession); do not construct one directly. - -| Member | Description | -|---|---| -| `id` (property) | The session's id. | -| `close()` | Close the session. Calls made after `close()` raise `ToolError`. | -| `call(action, **kwargs)` | Invoke any action by name. The methods documented below route through this; it also accepts the action and argument names exactly as the browser declares them, for example `page.call("tree", maxDepth=1)`. | - -Sessions are context managers too: `with browser.new_session() as page:` closes the session on exit. Closing the browser ends every session anyway. - -## Calling an action - -Every browser action is a method on `Session`/`AsyncSession`, keyword-only, with the action and its arguments in snake_case: the `waitForSelector` action is `wait_for_selector`, and its `backendNodeId` argument is `backend_node_id`. The methods are generated from the bundled browser's action schemas, so the signatures and docstrings your IDE shows come straight from the binary. The [Python API](/reference/python-api#session) page lists every method with its exact signature and docstring. - -A failed action raises `ToolError`. - -In the Arguments column below, `?` marks an optional keyword argument, and `selector` / `backend_node_id` marks a pair where one of the two is required. Prefer `selector` for reproducibility; it also wins when you pass both. `backend_node_id` takes the `backendNodeId` values returned by a prior `tree`, `links`, or `find_element` call. In the Returns column, `JSON` is a parsed Python `dict` or `list`, `text` is a plain string. - -### Navigation and search - -These methods bring a page into the browser: - -| Method | Arguments | Returns | Description | -|---|---|---|---| -| `goto` | `url`, `timeout?`, `wait_until?` | text | Navigate to a URL and load the page in memory so it can be reused later for info extraction. `wait_until` accepts the same states as `wait_for_state` and defaults to `load`. | -| `search` | `query`, `timeout?` | text | Run a web search and return results as markdown: a numbered list of `{title, url, snippet}`. The browser does not navigate; to open a result, call `goto` with its URL. | - -### Reading the page - -These methods read the loaded page without modifying it: - -| Method | Arguments | Returns | Description | -|---|---|---|---| -| `markdown` | `selector?`, `backend_node_id?`, `max_bytes?`, `url?`, `timeout?` | text | Render the page, or a subtree, as markdown. Scope with `selector` or `backend_node_id` to read just the relevant region; use `max_bytes` to cap long pages. | -| `html` | `selector?`, `backend_node_id?`, `max_bytes?`, `strip?`, `url?`, `timeout?` | text | Raw HTML for the document, or a single node's outerHTML when scoped. Verbose; use only when you need attributes that markdown discards. Use `max_bytes` to cap long pages. `strip` is an object of element groups to omit: `js` (script, noscript, script preloads), `css` (style, stylesheet links), `ui` (css plus img, picture, video, audio, svg, canvas, iframe) and `invisible` (elements set to display:none). `{"js": True, "css": True}` keeps a page dump small. | -| `screenshot` | `path?`, `selector?`, `backend_node_id?`, `full_page?`, `url?`, `timeout?` | text or bytes | Render the page, or one node, as a PNG: the text layout Lightpanda computes, not a pixel-accurate rendering (no images, fonts, or CSS colors). `path` must be a relative path. Without `path`, the PNG is returned as `bytes`. | -| `tree` | `url?`, `timeout?`, `backend_node_id?`, `max_depth?` | text | Simplified semantic DOM tree: role, name, value, and `backendNodeId` per node. | -| `links` | `limit?`, `url?`, `timeout?` | JSON | Extract all links as `text` (visible anchor text, falling back to aria-label/title/image alt), `href` (resolved URL), and `backendNodeId` (pass to `node_details`). One entry per href; hidden links are omitted. `limit` returns at most that many links, in document order. | -| `node_details` | `backend_node_id` | JSON | Tag, role, name, value, and other state for a node, plus a ready-to-use CSS `selector` that resolves to it. The way to turn a `backendNodeId` into a selector. | -| `find_element` | `role?`, `name?` | JSON | Find interactive elements by role and/or accessible name, with their `backendNodeId`. | -| `interactive_elements` | `url?`, `timeout?` | JSON | Every interactive element on the page. | -| `structured_data` | `url?`, `timeout?` | JSON | Structured data on the page, such as JSON-LD or OpenGraph tags. | -| `detect_forms` | `url?`, `timeout?` | JSON | Forms on the page: fields, types, and required status. | - -### Data extraction and scripting - -| Method | Arguments | Returns | Description | -|---|---|---|---| -| `extract` | `schema`, `save?` | JSON | Extract structured data from the current page using a schema mapping output field names to CSS-selector specs. | -| `evaluate` | `script`, `url?`, `timeout?`, `save?` | typed | Evaluate a JavaScript string in the page context and return its value. Runs in the page, so it cannot see your Python variables. | - -**`evaluate`'s return** is typed like the JavaScript result: for example `1+1` comes back as the `int` `2`, and `({a:1})` comes back as the `dict` `{"a": 1}`. - -**`extract`'s `schema`** maps output field names to CSS-selector specs. Pass it as a Python `dict` or `list`, it's encoded for you; a JSON string also works: - -| Schema value | Result | -|---|---| -| `""` | First match's text, or `None` | -| `[""]` | Every match's text | -| `{"selector": "", "attr": ""}` | First match's attribute (`href`/`src` resolve to absolute URLs) | -| `[{"selector": "", "attr": ""}]` | Every match's attribute | -| `[{"selector": "", "fields": {...}}]` | One dict per match, with fields resolved relative to each match | - -Add `"limit": N` inside any array spec to cap matches. Every extracted value is a string or `None`; parse numbers yourself. - -### Interacting with the page - -These methods dispatch real DOM events on the page: - -| Method | Arguments | Returns | Description | -|---|---|---|---| -| `click` | `selector` / `backend_node_id` | text | Click an interactive element. | -| `fill` | `selector` / `backend_node_id`, `value` | text | Fill text into an input element. | -| `scroll` | `backend_node_id?`, `x?`, `y?` | text | Scroll the page, or a specific element if `backend_node_id` is given. | -| `hover` | `selector` / `backend_node_id` | text | Hover over an element, triggering `mouseover` and `mouseenter`. | -| `press` | `key`, `selector?`, `backend_node_id?` | text | Press a keyboard key, dispatching `keydown` and `keyup`. Targets the document if no element is given. | -| `select_option` | `selector` / `backend_node_id`, `value` | text | Select an option in a `