mirror of
https://github.com/HKUDS/nanobot.git
synced 2026-08-09 05:48:38 +03:00
fix(tools): harden computer use integration
This commit is contained in:
@@ -101,6 +101,3 @@ exp/
|
|||||||
.playwright-mcp/
|
.playwright-mcp/
|
||||||
bridge/node_modules/
|
bridge/node_modules/
|
||||||
webui/.verify-*
|
webui/.verify-*
|
||||||
|
|
||||||
# Computer-use E2E run artifacts
|
|
||||||
tests/e2e/runs/
|
|
||||||
|
|||||||
@@ -125,6 +125,7 @@ Important files:
|
|||||||
| Shell execution | `nanobot/agent/tools/shell.py` |
|
| Shell execution | `nanobot/agent/tools/shell.py` |
|
||||||
| Filesystem tools | `nanobot/agent/tools/filesystem.py` |
|
| Filesystem tools | `nanobot/agent/tools/filesystem.py` |
|
||||||
| Web search/fetch | `nanobot/agent/tools/web.py` |
|
| Web search/fetch | `nanobot/agent/tools/web.py` |
|
||||||
|
| Browser and computer use | `nanobot/agent/tools/browser_tool.py`, `nanobot/agent/tools/computer_use.py` |
|
||||||
| MCP tools | `nanobot/agent/tools/mcp.py` |
|
| MCP tools | `nanobot/agent/tools/mcp.py` |
|
||||||
| Cron | `nanobot/agent/tools/cron.py`, `nanobot/cron/` |
|
| Cron | `nanobot/agent/tools/cron.py`, `nanobot/cron/` |
|
||||||
| Image generation | `nanobot/agent/tools/image_generation.py` |
|
| Image generation | `nanobot/agent/tools/image_generation.py` |
|
||||||
@@ -188,7 +189,7 @@ Security-sensitive code paths include:
|
|||||||
|---|---|
|
|---|---|
|
||||||
| Workspace scope | `nanobot/security/workspace_access.py`, `nanobot/security/workspace_policy.py` |
|
| Workspace scope | `nanobot/security/workspace_access.py`, `nanobot/security/workspace_policy.py` |
|
||||||
| Shell sandboxing | `nanobot/agent/tools/shell.py` |
|
| Shell sandboxing | `nanobot/agent/tools/shell.py` |
|
||||||
| SSRF/network checks | `nanobot/security/network.py`, `nanobot/agent/tools/web.py` |
|
| SSRF/network checks | `nanobot/security/network.py`, `nanobot/agent/tools/web.py`, `nanobot/agent/tools/computer_use_backends/browser_playwright.py` |
|
||||||
| PTH guard and CLI startup security | `nanobot/security/` and CLI entrypoints |
|
| PTH guard and CLI startup security | `nanobot/security/` and CLI entrypoints |
|
||||||
| Channel access control | channel config in `nanobot/channels/*.py` |
|
| Channel access control | channel config in `nanobot/channels/*.py` |
|
||||||
|
|
||||||
|
|||||||
@@ -42,6 +42,7 @@ the focused guides first and come back here for exact fields and defaults.
|
|||||||
| Add fallback chains | [Model Fallbacks](#model-fallbacks) |
|
| Add fallback chains | [Model Fallbacks](#model-fallbacks) |
|
||||||
| Configure voice transcription | [Transcription Settings](#transcription-settings) |
|
| Configure voice transcription | [Transcription Settings](#transcription-settings) |
|
||||||
| Tune channel defaults | [Channel Settings](#channel-settings) |
|
| Tune channel defaults | [Channel Settings](#channel-settings) |
|
||||||
|
| Enable browser or desktop control | [Browser and Computer Use](#browser-and-computer-use) |
|
||||||
| Configure web search and fetch | [Web Tools](#web-tools) |
|
| Configure web search and fetch | [Web Tools](#web-tools) |
|
||||||
| Enable image generation | [Image Generation](#image-generation) |
|
| Enable image generation | [Image Generation](#image-generation) |
|
||||||
| Add MCP servers | [MCP](#mcp-model-context-protocol) |
|
| Add MCP servers | [MCP](#mcp-model-context-protocol) |
|
||||||
@@ -1670,6 +1671,58 @@ When a channel `send()` raises, nanobot retries at the channel-manager layer. By
|
|||||||
>
|
>
|
||||||
> If a channel is completely unreachable, nanobot cannot notify the user through that same channel. Watch logs for `Failed to send to {channel} after N attempts` to spot persistent delivery failures.
|
> If a channel is completely unreachable, nanobot cannot notify the user through that same channel. Watch logs for `Failed to send to {channel} after N attempts` to spot persistent delivery failures.
|
||||||
|
|
||||||
|
## Browser and Computer Use
|
||||||
|
|
||||||
|
Browser and desktop control are optional and disabled by default. Install their runtime first:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install 'nanobot-ai[computer-use]'
|
||||||
|
playwright install chromium
|
||||||
|
```
|
||||||
|
|
||||||
|
For normal web interaction, prefer the DOM-based `browser` tool. It gives the model numbered
|
||||||
|
element references and works without vision. Use `computer_use` when the model must see and act
|
||||||
|
on pixels; its `desktop` backend controls the real local machine, while its `browser` backend
|
||||||
|
controls an isolated Playwright page.
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"tools": {
|
||||||
|
"browser": {
|
||||||
|
"enable": true,
|
||||||
|
"allowedDomains": ["example.com"]
|
||||||
|
},
|
||||||
|
"computerUse": {
|
||||||
|
"enable": false,
|
||||||
|
"backend": "desktop"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
| Option | Default | Description |
|
||||||
|
|---|---|---|
|
||||||
|
| `tools.browser.enable` | `false` | Register the DOM-based `browser` tool |
|
||||||
|
| `tools.browser.allowedDomains` | `[]` | Optional top-level navigation allowlist; entries include subdomains |
|
||||||
|
| `tools.browser.includeScreenshot` | `false` | Attach a screenshot after browser actions |
|
||||||
|
| `tools.computerUse.enable` | `false` | Register pixel-based `computer_use` |
|
||||||
|
| `tools.computerUse.backend` | `"desktop"` | `"desktop"` or `"browser"` |
|
||||||
|
| `tools.computerUse.allowedDomains` | `[]` | Navigation allowlist for the browser backend |
|
||||||
|
| `tools.computerUse.targetWidth` / `targetHeight` | `1280` / `800` | Maximum screenshot dimensions exposed to the model |
|
||||||
|
|
||||||
|
Each nanobot session gets separate browser state. Browser HTTP and WebSocket traffic passes
|
||||||
|
through the shared SSRF policy; local, private, link-local, and metadata targets are blocked
|
||||||
|
unless explicitly permitted with `tools.ssrfWhitelist`. `file:` URLs are not accepted.
|
||||||
|
|
||||||
|
> [!IMPORTANT]
|
||||||
|
> Browser URL checks are defense in depth, not an egress sandbox: Chromium performs its own DNS
|
||||||
|
> resolution after validation. Use OS/container network isolation when browsing hostile pages.
|
||||||
|
|
||||||
|
> [!WARNING]
|
||||||
|
> The desktop backend can click, type, and change state outside the workspace. Enabling it is an
|
||||||
|
> explicit trust decision: use a trusted model and input source, and run nanobot in a disposable
|
||||||
|
> OS account or VM when unattended. The workspace restriction is not an OS sandbox.
|
||||||
|
|
||||||
## Web Tools
|
## Web Tools
|
||||||
|
|
||||||
nanobot incorporates basic tools for accessing the web. These include searching via APIs, and fetching arbitrary web pages in Markdown format. They are enabled by default, and can be configured in `~/.nanobot/config.json` under `tools.web`.
|
nanobot incorporates basic tools for accessing the web. These include searching via APIs, and fetching arbitrary web pages in Markdown format. They are enabled by default, and can be configured in `~/.nanobot/config.json` under `tools.web`.
|
||||||
|
|||||||
@@ -1397,6 +1397,7 @@ class AgentLoop:
|
|||||||
cleanup_steps = (
|
cleanup_steps = (
|
||||||
self.subagents.close,
|
self.subagents.close,
|
||||||
self._exec_session_manager.close_all,
|
self._exec_session_manager.close_all,
|
||||||
|
*(() if not hasattr(self, "tools") else (self.tools.close,)),
|
||||||
lambda: agent_context.close_mcp(self),
|
lambda: agent_context.close_mcp(self),
|
||||||
)
|
)
|
||||||
for cleanup in cleanup_steps:
|
for cleanup in cleanup_steps:
|
||||||
|
|||||||
@@ -220,6 +220,10 @@ class Tool(ABC):
|
|||||||
"""Return optional per-turn prompt context owned by this tool."""
|
"""Return optional per-turn prompt context owned by this tool."""
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
"""Release resources owned by the tool. Safe to call repeatedly."""
|
||||||
|
return None
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
async def execute(self, **kwargs: Any) -> Any:
|
async def execute(self, **kwargs: Any) -> Any:
|
||||||
"""Run the tool; return content, or ``ToolResult.error(...)`` for failures."""
|
"""Run the tool; return content, or ``ToolResult.error(...)`` for failures."""
|
||||||
|
|||||||
@@ -1,33 +1,23 @@
|
|||||||
"""browser tool: DOM/accessibility-based web automation (model-agnostic actions).
|
"""DOM-based browser automation by element reference."""
|
||||||
|
|
||||||
Unlike the pixel-based ``computer_use`` tool, this drives the page by **element
|
# pyright: reportIncompatibleMethodOverride=false
|
||||||
ref** instead of by screen coordinates. Each action returns a fresh snapshot of
|
|
||||||
the page's interactive elements (a numbered list), and the model acts by picking
|
|
||||||
a ``[ref]`` — no pixel grounding required. This makes web *actions* reliable
|
|
||||||
across ANY tool-calling model (including non-vision models), which pixel-based
|
|
||||||
computer use cannot do (only computer-use-trained models ground pixels well).
|
|
||||||
|
|
||||||
It reuses the Playwright ``BrowserBackend`` from ``computer_use_backends``. The
|
|
||||||
heavy ``playwright`` dependency is imported lazily, so importing this module at
|
|
||||||
tool auto-discovery time stays cheap.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from typing import Any
|
from typing import Any
|
||||||
from urllib.parse import urlparse
|
|
||||||
|
|
||||||
from loguru import logger
|
|
||||||
from pydantic import Field
|
from pydantic import Field
|
||||||
|
|
||||||
from nanobot.agent.tools.base import Tool, tool_parameters
|
from nanobot.agent.tools.base import Tool, tool_parameters
|
||||||
|
from nanobot.agent.tools.computer_use_backends.base import SessionBackendPool
|
||||||
|
from nanobot.agent.tools.context import ToolContext
|
||||||
from nanobot.agent.tools.schema import (
|
from nanobot.agent.tools.schema import (
|
||||||
BooleanSchema,
|
BooleanSchema,
|
||||||
IntegerSchema,
|
IntegerSchema,
|
||||||
StringSchema,
|
StringSchema,
|
||||||
tool_parameters_schema,
|
tool_parameters_schema,
|
||||||
)
|
)
|
||||||
from nanobot.config.schema import Base
|
from nanobot.config_base import Base
|
||||||
from nanobot.utils.helpers import build_image_content_blocks
|
from nanobot.utils.helpers import build_image_content_blocks
|
||||||
|
|
||||||
_ACTIONS = [
|
_ACTIONS = [
|
||||||
@@ -41,35 +31,34 @@ _ACTIONS = [
|
|||||||
"back",
|
"back",
|
||||||
"read_text",
|
"read_text",
|
||||||
]
|
]
|
||||||
_ENABLE_WARNED = False
|
|
||||||
|
|
||||||
|
|
||||||
class BrowserToolConfig(Base):
|
class BrowserToolConfig(Base):
|
||||||
"""browser (DOM) tool configuration."""
|
"""browser (DOM) tool configuration."""
|
||||||
|
|
||||||
enable: bool = False # off by default — opt-in
|
enable: bool = False
|
||||||
start_url: str = "about:blank"
|
start_url: str = "about:blank"
|
||||||
headless: bool = True
|
headless: bool = True
|
||||||
width: int = 1280
|
width: int = Field(default=1280, ge=320, le=4096)
|
||||||
height: int = 800
|
height: int = Field(default=800, ge=240, le=4096)
|
||||||
allowed_domains: list[str] = Field(default_factory=list) # empty = all
|
allowed_domains: list[str] = Field(default_factory=list)
|
||||||
include_screenshot: bool = False # also attach a screenshot (for vision models)
|
include_screenshot: bool = False
|
||||||
max_elements: int = 200 # cap interactive elements per snapshot
|
max_elements: int = Field(default=200, ge=1, le=1000)
|
||||||
|
|
||||||
|
|
||||||
def _format_elements(elements: list[dict]) -> str:
|
def _format_elements(elements: list[dict[str, Any]]) -> str:
|
||||||
if not elements:
|
if not elements:
|
||||||
return "Interactive elements: (none found — try scrolling or read_text)"
|
return "Interactive elements: (none found — try scrolling or read_text)"
|
||||||
lines = []
|
lines: list[str] = []
|
||||||
for e in elements:
|
for e in elements:
|
||||||
tag = e.get("tag", "")
|
tag = str(e.get("tag") or "")
|
||||||
typ = e.get("type") or ""
|
typ = str(e.get("type") or "")
|
||||||
label = tag + (f"[{typ}]" if typ else "")
|
label = tag + (f"[{typ}]" if typ else "")
|
||||||
line = f"[{e.get('ref')}] {label}"
|
line = f"[{e.get('ref')}] {label}"
|
||||||
name = (e.get("name") or "").strip()
|
name = str(e.get("name") or "").strip()
|
||||||
if name:
|
if name:
|
||||||
line += f' "{name}"'
|
line += f' "{name}"'
|
||||||
href = e.get("href") or ""
|
href = str(e.get("href") or "")
|
||||||
if href and tag == "a":
|
if href and tag == "a":
|
||||||
line += f" -> {href[:60]}"
|
line += f" -> {href[:60]}"
|
||||||
lines.append(line)
|
lines.append(line)
|
||||||
@@ -81,6 +70,7 @@ def _format_elements(elements: list[dict]) -> str:
|
|||||||
action=StringSchema("The action to perform.", enum=_ACTIONS),
|
action=StringSchema("The action to perform.", enum=_ACTIONS),
|
||||||
ref=IntegerSchema(
|
ref=IntegerSchema(
|
||||||
description="Element ref number from the latest snapshot (click/type/select).",
|
description="Element ref number from the latest snapshot (click/type/select).",
|
||||||
|
minimum=1,
|
||||||
nullable=True,
|
nullable=True,
|
||||||
),
|
),
|
||||||
text=StringSchema(
|
text=StringSchema(
|
||||||
@@ -93,7 +83,12 @@ def _format_elements(elements: list[dict]) -> str:
|
|||||||
scroll_direction=StringSchema(
|
scroll_direction=StringSchema(
|
||||||
"Scroll direction (action=scroll).", enum=["up", "down", "left", "right"], nullable=True
|
"Scroll direction (action=scroll).", enum=["up", "down", "left", "right"], nullable=True
|
||||||
),
|
),
|
||||||
scroll_amount=IntegerSchema(description="Scroll clicks (action=scroll).", nullable=True),
|
scroll_amount=IntegerSchema(
|
||||||
|
description="Scroll clicks (action=scroll).",
|
||||||
|
minimum=1,
|
||||||
|
maximum=100,
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
required=["action"],
|
required=["action"],
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
@@ -102,8 +97,8 @@ class BrowserTool(Tool):
|
|||||||
|
|
||||||
_scopes = {"core"}
|
_scopes = {"core"}
|
||||||
|
|
||||||
name = "browser"
|
name = "browser" # pyright: ignore[reportIncompatibleMethodOverride, reportAssignmentType]
|
||||||
description = (
|
description = ( # pyright: ignore[reportIncompatibleMethodOverride, reportAssignmentType]
|
||||||
"Control a web browser by acting on page elements by their [ref] number. "
|
"Control a web browser by acting on page elements by their [ref] number. "
|
||||||
"Each call returns the current page URL plus a fresh numbered list of the page's "
|
"Each call returns the current page URL plus a fresh numbered list of the page's "
|
||||||
"interactive elements; pick a [ref] to click/type/select — no pixel coordinates "
|
"interactive elements; pick a [ref] to click/type/select — no pixel coordinates "
|
||||||
@@ -120,50 +115,30 @@ class BrowserTool(Tool):
|
|||||||
return BrowserToolConfig
|
return BrowserToolConfig
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def enabled(cls, ctx: Any) -> bool:
|
def enabled(cls, ctx: ToolContext) -> bool:
|
||||||
return bool(ctx.config.browser.enable)
|
return bool(ctx.config.browser.enable)
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def create(cls, ctx: Any) -> Tool:
|
def create(cls, ctx: ToolContext) -> Tool:
|
||||||
cfg = ctx.config.browser
|
return cls(ctx.config.browser)
|
||||||
global _ENABLE_WARNED
|
|
||||||
if not _ENABLE_WARNED:
|
|
||||||
logger.warning(
|
|
||||||
"browser tool is ENABLED — it can navigate and act on web pages. "
|
|
||||||
"Restrict with allowed_domains, run in a sandbox, and beware prompt "
|
|
||||||
"injection from page content."
|
|
||||||
)
|
|
||||||
_ENABLE_WARNED = True
|
|
||||||
return cls(
|
|
||||||
start_url=cfg.start_url,
|
|
||||||
headless=cfg.headless,
|
|
||||||
width=cfg.width,
|
|
||||||
height=cfg.height,
|
|
||||||
allowed_domains=list(cfg.allowed_domains),
|
|
||||||
include_screenshot=cfg.include_screenshot,
|
|
||||||
max_elements=cfg.max_elements,
|
|
||||||
)
|
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
|
config: BrowserToolConfig | None = None,
|
||||||
*,
|
*,
|
||||||
start_url: str = "about:blank",
|
|
||||||
headless: bool = True,
|
|
||||||
width: int = 1280,
|
|
||||||
height: int = 800,
|
|
||||||
allowed_domains: list[str] | None = None,
|
|
||||||
include_screenshot: bool = False,
|
|
||||||
max_elements: int = 200,
|
|
||||||
backend_impl: Any = None,
|
backend_impl: Any = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
self.start_url = start_url
|
self.config = config or BrowserToolConfig()
|
||||||
self.headless = headless
|
runtime = None
|
||||||
self.width = width
|
if backend_impl is None:
|
||||||
self.height = height
|
from nanobot.agent.tools.computer_use_backends.browser_playwright import BrowserRuntime
|
||||||
self.allowed_domains = allowed_domains or []
|
runtime = BrowserRuntime(headless=self.config.headless)
|
||||||
self.include_screenshot = include_screenshot
|
self._runtime = runtime
|
||||||
self.max_elements = max_elements
|
self._backends = SessionBackendPool(
|
||||||
self._backend = backend_impl
|
self._make_backend,
|
||||||
|
backend_impl,
|
||||||
|
finalizer=runtime.close if runtime is not None else None,
|
||||||
|
)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def read_only(self) -> bool:
|
def read_only(self) -> bool:
|
||||||
@@ -173,29 +148,15 @@ class BrowserTool(Tool):
|
|||||||
def exclusive(self) -> bool:
|
def exclusive(self) -> bool:
|
||||||
return True
|
return True
|
||||||
|
|
||||||
async def _get_backend(self) -> Any:
|
def _make_backend(self) -> Any:
|
||||||
if self._backend is not None:
|
|
||||||
return self._backend
|
|
||||||
from nanobot.agent.tools.computer_use_backends.browser_playwright import BrowserBackend
|
from nanobot.agent.tools.computer_use_backends.browser_playwright import BrowserBackend
|
||||||
self._backend = BrowserBackend(
|
return BrowserBackend(
|
||||||
width=self.width,
|
width=self.config.width,
|
||||||
height=self.height,
|
height=self.config.height,
|
||||||
headless=self.headless,
|
start_url=self.config.start_url,
|
||||||
start_url=self.start_url,
|
allowed_domains=self.config.allowed_domains,
|
||||||
|
runtime=self._runtime,
|
||||||
)
|
)
|
||||||
return self._backend
|
|
||||||
|
|
||||||
def _domain_allowed(self, url: str) -> bool:
|
|
||||||
if not self.allowed_domains:
|
|
||||||
return True
|
|
||||||
host = (urlparse(url).hostname or "").lower()
|
|
||||||
if not host:
|
|
||||||
return False
|
|
||||||
for dom in self.allowed_domains:
|
|
||||||
d = dom.lower().lstrip(".")
|
|
||||||
if d and (host == d or host.endswith("." + d)):
|
|
||||||
return True
|
|
||||||
return False
|
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _req_ref(params: dict[str, Any], action: str) -> Any:
|
def _req_ref(params: dict[str, Any], action: str) -> Any:
|
||||||
@@ -211,8 +172,6 @@ class BrowserTool(Tool):
|
|||||||
url = p.get("url")
|
url = p.get("url")
|
||||||
if not url:
|
if not url:
|
||||||
raise ValueError("action 'navigate' requires 'url'")
|
raise ValueError("action 'navigate' requires 'url'")
|
||||||
if not self._domain_allowed(str(url)):
|
|
||||||
raise ValueError(f"navigation to '{url}' is blocked by the allowed_domains policy")
|
|
||||||
await backend.navigate(str(url))
|
await backend.navigate(str(url))
|
||||||
return f"Navigated to {url}", None
|
return f"Navigated to {url}", None
|
||||||
|
|
||||||
@@ -271,7 +230,7 @@ class BrowserTool(Tool):
|
|||||||
return f"Error: unknown action '{action}'. Valid actions: {', '.join(_ACTIONS)}"
|
return f"Error: unknown action '{action}'. Valid actions: {', '.join(_ACTIONS)}"
|
||||||
|
|
||||||
try:
|
try:
|
||||||
backend = await self._get_backend()
|
backend = self._backends.get()
|
||||||
except ImportError as exc:
|
except ImportError as exc:
|
||||||
return f"Error: {exc}"
|
return f"Error: {exc}"
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -279,6 +238,8 @@ class BrowserTool(Tool):
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
status, direct = await self._dispatch(backend, action, kwargs)
|
status, direct = await self._dispatch(backend, action, kwargs)
|
||||||
|
if blocked := getattr(backend, "pop_blocked_navigation", lambda: None)():
|
||||||
|
raise ValueError(f"navigation was blocked: {blocked}")
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
return f"Error: {exc}"
|
return f"Error: {exc}"
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -288,7 +249,7 @@ class BrowserTool(Tool):
|
|||||||
return direct
|
return direct
|
||||||
|
|
||||||
try:
|
try:
|
||||||
elements = await backend.dom_snapshot(self.max_elements)
|
elements = await backend.dom_snapshot(self.config.max_elements)
|
||||||
snapshot = _format_elements(elements)
|
snapshot = _format_elements(elements)
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
snapshot = f"(could not read page elements: {type(exc).__name__}: {exc})"
|
snapshot = f"(could not read page elements: {type(exc).__name__}: {exc})"
|
||||||
@@ -299,7 +260,7 @@ class BrowserTool(Tool):
|
|||||||
header = f"{status}\nCurrent page: {current}" if current else status
|
header = f"{status}\nCurrent page: {current}" if current else status
|
||||||
text_out = f"{header}\n\n{snapshot}"
|
text_out = f"{header}\n\n{snapshot}"
|
||||||
|
|
||||||
if self.include_screenshot:
|
if self.config.include_screenshot:
|
||||||
try:
|
try:
|
||||||
png = await backend.screenshot()
|
png = await backend.screenshot()
|
||||||
return build_image_content_blocks(png, "image/png", "", text_out)
|
return build_image_content_blocks(png, "image/png", "", text_out)
|
||||||
@@ -308,5 +269,4 @@ class BrowserTool(Tool):
|
|||||||
return text_out
|
return text_out
|
||||||
|
|
||||||
async def close(self) -> None:
|
async def close(self) -> None:
|
||||||
if self._backend is not None:
|
await self._backends.close()
|
||||||
await self._backend.close()
|
|
||||||
|
|||||||
@@ -1,38 +1,26 @@
|
|||||||
"""computer_use tool: control a desktop or browser via screenshots + mouse/keyboard.
|
"""Screenshot-based computer control."""
|
||||||
|
|
||||||
Model-agnostic by design. The tool returns each screenshot as ``image_url``
|
# pyright: reportIncompatibleMethodOverride=false
|
||||||
content blocks; the runner delivers those to the model as a follow-up user
|
|
||||||
message, so any vision + tool-calling model works (Claude, GPT, Gemini, ... via
|
|
||||||
any gateway such as OpenRouter) without provider-specific plumbing.
|
|
||||||
|
|
||||||
Backends (selected via config) are pluggable:
|
|
||||||
- ``desktop`` — PyAutoGUI, controls the local GUI (Codex-style).
|
|
||||||
- ``browser`` — Playwright, controls a headless web page (also supports navigate).
|
|
||||||
|
|
||||||
Both heavy deps are optional and imported lazily, so importing this module at
|
|
||||||
tool auto-discovery time is cheap and never requires pyautogui/playwright.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
import io
|
import io
|
||||||
from typing import Any
|
from typing import Any, Literal
|
||||||
from urllib.parse import urlparse
|
|
||||||
|
|
||||||
from loguru import logger
|
|
||||||
from pydantic import Field
|
from pydantic import Field
|
||||||
|
|
||||||
from nanobot.agent.tools.base import Tool, tool_parameters
|
from nanobot.agent.tools.base import Tool, tool_parameters
|
||||||
|
from nanobot.agent.tools.computer_use_backends.base import SessionBackendPool
|
||||||
|
from nanobot.agent.tools.context import ToolContext
|
||||||
from nanobot.agent.tools.schema import (
|
from nanobot.agent.tools.schema import (
|
||||||
IntegerSchema,
|
IntegerSchema,
|
||||||
NumberSchema,
|
NumberSchema,
|
||||||
StringSchema,
|
StringSchema,
|
||||||
tool_parameters_schema,
|
tool_parameters_schema,
|
||||||
)
|
)
|
||||||
from nanobot.config.schema import Base
|
from nanobot.config_base import Base
|
||||||
from nanobot.utils.helpers import build_image_content_blocks
|
from nanobot.utils.helpers import build_image_content_blocks
|
||||||
from nanobot.utils.screen_scale import ScreenScaler, fit_target_size
|
|
||||||
|
|
||||||
_ACTIONS = [
|
_ACTIONS = [
|
||||||
"screenshot",
|
"screenshot",
|
||||||
@@ -60,20 +48,41 @@ _CLICK_BUTTONS = {
|
|||||||
_CLICK_COUNTS = {"double_click": 2, "triple_click": 3}
|
_CLICK_COUNTS = {"double_click": 2, "triple_click": 3}
|
||||||
|
|
||||||
_MAX_WAIT_S = 10.0
|
_MAX_WAIT_S = 10.0
|
||||||
_ENABLE_WARNED = False # log the security warning at most once per process
|
|
||||||
|
|
||||||
|
|
||||||
class ComputerUseToolConfig(Base):
|
class ComputerUseToolConfig(Base):
|
||||||
"""computer_use tool configuration."""
|
"""computer_use tool configuration."""
|
||||||
|
|
||||||
enable: bool = False # off by default — security-sensitive, opt-in
|
enable: bool = False
|
||||||
backend: str = "desktop" # "desktop" | "browser"
|
backend: Literal["desktop", "browser"] = "desktop"
|
||||||
target_width: int = 1280 # screenshot is downscaled to fit this box (model space)
|
target_width: int = Field(default=1280, ge=320, le=4096)
|
||||||
target_height: int = 800
|
target_height: int = Field(default=800, ge=240, le=4096)
|
||||||
require_approval: bool = True # gate destructive actions (enforced by the agent layer)
|
allowed_domains: list[str] = Field(default_factory=list)
|
||||||
allowed_domains: list[str] = Field(default_factory=list) # browser allowlist; empty = all
|
start_url: str = "about:blank"
|
||||||
start_url: str = "about:blank" # browser initial page
|
headless: bool = True
|
||||||
headless: bool = True # browser headless mode
|
|
||||||
|
|
||||||
|
def _fit_size(width: int, height: int, max_width: int, max_height: int) -> tuple[int, int]:
|
||||||
|
if width <= 0 or height <= 0:
|
||||||
|
return max(1, max_width), max(1, max_height)
|
||||||
|
scale = min(max_width / width, max_height / height, 1.0)
|
||||||
|
return max(1, round(width * scale)), max(1, round(height * scale))
|
||||||
|
|
||||||
|
|
||||||
|
def _scale_point(
|
||||||
|
x: int,
|
||||||
|
y: int,
|
||||||
|
source: tuple[int, int],
|
||||||
|
target: tuple[int, int],
|
||||||
|
) -> tuple[int, int]:
|
||||||
|
width, height = source
|
||||||
|
target_width, target_height = target
|
||||||
|
real_x = round(x * width / target_width) if target_width else x
|
||||||
|
real_y = round(y * height / target_height) if target_height else y
|
||||||
|
return (
|
||||||
|
max(0, min(real_x, max(0, width - 1))),
|
||||||
|
max(0, min(real_y, max(0, height - 1))),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@tool_parameters(
|
@tool_parameters(
|
||||||
@@ -95,9 +104,17 @@ class ComputerUseToolConfig(Base):
|
|||||||
"Scroll direction (action=scroll).", enum=["up", "down", "left", "right"], nullable=True
|
"Scroll direction (action=scroll).", enum=["up", "down", "left", "right"], nullable=True
|
||||||
),
|
),
|
||||||
scroll_amount=IntegerSchema(
|
scroll_amount=IntegerSchema(
|
||||||
description="Number of scroll clicks (action=scroll).", nullable=True
|
description="Number of scroll clicks (action=scroll).",
|
||||||
|
minimum=1,
|
||||||
|
maximum=100,
|
||||||
|
nullable=True,
|
||||||
|
),
|
||||||
|
duration=NumberSchema(
|
||||||
|
description="Seconds to wait (action=wait).",
|
||||||
|
minimum=0,
|
||||||
|
maximum=_MAX_WAIT_S,
|
||||||
|
nullable=True,
|
||||||
),
|
),
|
||||||
duration=NumberSchema(description="Seconds to wait (action=wait).", nullable=True),
|
|
||||||
url=StringSchema("URL to open (action=navigate, browser backend only).", nullable=True),
|
url=StringSchema("URL to open (action=navigate, browser backend only).", nullable=True),
|
||||||
required=["action"],
|
required=["action"],
|
||||||
)
|
)
|
||||||
@@ -107,8 +124,8 @@ class ComputerUseTool(Tool):
|
|||||||
|
|
||||||
_scopes = {"core"} # never exposed to subagents — security-sensitive
|
_scopes = {"core"} # never exposed to subagents — security-sensitive
|
||||||
|
|
||||||
name = "computer_use"
|
name = "computer_use" # pyright: ignore[reportIncompatibleMethodOverride, reportAssignmentType]
|
||||||
description = (
|
description = ( # pyright: ignore[reportIncompatibleMethodOverride, reportAssignmentType]
|
||||||
"Control a computer via screenshots and mouse/keyboard. Each call performs ONE "
|
"Control a computer via screenshots and mouse/keyboard. Each call performs ONE "
|
||||||
"action and returns a fresh screenshot of the resulting screen. Coordinates (x, y) "
|
"action and returns a fresh screenshot of the resulting screen. Coordinates (x, y) "
|
||||||
"are in the pixel space of the screenshot you were last shown (top-left is 0,0). "
|
"are in the pixel space of the screenshot you were last shown (top-left is 0,0). "
|
||||||
@@ -124,53 +141,30 @@ class ComputerUseTool(Tool):
|
|||||||
return ComputerUseToolConfig
|
return ComputerUseToolConfig
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def enabled(cls, ctx: Any) -> bool:
|
def enabled(cls, ctx: ToolContext) -> bool:
|
||||||
return bool(ctx.config.computer_use.enable)
|
return bool(ctx.config.computer_use.enable)
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def create(cls, ctx: Any) -> Tool:
|
def create(cls, ctx: ToolContext) -> Tool:
|
||||||
cfg = ctx.config.computer_use
|
return cls(ctx.config.computer_use)
|
||||||
global _ENABLE_WARNED
|
|
||||||
if not _ENABLE_WARNED:
|
|
||||||
logger.warning(
|
|
||||||
"computer_use tool is ENABLED (backend={}). It can control the real "
|
|
||||||
"{} and affect state outside the workspace. Run nanobot in a sandbox/VM, "
|
|
||||||
"restrict to trusted models/inputs, and beware prompt injection from "
|
|
||||||
"on-screen/web content.",
|
|
||||||
cfg.backend,
|
|
||||||
"browser" if cfg.backend == "browser" else "desktop",
|
|
||||||
)
|
|
||||||
_ENABLE_WARNED = True
|
|
||||||
return cls(
|
|
||||||
backend=cfg.backend,
|
|
||||||
target_width=cfg.target_width,
|
|
||||||
target_height=cfg.target_height,
|
|
||||||
require_approval=cfg.require_approval,
|
|
||||||
allowed_domains=list(cfg.allowed_domains),
|
|
||||||
start_url=cfg.start_url,
|
|
||||||
headless=cfg.headless,
|
|
||||||
)
|
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
|
config: ComputerUseToolConfig | None = None,
|
||||||
*,
|
*,
|
||||||
backend: str = "desktop",
|
|
||||||
target_width: int = 1280,
|
|
||||||
target_height: int = 800,
|
|
||||||
require_approval: bool = True,
|
|
||||||
allowed_domains: list[str] | None = None,
|
|
||||||
start_url: str = "about:blank",
|
|
||||||
headless: bool = True,
|
|
||||||
backend_impl: Any = None,
|
backend_impl: Any = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
self.backend_name = backend
|
self.config = config or ComputerUseToolConfig()
|
||||||
self.target_width = target_width
|
runtime = None
|
||||||
self.target_height = target_height
|
if backend_impl is None and self.config.backend == "browser":
|
||||||
self.require_approval = require_approval
|
from nanobot.agent.tools.computer_use_backends.browser_playwright import BrowserRuntime
|
||||||
self.allowed_domains = allowed_domains or []
|
runtime = BrowserRuntime(headless=self.config.headless)
|
||||||
self.start_url = start_url
|
self._runtime = runtime
|
||||||
self.headless = headless
|
self._backends = SessionBackendPool(
|
||||||
self._backend = backend_impl # injectable for tests
|
self._make_backend,
|
||||||
|
backend_impl,
|
||||||
|
finalizer=runtime.close if runtime is not None else None,
|
||||||
|
)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def read_only(self) -> bool:
|
def read_only(self) -> bool:
|
||||||
@@ -181,21 +175,18 @@ class ComputerUseTool(Tool):
|
|||||||
# Stateful single environment; must not run alongside other tools.
|
# Stateful single environment; must not run alongside other tools.
|
||||||
return True
|
return True
|
||||||
|
|
||||||
async def _get_backend(self) -> Any:
|
def _make_backend(self) -> Any:
|
||||||
if self._backend is not None:
|
if self.config.backend == "browser":
|
||||||
return self._backend
|
|
||||||
if self.backend_name == "browser":
|
|
||||||
from nanobot.agent.tools.computer_use_backends.browser_playwright import BrowserBackend
|
from nanobot.agent.tools.computer_use_backends.browser_playwright import BrowserBackend
|
||||||
self._backend = BrowserBackend(
|
return BrowserBackend(
|
||||||
width=self.target_width,
|
width=self.config.target_width,
|
||||||
height=self.target_height,
|
height=self.config.target_height,
|
||||||
headless=self.headless,
|
start_url=self.config.start_url,
|
||||||
start_url=self.start_url,
|
allowed_domains=self.config.allowed_domains,
|
||||||
|
runtime=self._runtime,
|
||||||
)
|
)
|
||||||
else:
|
from nanobot.agent.tools.computer_use_backends.desktop_pyautogui import DesktopBackend
|
||||||
from nanobot.agent.tools.computer_use_backends.desktop_pyautogui import DesktopBackend
|
return DesktopBackend()
|
||||||
self._backend = DesktopBackend()
|
|
||||||
return self._backend
|
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _downscale_png(png: bytes, target: tuple[int, int]) -> bytes:
|
def _downscale_png(png: bytes, target: tuple[int, int]) -> bytes:
|
||||||
@@ -209,45 +200,31 @@ class ComputerUseTool(Tool):
|
|||||||
with Image.open(io.BytesIO(png)) as img:
|
with Image.open(io.BytesIO(png)) as img:
|
||||||
if (img.width, img.height) == (tw, th):
|
if (img.width, img.height) == (tw, th):
|
||||||
return png
|
return png
|
||||||
resized = img.convert("RGB").resize((tw, th))
|
resized = img.convert("RGB").resize((tw, th)) # pyright: ignore[reportUnknownMemberType]
|
||||||
out = io.BytesIO()
|
out = io.BytesIO()
|
||||||
resized.save(out, format="PNG")
|
resized.save(out, format="PNG")
|
||||||
return out.getvalue()
|
return out.getvalue()
|
||||||
|
|
||||||
def _domain_allowed(self, url: str) -> bool:
|
async def _dispatch(
|
||||||
"""Check a navigation URL against the browser allowed_domains policy.
|
self,
|
||||||
|
backend: Any,
|
||||||
Empty allowlist = allow all. A domain entry matches the exact host or any
|
action: str,
|
||||||
subdomain of it (``example.com`` allows ``app.example.com``).
|
params: dict[str, Any],
|
||||||
"""
|
source: tuple[int, int],
|
||||||
if not self.allowed_domains:
|
target: tuple[int, int],
|
||||||
return True
|
) -> str:
|
||||||
host = (urlparse(url).hostname or "").lower()
|
|
||||||
if not host:
|
|
||||||
return False
|
|
||||||
for dom in self.allowed_domains:
|
|
||||||
d = dom.lower().lstrip(".")
|
|
||||||
if d and (host == d or host.endswith("." + d)):
|
|
||||||
return True
|
|
||||||
return False
|
|
||||||
|
|
||||||
async def _dispatch(self, backend: Any, scaler: ScreenScaler, action: str, params: dict[str, Any]) -> str:
|
|
||||||
def _xy() -> tuple[int, int]:
|
def _xy() -> tuple[int, int]:
|
||||||
x, y = params.get("x"), params.get("y")
|
x, y = params.get("x"), params.get("y")
|
||||||
# Some models emit a combined [x, y] array in the x field (this is
|
|
||||||
# Anthropic's native ``coordinate`` convention). Accept that too so
|
|
||||||
# the tool is not tied to one provider's calling style.
|
|
||||||
if isinstance(x, (list, tuple)) and len(x) == 2 and y is None:
|
|
||||||
x, y = x[0], x[1]
|
|
||||||
if x is None or y is None:
|
if x is None or y is None:
|
||||||
raise ValueError(f"action '{action}' requires integer 'x' and 'y'")
|
raise ValueError(f"action '{action}' requires integer 'x' and 'y'")
|
||||||
return scaler.to_real(int(x), int(y))
|
return _scale_point(int(x), int(y), source, target)
|
||||||
|
|
||||||
if action == "screenshot":
|
if action == "screenshot":
|
||||||
return "Took a screenshot"
|
return "Took a screenshot"
|
||||||
|
|
||||||
if action == "wait":
|
if action == "wait":
|
||||||
secs = float(params.get("duration") or 1.0)
|
duration = params.get("duration")
|
||||||
|
secs = 1.0 if duration is None else float(duration)
|
||||||
secs = max(0.0, min(secs, _MAX_WAIT_S))
|
secs = max(0.0, min(secs, _MAX_WAIT_S))
|
||||||
await asyncio.sleep(secs)
|
await asyncio.sleep(secs)
|
||||||
return f"Waited {secs:g}s"
|
return f"Waited {secs:g}s"
|
||||||
@@ -294,10 +271,6 @@ class ComputerUseTool(Tool):
|
|||||||
url = params.get("url")
|
url = params.get("url")
|
||||||
if not url:
|
if not url:
|
||||||
raise ValueError("action 'navigate' requires 'url'")
|
raise ValueError("action 'navigate' requires 'url'")
|
||||||
if not self._domain_allowed(str(url)):
|
|
||||||
raise ValueError(
|
|
||||||
f"navigation to '{url}' is blocked by the allowed_domains policy"
|
|
||||||
)
|
|
||||||
await backend.navigate(str(url))
|
await backend.navigate(str(url))
|
||||||
return f"Navigated to {url}"
|
return f"Navigated to {url}"
|
||||||
|
|
||||||
@@ -309,17 +282,20 @@ class ComputerUseTool(Tool):
|
|||||||
return f"Error: unknown action '{action}'. Valid actions: {', '.join(_ACTIONS)}"
|
return f"Error: unknown action '{action}'. Valid actions: {', '.join(_ACTIONS)}"
|
||||||
|
|
||||||
try:
|
try:
|
||||||
backend = await self._get_backend()
|
backend = self._backends.get()
|
||||||
real_w, real_h = await backend.dimensions()
|
real_w, real_h = await backend.dimensions()
|
||||||
except ImportError as exc:
|
except ImportError as exc:
|
||||||
return f"Error: {exc}"
|
return f"Error: {exc}"
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
return f"Error: could not initialize computer_use backend: {type(exc).__name__}: {exc}"
|
return f"Error: could not initialize computer_use backend: {type(exc).__name__}: {exc}"
|
||||||
|
|
||||||
scaler = ScreenScaler.for_screen(real_w, real_h, self.target_width, self.target_height)
|
source = (real_w, real_h)
|
||||||
|
target = _fit_size(real_w, real_h, self.config.target_width, self.config.target_height)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
status = await self._dispatch(backend, scaler, action, kwargs)
|
status = await self._dispatch(backend, action, kwargs, source, target)
|
||||||
|
if blocked := getattr(backend, "pop_blocked_navigation", lambda: None)():
|
||||||
|
raise ValueError(f"navigation was blocked: {blocked}")
|
||||||
except ValueError as exc:
|
except ValueError as exc:
|
||||||
return f"Error: {exc}"
|
return f"Error: {exc}"
|
||||||
except NotImplementedError as exc:
|
except NotImplementedError as exc:
|
||||||
@@ -330,7 +306,6 @@ class ComputerUseTool(Tool):
|
|||||||
# Return a fresh screenshot so the model sees the result of its action.
|
# Return a fresh screenshot so the model sees the result of its action.
|
||||||
try:
|
try:
|
||||||
png = await backend.screenshot()
|
png = await backend.screenshot()
|
||||||
target = fit_target_size(real_w, real_h, self.target_width, self.target_height)
|
|
||||||
png = self._downscale_png(png, target)
|
png = self._downscale_png(png, target)
|
||||||
except ImportError as exc:
|
except ImportError as exc:
|
||||||
return f"Error: {exc}"
|
return f"Error: {exc}"
|
||||||
@@ -341,5 +316,4 @@ class ComputerUseTool(Tool):
|
|||||||
return build_image_content_blocks(png, "image/png", "", label)
|
return build_image_content_blocks(png, "image/png", "", label)
|
||||||
|
|
||||||
async def close(self) -> None:
|
async def close(self) -> None:
|
||||||
if self._backend is not None:
|
await self._backends.close()
|
||||||
await self._backend.close()
|
|
||||||
|
|||||||
@@ -1,11 +1 @@
|
|||||||
"""Pluggable execution backends for the ``computer_use`` tool.
|
"""Computer-use backend adapters."""
|
||||||
|
|
||||||
Backends are plain classes (NOT ``Tool`` subclasses), so ``ToolLoader`` never
|
|
||||||
mistakes them for tools. Importing this package must stay cheap: the heavy,
|
|
||||||
optional dependencies (pyautogui / Pillow / playwright) are imported lazily
|
|
||||||
inside the concrete backend modules and never at import time here.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from nanobot.agent.tools.computer_use_backends.base import ComputerBackend
|
|
||||||
|
|
||||||
__all__ = ["ComputerBackend"]
|
|
||||||
|
|||||||
@@ -13,7 +13,12 @@ backends never deal with the downscaled space.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
|
from collections.abc import Awaitable, Callable
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from nanobot.agent.tools.context import current_request_session_key
|
||||||
|
|
||||||
|
|
||||||
class ComputerBackend(ABC):
|
class ComputerBackend(ABC):
|
||||||
@@ -63,3 +68,48 @@ class ComputerBackend(ABC):
|
|||||||
async def close(self) -> None:
|
async def close(self) -> None:
|
||||||
"""Release any resources (browser process, etc.). Safe to call repeatedly."""
|
"""Release any resources (browser process, etc.). Safe to call repeatedly."""
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
class SessionBackendPool:
|
||||||
|
"""Keep stateful backends isolated by nanobot session."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
factory: Callable[[], Any],
|
||||||
|
injected: Any = None,
|
||||||
|
*,
|
||||||
|
finalizer: Callable[[], Awaitable[None]] | None = None,
|
||||||
|
) -> None:
|
||||||
|
self._factory = factory
|
||||||
|
self._injected = injected
|
||||||
|
self._finalizer = finalizer
|
||||||
|
self._backends: dict[str, Any] = {}
|
||||||
|
|
||||||
|
def get(self) -> Any:
|
||||||
|
if self._injected is not None:
|
||||||
|
return self._injected
|
||||||
|
key = current_request_session_key() or "default"
|
||||||
|
backend = self._backends.get(key)
|
||||||
|
if backend is None:
|
||||||
|
backend = self._backends[key] = self._factory()
|
||||||
|
return backend
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
backends = [self._injected] if self._injected is not None else list(self._backends.values())
|
||||||
|
self._injected = None
|
||||||
|
self._backends.clear()
|
||||||
|
results = await asyncio.gather(
|
||||||
|
*(backend.close() for backend in backends if backend is not None),
|
||||||
|
return_exceptions=True,
|
||||||
|
)
|
||||||
|
errors = [result for result in results if isinstance(result, BaseException)]
|
||||||
|
if self._finalizer is not None:
|
||||||
|
try:
|
||||||
|
await self._finalizer()
|
||||||
|
except BaseException as exc:
|
||||||
|
errors.append(exc)
|
||||||
|
self._finalizer = None
|
||||||
|
if len(errors) == 1:
|
||||||
|
raise errors[0]
|
||||||
|
if errors:
|
||||||
|
raise BaseExceptionGroup("failed to close computer-use backends", errors)
|
||||||
|
|||||||
@@ -1,17 +1,17 @@
|
|||||||
"""Browser backend using Playwright (DOM-aware, deterministic, sandbox-friendly).
|
"""Playwright backend shared by browser and computer_use."""
|
||||||
|
|
||||||
Drives a headless Chromium page via pixel coordinates + screenshots, mirroring
|
|
||||||
the desktop backend's surface so the same model/tool loop works against the web.
|
|
||||||
The viewport uses ``deviceScaleFactor=1`` so screenshot pixels == mouse
|
|
||||||
coordinates (no HiDPI scaling to undo). The browser/page persist across actions
|
|
||||||
within one tool instance so navigation and state carry between turns.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from typing import Any
|
import asyncio
|
||||||
|
import importlib
|
||||||
|
from collections.abc import Sequence
|
||||||
|
from typing import Any, cast
|
||||||
|
from urllib.parse import urlparse, urlunparse
|
||||||
|
|
||||||
|
from loguru import logger
|
||||||
|
|
||||||
from nanobot.agent.tools.computer_use_backends.base import ComputerBackend
|
from nanobot.agent.tools.computer_use_backends.base import ComputerBackend
|
||||||
|
from nanobot.security.network import validate_url_target
|
||||||
|
|
||||||
_MISSING = (
|
_MISSING = (
|
||||||
"Browser computer-use backend needs 'playwright'. Install with: "
|
"Browser computer-use backend needs 'playwright'. Install with: "
|
||||||
@@ -70,6 +70,36 @@ _KEYS = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_browser_url(
|
||||||
|
url: str,
|
||||||
|
allowed_domains: Sequence[str] = (),
|
||||||
|
*,
|
||||||
|
navigation: bool = True,
|
||||||
|
) -> tuple[bool, str]:
|
||||||
|
if url == "about:blank":
|
||||||
|
return True, ""
|
||||||
|
|
||||||
|
parsed = urlparse(url)
|
||||||
|
if not navigation and parsed.scheme in {"blob", "data"}:
|
||||||
|
return True, ""
|
||||||
|
|
||||||
|
target = url
|
||||||
|
if parsed.scheme in {"ws", "wss"}:
|
||||||
|
target = urlunparse(parsed._replace(scheme="https" if parsed.scheme == "wss" else "http"))
|
||||||
|
|
||||||
|
if navigation and allowed_domains:
|
||||||
|
host = (parsed.hostname or "").rstrip(".").lower()
|
||||||
|
allowed = any(
|
||||||
|
normalized and (host == normalized or host.endswith(f".{normalized}"))
|
||||||
|
for domain in allowed_domains
|
||||||
|
if (normalized := domain.strip().lstrip(".").rstrip(".").lower())
|
||||||
|
)
|
||||||
|
if not allowed:
|
||||||
|
return False, f"host {host or '<missing>'} is not in allowed_domains"
|
||||||
|
|
||||||
|
return validate_url_target(target)
|
||||||
|
|
||||||
|
|
||||||
def _playwright_key(combo: str) -> str:
|
def _playwright_key(combo: str) -> str:
|
||||||
parts = [p.strip() for p in combo.split("+") if p.strip()]
|
parts = [p.strip() for p in combo.split("+") if p.strip()]
|
||||||
out: list[str] = []
|
out: list[str] = []
|
||||||
@@ -86,6 +116,57 @@ def _playwright_key(combo: str) -> str:
|
|||||||
return "+".join(out)
|
return "+".join(out)
|
||||||
|
|
||||||
|
|
||||||
|
class BrowserRuntime:
|
||||||
|
"""One lazily started browser process shared by isolated session contexts."""
|
||||||
|
|
||||||
|
def __init__(self, *, headless: bool = True) -> None:
|
||||||
|
self._headless = headless
|
||||||
|
self._lock = asyncio.Lock()
|
||||||
|
self._playwright: Any = None
|
||||||
|
self._browser: Any = None
|
||||||
|
|
||||||
|
async def get(self) -> Any:
|
||||||
|
if self._browser is not None:
|
||||||
|
return self._browser
|
||||||
|
async with self._lock:
|
||||||
|
if self._browser is not None:
|
||||||
|
return self._browser
|
||||||
|
try:
|
||||||
|
playwright = importlib.import_module("playwright.async_api")
|
||||||
|
async_playwright = cast(Any, playwright).async_playwright
|
||||||
|
except ImportError as exc:
|
||||||
|
raise ImportError(_MISSING) from exc
|
||||||
|
self._playwright = await async_playwright().start()
|
||||||
|
try:
|
||||||
|
self._browser = await self._playwright.chromium.launch(
|
||||||
|
headless=self._headless
|
||||||
|
)
|
||||||
|
except BaseException:
|
||||||
|
await self.close()
|
||||||
|
raise
|
||||||
|
return self._browser
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
browser, playwright = self._browser, self._playwright
|
||||||
|
self._browser = self._playwright = None
|
||||||
|
errors: list[BaseException] = []
|
||||||
|
closers = (
|
||||||
|
browser.close if browser is not None else None,
|
||||||
|
playwright.stop if playwright is not None else None,
|
||||||
|
)
|
||||||
|
for close in closers:
|
||||||
|
if close is None:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
await close()
|
||||||
|
except BaseException as exc:
|
||||||
|
errors.append(exc)
|
||||||
|
if len(errors) == 1:
|
||||||
|
raise errors[0]
|
||||||
|
if errors:
|
||||||
|
raise BaseExceptionGroup("failed to close browser runtime", errors)
|
||||||
|
|
||||||
|
|
||||||
class BrowserBackend(ComputerBackend):
|
class BrowserBackend(ComputerBackend):
|
||||||
environment = "browser"
|
environment = "browser"
|
||||||
|
|
||||||
@@ -96,33 +177,84 @@ class BrowserBackend(ComputerBackend):
|
|||||||
height: int = 800,
|
height: int = 800,
|
||||||
headless: bool = True,
|
headless: bool = True,
|
||||||
start_url: str = "about:blank",
|
start_url: str = "about:blank",
|
||||||
|
allowed_domains: Sequence[str] = (),
|
||||||
|
runtime: BrowserRuntime | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
self._width = width
|
self._width = width
|
||||||
self._height = height
|
self._height = height
|
||||||
self._headless = headless
|
|
||||||
self._start_url = start_url
|
self._start_url = start_url
|
||||||
self._pw: Any = None
|
self._allowed_domains = tuple(allowed_domains)
|
||||||
self._browser: Any = None
|
self._runtime = runtime or BrowserRuntime(headless=headless)
|
||||||
|
self._owns_runtime = runtime is None
|
||||||
|
self._context: Any = None
|
||||||
self._page: Any = None
|
self._page: Any = None
|
||||||
self._last_pos = (0, 0)
|
self._last_pos = (0, 0)
|
||||||
|
self._blocked_navigation: str | None = None
|
||||||
|
|
||||||
|
async def _require_url(self, url: str, label: str) -> None:
|
||||||
|
ok, error = await asyncio.to_thread(
|
||||||
|
_validate_browser_url,
|
||||||
|
url,
|
||||||
|
self._allowed_domains,
|
||||||
|
)
|
||||||
|
if not ok:
|
||||||
|
raise ValueError(f"{label} is blocked: {error}")
|
||||||
|
|
||||||
|
async def _route_request(self, route: Any) -> None:
|
||||||
|
request = route.request
|
||||||
|
navigation = bool(request.is_navigation_request())
|
||||||
|
ok, error = await asyncio.to_thread(
|
||||||
|
_validate_browser_url,
|
||||||
|
request.url,
|
||||||
|
self._allowed_domains,
|
||||||
|
navigation=navigation,
|
||||||
|
)
|
||||||
|
if ok:
|
||||||
|
await route.continue_()
|
||||||
|
return
|
||||||
|
if navigation:
|
||||||
|
self._blocked_navigation = error
|
||||||
|
logger.warning("Blocked browser request to {}: {}", request.url, error)
|
||||||
|
await route.abort("blockedbyclient")
|
||||||
|
|
||||||
|
async def _route_web_socket(self, web_socket: Any) -> None:
|
||||||
|
ok, error = await asyncio.to_thread(
|
||||||
|
_validate_browser_url,
|
||||||
|
web_socket.url,
|
||||||
|
self._allowed_domains,
|
||||||
|
navigation=False,
|
||||||
|
)
|
||||||
|
if not ok:
|
||||||
|
logger.warning("Blocked browser WebSocket to {}: {}", web_socket.url, error)
|
||||||
|
await web_socket.close(code=1008, reason="Blocked by nanobot network policy")
|
||||||
|
return
|
||||||
|
await web_socket.connect_to_server()
|
||||||
|
|
||||||
|
def pop_blocked_navigation(self) -> str | None:
|
||||||
|
error = self._blocked_navigation
|
||||||
|
self._blocked_navigation = None
|
||||||
|
return error
|
||||||
|
|
||||||
async def _ensure(self) -> Any:
|
async def _ensure(self) -> Any:
|
||||||
if self._page is not None:
|
if self._page is not None:
|
||||||
return self._page
|
return self._page
|
||||||
|
await self._require_url(self._start_url, "start_url")
|
||||||
try:
|
try:
|
||||||
from playwright.async_api import async_playwright # noqa: PLC0415
|
browser = await self._runtime.get()
|
||||||
except Exception as exc:
|
self._context = await browser.new_context(
|
||||||
raise ImportError(_MISSING) from exc
|
viewport={"width": self._width, "height": self._height},
|
||||||
self._pw = await async_playwright().start()
|
device_scale_factor=1,
|
||||||
self._browser = await self._pw.chromium.launch(headless=self._headless)
|
service_workers="block",
|
||||||
context = await self._browser.new_context(
|
)
|
||||||
viewport={"width": self._width, "height": self._height},
|
await self._context.route("**/*", self._route_request)
|
||||||
device_scale_factor=1,
|
await self._context.route_web_socket("**/*", self._route_web_socket)
|
||||||
)
|
self._page = await self._context.new_page()
|
||||||
self._page = await context.new_page()
|
if self._start_url != "about:blank":
|
||||||
if self._start_url and self._start_url != "about:blank":
|
await self._page.goto(self._start_url)
|
||||||
await self._page.goto(self._start_url)
|
return self._page
|
||||||
return self._page
|
except BaseException:
|
||||||
|
await self.close()
|
||||||
|
raise
|
||||||
|
|
||||||
async def dimensions(self) -> tuple[int, int]:
|
async def dimensions(self) -> tuple[int, int]:
|
||||||
await self._ensure()
|
await self._ensure()
|
||||||
@@ -171,36 +303,37 @@ class BrowserBackend(ComputerBackend):
|
|||||||
await page.keyboard.press(key)
|
await page.keyboard.press(key)
|
||||||
|
|
||||||
async def navigate(self, url: str) -> None:
|
async def navigate(self, url: str) -> None:
|
||||||
|
await self._require_url(url, "navigation")
|
||||||
page = await self._ensure()
|
page = await self._ensure()
|
||||||
await page.goto(url)
|
await page.goto(url)
|
||||||
self._last_pos = (0, 0)
|
self._last_pos = (0, 0)
|
||||||
|
|
||||||
# --- DOM / accessibility mode (act by element ref, not pixels) ---
|
# --- DOM / accessibility mode (act by element ref, not pixels) ---
|
||||||
|
|
||||||
async def dom_snapshot(self, max_elements: int = 200) -> list[dict]:
|
async def dom_snapshot(self, max_elements: int = 200) -> list[dict[str, Any]]:
|
||||||
"""Tag visible interactive elements with ``data-nanobot-ref`` and return them.
|
"""Tag visible interactive elements with ``data-nanobot-ref`` and return them.
|
||||||
|
|
||||||
Each entry: ``{ref, tag, role, type, name, href}``. Refs are reassigned on
|
Each entry: ``{ref, tag, role, type, name, href}``. Refs are reassigned on
|
||||||
every snapshot, so callers should act on the latest snapshot.
|
every snapshot, so callers should act on the latest snapshot.
|
||||||
"""
|
"""
|
||||||
page = await self._ensure()
|
page = await self._ensure()
|
||||||
return await page.evaluate(_SNAPSHOT_JS, max_elements)
|
return cast(list[dict[str, Any]], await page.evaluate(_SNAPSHOT_JS, max_elements))
|
||||||
|
|
||||||
def _ref_selector(self, ref) -> str:
|
def _ref_selector(self, ref: int) -> str:
|
||||||
return f'[data-nanobot-ref="{int(ref)}"]'
|
return f'[data-nanobot-ref="{int(ref)}"]'
|
||||||
|
|
||||||
async def click_ref(self, ref) -> None:
|
async def click_ref(self, ref: int) -> None:
|
||||||
page = await self._ensure()
|
page = await self._ensure()
|
||||||
await page.click(self._ref_selector(ref), timeout=5000)
|
await page.click(self._ref_selector(ref), timeout=5000)
|
||||||
|
|
||||||
async def fill_ref(self, ref, text: str, submit: bool = False) -> None:
|
async def fill_ref(self, ref: int, text: str, submit: bool = False) -> None:
|
||||||
page = await self._ensure()
|
page = await self._ensure()
|
||||||
sel = self._ref_selector(ref)
|
sel = self._ref_selector(ref)
|
||||||
await page.fill(sel, text, timeout=5000)
|
await page.fill(sel, text, timeout=5000)
|
||||||
if submit:
|
if submit:
|
||||||
await page.press(sel, "Enter")
|
await page.press(sel, "Enter")
|
||||||
|
|
||||||
async def select_ref(self, ref, value: str) -> None:
|
async def select_ref(self, ref: int, value: str) -> None:
|
||||||
page = await self._ensure()
|
page = await self._ensure()
|
||||||
sel = self._ref_selector(ref)
|
sel = self._ref_selector(ref)
|
||||||
try:
|
try:
|
||||||
@@ -230,12 +363,20 @@ class BrowserBackend(ComputerBackend):
|
|||||||
return page.url
|
return page.url
|
||||||
|
|
||||||
async def close(self) -> None:
|
async def close(self) -> None:
|
||||||
try:
|
context = self._context
|
||||||
if self._browser is not None:
|
self._context = self._page = None
|
||||||
await self._browser.close()
|
error: BaseException | None = None
|
||||||
finally:
|
if context is not None:
|
||||||
if self._pw is not None:
|
try:
|
||||||
await self._pw.stop()
|
await context.close()
|
||||||
self._browser = None
|
except BaseException as exc:
|
||||||
self._page = None
|
error = exc
|
||||||
self._pw = None
|
if self._owns_runtime:
|
||||||
|
try:
|
||||||
|
await self._runtime.close()
|
||||||
|
except BaseException as exc:
|
||||||
|
if error is not None:
|
||||||
|
raise BaseExceptionGroup("failed to close browser backend", [error, exc])
|
||||||
|
raise
|
||||||
|
if error is not None:
|
||||||
|
raise error
|
||||||
|
|||||||
@@ -1,15 +1,4 @@
|
|||||||
"""Desktop GUI backend using PyAutoGUI (the Codex-style "control the real machine").
|
"""PyAutoGUI desktop backend with HiDPI coordinate correction."""
|
||||||
|
|
||||||
Drives the local desktop: screenshots via PyAutoGUI/Pillow, mouse + keyboard via
|
|
||||||
PyAutoGUI. Works on macOS (needs Screen Recording + Accessibility permissions),
|
|
||||||
Windows, and Linux/X11 (incl. a headless Xvfb display, which is how the e2e
|
|
||||||
sandbox runs it).
|
|
||||||
|
|
||||||
Retina / HiDPI note: on macOS the screenshot is in *physical* pixels (e.g. 2x)
|
|
||||||
while PyAutoGUI's mouse API uses *logical* points. We compute the ratio from the
|
|
||||||
screenshot size vs ``pyautogui.size()`` and convert real (screenshot-pixel)
|
|
||||||
coordinates to logical points before actuating.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -40,15 +29,12 @@ _KEY_ALIASES = {
|
|||||||
"escape": "esc",
|
"escape": "esc",
|
||||||
}
|
}
|
||||||
|
|
||||||
_SCROLL_CLICK_PIXELS = 100 # one "scroll click" ~= this many pixels
|
|
||||||
|
|
||||||
|
|
||||||
class DesktopBackend(ComputerBackend):
|
class DesktopBackend(ComputerBackend):
|
||||||
environment = "desktop"
|
environment = "desktop"
|
||||||
|
|
||||||
def __init__(self) -> None:
|
def __init__(self) -> None:
|
||||||
self._pg: Any = None
|
self._pg: Any = None
|
||||||
self._image_mod: Any = None
|
|
||||||
self._ratio_x = 1.0
|
self._ratio_x = 1.0
|
||||||
self._ratio_y = 1.0
|
self._ratio_y = 1.0
|
||||||
self._dims: tuple[int, int] | None = None
|
self._dims: tuple[int, int] | None = None
|
||||||
@@ -58,12 +44,9 @@ class DesktopBackend(ComputerBackend):
|
|||||||
return self._pg
|
return self._pg
|
||||||
try:
|
try:
|
||||||
import pyautogui # noqa: PLC0415
|
import pyautogui # noqa: PLC0415
|
||||||
from PIL import Image # noqa: PLC0415
|
|
||||||
except Exception as exc: # ImportError, or platform display errors
|
except Exception as exc: # ImportError, or platform display errors
|
||||||
raise ImportError(_MISSING) from exc
|
raise ImportError(_MISSING) from exc
|
||||||
pyautogui.FAILSAFE = False
|
|
||||||
self._pg = pyautogui
|
self._pg = pyautogui
|
||||||
self._image_mod = Image
|
|
||||||
return pyautogui
|
return pyautogui
|
||||||
|
|
||||||
def _grab_png_and_size(self) -> tuple[bytes, int, int]:
|
def _grab_png_and_size(self) -> tuple[bytes, int, int]:
|
||||||
@@ -98,7 +81,7 @@ class DesktopBackend(ComputerBackend):
|
|||||||
async def click(self, x: int, y: int, button: str = "left", count: int = 1) -> None:
|
async def click(self, x: int, y: int, button: str = "left", count: int = 1) -> None:
|
||||||
pg = self._ensure()
|
pg = self._ensure()
|
||||||
lx, ly = self._to_logical(x, y)
|
lx, ly = self._to_logical(x, y)
|
||||||
await asyncio.to_thread(pg.click, lx, ly, count, 0.0, button)
|
await asyncio.to_thread(pg.click, lx, ly, clicks=count, button=button)
|
||||||
|
|
||||||
async def move(self, x: int, y: int) -> None:
|
async def move(self, x: int, y: int) -> None:
|
||||||
pg = self._ensure()
|
pg = self._ensure()
|
||||||
@@ -108,12 +91,19 @@ class DesktopBackend(ComputerBackend):
|
|||||||
async def drag(self, x: int, y: int) -> None:
|
async def drag(self, x: int, y: int) -> None:
|
||||||
pg = self._ensure()
|
pg = self._ensure()
|
||||||
lx, ly = self._to_logical(x, y)
|
lx, ly = self._to_logical(x, y)
|
||||||
await asyncio.to_thread(pg.dragTo, lx, ly, 0.3, pg.easeInOutQuad, False, "left")
|
await asyncio.to_thread(
|
||||||
|
pg.dragTo,
|
||||||
|
lx,
|
||||||
|
ly,
|
||||||
|
duration=0.3,
|
||||||
|
tween=pg.easeInOutQuad,
|
||||||
|
button="left",
|
||||||
|
)
|
||||||
|
|
||||||
async def scroll(self, x: int, y: int, direction: str, amount: int) -> None:
|
async def scroll(self, x: int, y: int, direction: str, amount: int) -> None:
|
||||||
pg = self._ensure()
|
pg = self._ensure()
|
||||||
lx, ly = self._to_logical(x, y)
|
lx, ly = self._to_logical(x, y)
|
||||||
clicks = max(1, amount) * _SCROLL_CLICK_PIXELS
|
clicks = max(1, amount)
|
||||||
await asyncio.to_thread(pg.moveTo, lx, ly)
|
await asyncio.to_thread(pg.moveTo, lx, ly)
|
||||||
if direction in ("up", "down"):
|
if direction in ("up", "down"):
|
||||||
await asyncio.to_thread(pg.scroll, clicks if direction == "up" else -clicks)
|
await asyncio.to_thread(pg.scroll, clicks if direction == "up" else -clicks)
|
||||||
|
|||||||
@@ -187,5 +187,8 @@ class _LegacyErrorPrefixTool(Tool):
|
|||||||
return ToolResult.error(result)
|
return ToolResult.error(result)
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
await self._wrapped.close()
|
||||||
|
|
||||||
def __getattr__(self, name: str) -> Any:
|
def __getattr__(self, name: str) -> Any:
|
||||||
return getattr(self._wrapped, name)
|
return getattr(self._wrapped, name)
|
||||||
|
|||||||
@@ -200,6 +200,19 @@ class ToolRegistry:
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
return ToolResult.error(f"Error executing {name}: {str(e)}" + hint)
|
return ToolResult.error(f"Error executing {name}: {str(e)}" + hint)
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
"""Close every registered tool, attempting all cleanups."""
|
||||||
|
errors: list[BaseException] = []
|
||||||
|
for tool in self._tools.values():
|
||||||
|
try:
|
||||||
|
await tool.close()
|
||||||
|
except BaseException as exc:
|
||||||
|
errors.append(exc)
|
||||||
|
if len(errors) == 1:
|
||||||
|
raise errors[0]
|
||||||
|
if errors:
|
||||||
|
raise BaseExceptionGroup("failed to close tools", errors)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def tool_names(self) -> list[str]:
|
def tool_names(self) -> list[str]:
|
||||||
"""Get list of registered tool names."""
|
"""Get list of registered tool names."""
|
||||||
|
|||||||
@@ -671,6 +671,73 @@ class OpenAICompatProvider(LLMProvider):
|
|||||||
dumped = str(content)
|
dumped = str(content)
|
||||||
return dumped or "(empty)"
|
return dumped or "(empty)"
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _move_tool_images_to_user(
|
||||||
|
cls,
|
||||||
|
messages: list[dict[str, Any]],
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""Adapt multimodal tool results to Chat Completions' text-only tool role."""
|
||||||
|
updated: list[dict[str, Any]] = []
|
||||||
|
pending_images: list[dict[str, Any]] = []
|
||||||
|
|
||||||
|
def flush_images(next_message: dict[str, Any] | None = None) -> None:
|
||||||
|
if not pending_images:
|
||||||
|
if next_message is not None:
|
||||||
|
updated.append(next_message)
|
||||||
|
return
|
||||||
|
content: list[dict[str, Any]] = [
|
||||||
|
*pending_images,
|
||||||
|
{"type": "text", "text": "Images returned by the preceding tool call(s)."},
|
||||||
|
]
|
||||||
|
pending_images.clear()
|
||||||
|
if next_message is not None and next_message.get("role") == "user":
|
||||||
|
existing = next_message.get("content")
|
||||||
|
if isinstance(existing, str):
|
||||||
|
content.append({"type": "text", "text": existing})
|
||||||
|
elif isinstance(existing, list):
|
||||||
|
content.extend(cast(list[dict[str, Any]], existing))
|
||||||
|
updated.append({**next_message, "content": content})
|
||||||
|
else:
|
||||||
|
updated.append({"role": "user", "content": content})
|
||||||
|
if next_message is not None:
|
||||||
|
updated.append(next_message)
|
||||||
|
|
||||||
|
for message in messages:
|
||||||
|
content = message.get("content")
|
||||||
|
if message.get("role") == "tool" and isinstance(content, list):
|
||||||
|
blocks = cast(list[object], content)
|
||||||
|
images: list[dict[str, Any]] = []
|
||||||
|
text_blocks: list[object] = []
|
||||||
|
for block in blocks:
|
||||||
|
if isinstance(block, dict):
|
||||||
|
block_data = cast(dict[str, Any], block)
|
||||||
|
image_url = block_data.get("image_url")
|
||||||
|
if block_data.get("type") == "image_url" and isinstance(
|
||||||
|
image_url, dict
|
||||||
|
):
|
||||||
|
images.append({"type": "image_url", "image_url": image_url})
|
||||||
|
continue
|
||||||
|
text_blocks.append(block_data)
|
||||||
|
else:
|
||||||
|
text_blocks.append(block)
|
||||||
|
if images:
|
||||||
|
updated.append({
|
||||||
|
**message,
|
||||||
|
"content": (
|
||||||
|
cls._coerce_content_to_string(text_blocks)
|
||||||
|
if text_blocks
|
||||||
|
else "(image returned)"
|
||||||
|
),
|
||||||
|
})
|
||||||
|
pending_images.extend(images)
|
||||||
|
continue
|
||||||
|
if message.get("role") != "tool":
|
||||||
|
flush_images(message)
|
||||||
|
else:
|
||||||
|
updated.append(message)
|
||||||
|
flush_images()
|
||||||
|
return updated
|
||||||
|
|
||||||
def _sanitize_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
def _sanitize_messages(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||||
"""Strip non-standard keys, normalize tool_call IDs."""
|
"""Strip non-standard keys, normalize tool_call IDs."""
|
||||||
sanitized = LLMProvider._sanitize_request_messages(messages, _ALLOWED_MSG_KEYS)
|
sanitized = LLMProvider._sanitize_request_messages(messages, _ALLOWED_MSG_KEYS)
|
||||||
@@ -824,9 +891,10 @@ class OpenAICompatProvider(LLMProvider):
|
|||||||
|
|
||||||
model_name = self._request_model_name(model_name)
|
model_name = self._request_model_name(model_name)
|
||||||
|
|
||||||
|
sanitized_messages = self._sanitize_messages(self._sanitize_empty_content(messages))
|
||||||
kwargs: dict[str, Any] = {
|
kwargs: dict[str, Any] = {
|
||||||
"model": model_name,
|
"model": model_name,
|
||||||
"messages": self._sanitize_messages(self._sanitize_empty_content(messages)),
|
"messages": self._move_tool_images_to_user(sanitized_messages),
|
||||||
}
|
}
|
||||||
|
|
||||||
# GPT-5 and reasoning models (o1/o3/o4) reject temperature when
|
# GPT-5 and reasoning models (o1/o3/o4) reject temperature when
|
||||||
|
|||||||
@@ -1,72 +0,0 @@
|
|||||||
"""Coordinate scaling between the screenshot the model sees and the real device.
|
|
||||||
|
|
||||||
The computer-use loop sends the model a screenshot that is downscaled to a
|
|
||||||
target size (smaller screenshots cost fewer tokens, are faster, and match the
|
|
||||||
resolutions vision models are tuned for). The model then replies with click /
|
|
||||||
move coordinates *in that downscaled space*. Those must be scaled back to the
|
|
||||||
real device's pixel space before we actuate mouse events — getting this wrong is
|
|
||||||
the single most common cause of "the clicks miss" bugs.
|
|
||||||
|
|
||||||
This module is pure (no I/O, no optional deps) so it is cheap to import at tool
|
|
||||||
auto-discovery time and trivial to unit test.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from dataclasses import dataclass
|
|
||||||
|
|
||||||
|
|
||||||
def fit_target_size(
|
|
||||||
real_width: int,
|
|
||||||
real_height: int,
|
|
||||||
max_width: int,
|
|
||||||
max_height: int,
|
|
||||||
) -> tuple[int, int]:
|
|
||||||
"""Target size fitting within ``(max_width, max_height)``, preserving aspect.
|
|
||||||
|
|
||||||
Never upscales: if the real screen is already smaller than the max, the real
|
|
||||||
size is returned unchanged.
|
|
||||||
"""
|
|
||||||
if real_width <= 0 or real_height <= 0:
|
|
||||||
return max(1, max_width), max(1, max_height)
|
|
||||||
if max_width <= 0 or max_height <= 0:
|
|
||||||
return real_width, real_height
|
|
||||||
scale = min(max_width / real_width, max_height / real_height, 1.0)
|
|
||||||
return max(1, round(real_width * scale)), max(1, round(real_height * scale))
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass(frozen=True)
|
|
||||||
class ScreenScaler:
|
|
||||||
"""Maps model (target-space) coordinates onto real device pixels."""
|
|
||||||
|
|
||||||
real_width: int
|
|
||||||
real_height: int
|
|
||||||
target_width: int
|
|
||||||
target_height: int
|
|
||||||
|
|
||||||
@property
|
|
||||||
def scale_x(self) -> float:
|
|
||||||
return self.real_width / self.target_width if self.target_width else 1.0
|
|
||||||
|
|
||||||
@property
|
|
||||||
def scale_y(self) -> float:
|
|
||||||
return self.real_height / self.target_height if self.target_height else 1.0
|
|
||||||
|
|
||||||
def to_real(self, x: float, y: float) -> tuple[int, int]:
|
|
||||||
"""Scale a model-space ``(x, y)`` to a real device pixel, clamped in-bounds."""
|
|
||||||
rx = round(x * self.scale_x)
|
|
||||||
ry = round(y * self.scale_y)
|
|
||||||
rx = max(0, min(rx, max(0, self.real_width - 1)))
|
|
||||||
ry = max(0, min(ry, max(0, self.real_height - 1)))
|
|
||||||
return rx, ry
|
|
||||||
|
|
||||||
@classmethod
|
|
||||||
def for_screen(
|
|
||||||
cls,
|
|
||||||
real_width: int,
|
|
||||||
real_height: int,
|
|
||||||
max_width: int,
|
|
||||||
max_height: int,
|
|
||||||
) -> ScreenScaler:
|
|
||||||
tw, th = fit_target_size(real_width, real_height, max_width, max_height)
|
|
||||||
return cls(real_width, real_height, tw, th)
|
|
||||||
+1
-1
@@ -91,7 +91,7 @@ olostep = [
|
|||||||
computer-use = [
|
computer-use = [
|
||||||
"pyautogui>=0.9.54",
|
"pyautogui>=0.9.54",
|
||||||
"pillow>=10.0.0",
|
"pillow>=10.0.0",
|
||||||
"playwright>=1.40.0",
|
"playwright>=1.48.0",
|
||||||
]
|
]
|
||||||
dev = [
|
dev = [
|
||||||
"pytest>=9.0.0,<10.0.0",
|
"pytest>=9.0.0,<10.0.0",
|
||||||
|
|||||||
@@ -76,11 +76,13 @@ class TestHandleStop:
|
|||||||
|
|
||||||
loop.subagents.close = close_subagents
|
loop.subagents.close = close_subagents
|
||||||
loop._exec_session_manager.close_all = AsyncMock()
|
loop._exec_session_manager.close_all = AsyncMock()
|
||||||
|
loop.tools.close = AsyncMock()
|
||||||
with patch("nanobot.agent.loop.agent_context.close_mcp", AsyncMock()):
|
with patch("nanobot.agent.loop.agent_context.close_mcp", AsyncMock()):
|
||||||
await loop.close_mcp()
|
await loop.close_mcp()
|
||||||
|
|
||||||
assert events == ["turn_cancelled", "resources_closed"]
|
assert events == ["turn_cancelled", "resources_closed"]
|
||||||
assert task.cancelled()
|
assert task.cancelled()
|
||||||
|
loop.tools.close.assert_awaited_once()
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_close_mcp_serializes_duplicate_cleanup(self):
|
async def test_close_mcp_serializes_duplicate_cleanup(self):
|
||||||
|
|||||||
@@ -1,40 +0,0 @@
|
|||||||
# Sandbox for computer_use end-to-end tests.
|
|
||||||
#
|
|
||||||
# Provides BOTH backends in an isolated Linux container:
|
|
||||||
# - browser : headless Chromium via Playwright (no display needed)
|
|
||||||
# - desktop : PyAutoGUI driving a virtual X11 display (Xvfb) — Codex-style,
|
|
||||||
# but contained so it never touches a real machine.
|
|
||||||
#
|
|
||||||
# Build: docker build -f tests/e2e/Dockerfile -t nanobot-cu-e2e .
|
|
||||||
# Run: docker run --rm -e OPENROUTER_API_KEY=sk-or-... -e COMPUTER_USE_E2E=1 \
|
|
||||||
# nanobot-cu-e2e
|
|
||||||
#
|
|
||||||
# The desktop backend needs a display; the entrypoint starts Xvfb on :99 and
|
|
||||||
# exports DISPLAY so pyautogui/scrot work headlessly.
|
|
||||||
FROM python:3.11-slim
|
|
||||||
|
|
||||||
ENV DEBIAN_FRONTEND=noninteractive \
|
|
||||||
DISPLAY=:99 \
|
|
||||||
PYTHONUNBUFFERED=1
|
|
||||||
|
|
||||||
# X11 virtual display + screenshot/input tooling for the desktop backend,
|
|
||||||
# plus the system libs Chromium needs.
|
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
||||||
xvfb x11-utils scrot xauth \
|
|
||||||
libxtst6 libxrandr2 libxss1 libnss3 libasound2 libgbm1 libgtk-3-0 \
|
|
||||||
fonts-liberation ca-certificates \
|
|
||||||
&& rm -rf /var/lib/apt/lists/*
|
|
||||||
|
|
||||||
WORKDIR /app
|
|
||||||
COPY . /app
|
|
||||||
|
|
||||||
RUN pip install --no-cache-dir -e ".[computer-use,dev]" \
|
|
||||||
&& python -m playwright install --with-deps chromium
|
|
||||||
|
|
||||||
# Start Xvfb, wait for it, then run the e2e suite.
|
|
||||||
RUN printf '#!/bin/sh\nset -e\nXvfb :99 -screen 0 1280x800x24 >/tmp/xvfb.log 2>&1 &\nfor i in $(seq 1 30); do xdpyinfo -display :99 >/dev/null 2>&1 && break; sleep 0.3; done\nexec "$@"\n' > /usr/local/bin/with-xvfb \
|
|
||||||
&& chmod +x /usr/local/bin/with-xvfb
|
|
||||||
|
|
||||||
ENV COMPUTER_USE_E2E=1
|
|
||||||
ENTRYPOINT ["with-xvfb"]
|
|
||||||
CMD ["pytest", "tests/e2e/test_computer_use_e2e.py", "-v", "-s"]
|
|
||||||
@@ -1,124 +0,0 @@
|
|||||||
# computer_use end-to-end tests
|
|
||||||
|
|
||||||
Real vision models (via **OpenRouter**) drive the real backends through the full
|
|
||||||
agent loop. Gated by env vars, excluded from default CI.
|
|
||||||
|
|
||||||
## What it proves
|
|
||||||
|
|
||||||
- Model-agnostic screenshot delivery: every model goes through one
|
|
||||||
OpenAI-compatible endpoint (OpenRouter); screenshots come back as follow-up
|
|
||||||
**user** messages, so no provider-specific path is exercised.
|
|
||||||
- The `computer_use` tool + backends actually see the screen and act
|
|
||||||
(read a number, click a button, type into a field).
|
|
||||||
|
|
||||||
## Requirements
|
|
||||||
|
|
||||||
```bash
|
|
||||||
pip install 'nanobot-ai[computer-use]'
|
|
||||||
playwright install chromium # for the browser backend
|
|
||||||
```
|
|
||||||
|
|
||||||
Models must be **vision + tool-calling** capable on OpenRouter.
|
|
||||||
|
|
||||||
## Run the browser e2e locally (no Docker, headless)
|
|
||||||
|
|
||||||
```bash
|
|
||||||
COMPUTER_USE_E2E=1 OPENROUTER_API_KEY=sk-or-... \
|
|
||||||
pytest tests/e2e/test_computer_use_e2e.py -v -s
|
|
||||||
```
|
|
||||||
|
|
||||||
Pick models explicitly (default: claude-sonnet-4.5 / gpt-4o / gemini-2.5-pro):
|
|
||||||
|
|
||||||
```bash
|
|
||||||
COMPUTER_USE_E2E_MODELS="anthropic/claude-sonnet-4.5,openai/gpt-4o" \
|
|
||||||
COMPUTER_USE_E2E=1 OPENROUTER_API_KEY=sk-or-... \
|
|
||||||
pytest tests/e2e/test_computer_use_e2e.py -v -s
|
|
||||||
```
|
|
||||||
|
|
||||||
Vision models are non-deterministic; each scenario retries
|
|
||||||
`COMPUTER_USE_E2E_ATTEMPTS` (default 2) times and passes if any attempt succeeds.
|
|
||||||
|
|
||||||
## Run in the sandbox (Docker; also covers the desktop/pyautogui backend)
|
|
||||||
|
|
||||||
```bash
|
|
||||||
docker build -f tests/e2e/Dockerfile -t nanobot-cu-e2e .
|
|
||||||
docker run --rm -e OPENROUTER_API_KEY=sk-or-... -e COMPUTER_USE_E2E=1 nanobot-cu-e2e
|
|
||||||
```
|
|
||||||
|
|
||||||
The container starts Xvfb on `:99` so the desktop backend (PyAutoGUI + scrot)
|
|
||||||
works headlessly and contained — it never touches a real machine.
|
|
||||||
|
|
||||||
## Scenarios (`model × scenario` matrix)
|
|
||||||
|
|
||||||
| Test | What it checks |
|
|
||||||
|------|----------------|
|
|
||||||
| `test_read_screen` | reads the rendered number `42` from a screenshot (perception) |
|
|
||||||
| `test_click_button` | clicks **Submit**; asserts `#status == "SUBMITTED"` via the DOM (click accuracy) |
|
|
||||||
| `test_type_and_submit` | types a name + clicks **Greet**; asserts the greeting (multi-step loop) |
|
|
||||||
|
|
||||||
## Findings (first real run, June 2026, via OpenRouter)
|
|
||||||
|
|
||||||
Matrix of 5 model families × 3 scenarios:
|
|
||||||
|
|
||||||
| Model | read_screen | click_button | type_and_submit |
|
|
||||||
|-------|:----------:|:-----------:|:--------------:|
|
|
||||||
| anthropic/claude-sonnet-4.5 | ✅ | ✅ | ✅ |
|
|
||||||
| openai/gpt-4o | ✅ | ❌ | ❌ |
|
|
||||||
| google/gemini-2.5-pro | ✅ | ❌ | ❌ |
|
|
||||||
| qwen/qwen3-vl-235b-a22b-instruct | ✅ | ❌ | ❌ |
|
|
||||||
| x-ai/grok-4.3 | ✅ | ❌ | ❌ |
|
|
||||||
|
|
||||||
**Takeaway:** the model-agnostic *plumbing* works for everyone — all 5 models
|
|
||||||
received the screenshot (delivered as a user message) and read the on-screen
|
|
||||||
number. But **pixel-precise clicking only worked with the computer-use-trained
|
|
||||||
model (Claude)**. Diagnostics showed gpt-4o clicking a round-number guess
|
|
||||||
`(100,100)` for a button at `(37,156)`, and qwen3-vl emitting an `[x,y]` array
|
|
||||||
(now supported) but still mis-grounding the y coordinate. This matches the
|
|
||||||
field: general/GUI VLMs perceive well but ground pixel coordinates poorly.
|
|
||||||
|
|
||||||
**Implication for broad model support:** for the *browser* backend, a
|
|
||||||
DOM/accessibility-based interaction mode (click an element by ref/description
|
|
||||||
instead of by pixel — how browser-use / Playwright-MCP work) would make action
|
|
||||||
reliable across *any* tool-capable model, including non-vision ones. That is the
|
|
||||||
recommended next step beyond this pixel-based v1.
|
|
||||||
|
|
||||||
## Browser DOM mode (model-agnostic) — recommended for the web
|
|
||||||
|
|
||||||
`tests/e2e/test_browser_dom_e2e.py` and `tests/e2e/test_browser_complex_e2e.py`
|
|
||||||
exercise the **`browser`** tool, which acts by element **ref** (DOM/accessibility
|
|
||||||
snapshot) instead of pixels. This removes the pixel-grounding limitation above:
|
|
||||||
the same models that miss pixel clicks succeed here.
|
|
||||||
|
|
||||||
DOM matrix (simple scenarios) — the models that FAILED pixel clicking:
|
|
||||||
|
|
||||||
| Model | DOM click | DOM type+submit | read_text (no vision) |
|
|
||||||
|-------|:--------:|:--------------:|:--------------------:|
|
|
||||||
| openai/gpt-4o | ✅ | ✅ | flaky |
|
|
||||||
| google/gemini-2.5-pro | ✅ | flaky | ✅ |
|
|
||||||
| qwen/qwen3-vl-235b | ✅ | ✅ | ✅ |
|
|
||||||
|
|
||||||
Click went from **0/3 in pixel mode to 3/3 in DOM mode.** Because DOM mode needs
|
|
||||||
no pixel grounding (run with `include_screenshot=false`), it works with ANY
|
|
||||||
tool-calling model — including non-vision ones.
|
|
||||||
|
|
||||||
**Complex flows** (`test_browser_complex_e2e.py`) drive a small local web app
|
|
||||||
(served over HTTP so localStorage + navigation work), pages in `tests/e2e/pages/`:
|
|
||||||
|
|
||||||
- multi-field form: text inputs + `<select>` + radio group + checkbox + submit
|
|
||||||
- add-to-cart + checkout across two pages, verifying the cart total
|
|
||||||
- pick the correct row out of 60 (selecting among many similar elements)
|
|
||||||
- login (two fields + submit) following the redirect to a dashboard
|
|
||||||
|
|
||||||
Default models: `openai/gpt-5.1`, `anthropic/claude-sonnet-4.5` (override with
|
|
||||||
`COMPUTER_USE_E2E_MODELS`).
|
|
||||||
|
|
||||||
```bash
|
|
||||||
COMPUTER_USE_E2E=1 OPENROUTER_API_KEY=sk-or-... \
|
|
||||||
pytest tests/e2e/test_browser_complex_e2e.py -v
|
|
||||||
```
|
|
||||||
|
|
||||||
## Cost
|
|
||||||
|
|
||||||
Each scenario makes several model calls with screenshots (~1–2k tokens each).
|
|
||||||
Keep the model list small; check your OpenRouter spend. The full 5×3 matrix
|
|
||||||
above cost a few cents and took ~5 minutes.
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
<!doctype html>
|
|
||||||
<html><head><meta charset="utf-8"><title>Cart</title></head>
|
|
||||||
<body>
|
|
||||||
<h1>Your Cart</h1>
|
|
||||||
<ul id="items"></ul>
|
|
||||||
<div id="total">Total: $0</div>
|
|
||||||
<p><a href="shop.html">Back to shop</a></p>
|
|
||||||
<script>
|
|
||||||
var cart = JSON.parse(localStorage.getItem('cart') || '[]');
|
|
||||||
document.getElementById('items').innerHTML =
|
|
||||||
cart.map(function(i){ return '<li>' + i.n + ' $' + i.p + '</li>'; }).join('');
|
|
||||||
var total = cart.reduce(function(s, i){ return s + i.p; }, 0);
|
|
||||||
document.getElementById('total').textContent = 'Total: $' + total;
|
|
||||||
</script>
|
|
||||||
</body></html>
|
|
||||||
@@ -1,10 +0,0 @@
|
|||||||
<!doctype html>
|
|
||||||
<html><head><meta charset="utf-8"><title>Dashboard</title></head>
|
|
||||||
<body>
|
|
||||||
<h1 id="welcome"></h1>
|
|
||||||
<p>You are signed in.</p>
|
|
||||||
<script>
|
|
||||||
var u = new URLSearchParams(location.search).get('user') || '';
|
|
||||||
document.getElementById('welcome').textContent = 'Welcome ' + u;
|
|
||||||
</script>
|
|
||||||
</body></html>
|
|
||||||
@@ -1,41 +0,0 @@
|
|||||||
<!doctype html>
|
|
||||||
<html><head><meta charset="utf-8"><title>Sign up</title></head>
|
|
||||||
<body>
|
|
||||||
<h1>Create your account</h1>
|
|
||||||
<form id="f" onsubmit="return submitForm(event)">
|
|
||||||
<p><input id="name" placeholder="Full name"></p>
|
|
||||||
<p><input id="email" type="email" placeholder="Email address"></p>
|
|
||||||
<p>
|
|
||||||
<select id="country" aria-label="Country">
|
|
||||||
<option value="">-- choose country --</option>
|
|
||||||
<option value="us">United States</option>
|
|
||||||
<option value="cn">China</option>
|
|
||||||
<option value="de">Germany</option>
|
|
||||||
</select>
|
|
||||||
</p>
|
|
||||||
<fieldset>
|
|
||||||
<legend>Plan</legend>
|
|
||||||
<label><input type="radio" name="plan" value="free" aria-label="Free plan"> Free</label>
|
|
||||||
<label><input type="radio" name="plan" value="pro" aria-label="Pro plan"> Pro</label>
|
|
||||||
<label><input type="radio" name="plan" value="team" aria-label="Team plan"> Team</label>
|
|
||||||
</fieldset>
|
|
||||||
<p><label><input type="checkbox" id="agree" aria-label="Accept terms"> I accept the terms</label></p>
|
|
||||||
<button id="submit" type="submit">Create account</button>
|
|
||||||
</form>
|
|
||||||
<div id="result"></div>
|
|
||||||
<script>
|
|
||||||
function submitForm(e){
|
|
||||||
e.preventDefault();
|
|
||||||
var plan = (document.querySelector('input[name=plan]:checked') || {}).value || '';
|
|
||||||
var parts = [
|
|
||||||
document.getElementById('name').value,
|
|
||||||
document.getElementById('email').value,
|
|
||||||
document.getElementById('country').value,
|
|
||||||
plan,
|
|
||||||
document.getElementById('agree').checked ? 'agreed' : 'no'
|
|
||||||
];
|
|
||||||
document.getElementById('result').textContent = parts.join('|');
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
</script>
|
|
||||||
</body></html>
|
|
||||||
@@ -1,24 +0,0 @@
|
|||||||
<!doctype html>
|
|
||||||
<html><head><meta charset="utf-8"><title>Long list</title></head>
|
|
||||||
<body>
|
|
||||||
<h1>Pick a row</h1>
|
|
||||||
<div id="picked">none</div>
|
|
||||||
<ul id="list"></ul>
|
|
||||||
<script>
|
|
||||||
var ul = document.getElementById('list');
|
|
||||||
for (var i = 1; i <= 60; i++) {
|
|
||||||
(function(n){
|
|
||||||
var li = document.createElement('li');
|
|
||||||
li.textContent = 'Row ' + n + ' ';
|
|
||||||
var b = document.createElement('button');
|
|
||||||
b.textContent = 'Pick';
|
|
||||||
b.setAttribute('aria-label', 'Pick Row ' + n);
|
|
||||||
b.addEventListener('click', function(){
|
|
||||||
document.getElementById('picked').textContent = String(n);
|
|
||||||
});
|
|
||||||
li.appendChild(b);
|
|
||||||
ul.appendChild(li);
|
|
||||||
})(i);
|
|
||||||
}
|
|
||||||
</script>
|
|
||||||
</body></html>
|
|
||||||
@@ -1,24 +0,0 @@
|
|||||||
<!doctype html>
|
|
||||||
<html><head><meta charset="utf-8"><title>Login</title></head>
|
|
||||||
<body>
|
|
||||||
<h1>Sign in</h1>
|
|
||||||
<form onsubmit="return go(event)">
|
|
||||||
<p><input id="user" aria-label="Username" placeholder="Username"></p>
|
|
||||||
<p><input id="pass" type="password" aria-label="Password" placeholder="Password"></p>
|
|
||||||
<button type="submit">Log in</button>
|
|
||||||
</form>
|
|
||||||
<div id="err"></div>
|
|
||||||
<script>
|
|
||||||
function go(e){
|
|
||||||
e.preventDefault();
|
|
||||||
var u = document.getElementById('user').value;
|
|
||||||
var p = document.getElementById('pass').value;
|
|
||||||
if (u === 'admin' && p === 'secret') {
|
|
||||||
location.href = 'dashboard.html?user=' + encodeURIComponent(u);
|
|
||||||
} else {
|
|
||||||
document.getElementById('err').textContent = 'Invalid credentials';
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
</script>
|
|
||||||
</body></html>
|
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
<!doctype html>
|
|
||||||
<html><head><meta charset="utf-8"><title>Shop</title></head>
|
|
||||||
<body>
|
|
||||||
<h1>Gadget Shop</h1>
|
|
||||||
<div id="badge">Cart items: <span id="count">0</span></div>
|
|
||||||
<ul>
|
|
||||||
<li>Wireless Mouse - $25
|
|
||||||
<button aria-label="Add Wireless Mouse to cart" onclick="add('Wireless Mouse',25)">Add to cart</button></li>
|
|
||||||
<li>Mechanical Keyboard - $45
|
|
||||||
<button aria-label="Add Mechanical Keyboard to cart" onclick="add('Mechanical Keyboard',45)">Add to cart</button></li>
|
|
||||||
<li>4K Monitor - $200
|
|
||||||
<button aria-label="Add 4K Monitor to cart" onclick="add('4K Monitor',200)">Add to cart</button></li>
|
|
||||||
</ul>
|
|
||||||
<p><a href="cart.html" id="tocart">View cart</a></p>
|
|
||||||
<script>
|
|
||||||
function add(name, price){
|
|
||||||
var cart = JSON.parse(localStorage.getItem('cart') || '[]');
|
|
||||||
cart.push({n: name, p: price});
|
|
||||||
localStorage.setItem('cart', JSON.stringify(cart));
|
|
||||||
document.getElementById('count').textContent = cart.length;
|
|
||||||
}
|
|
||||||
</script>
|
|
||||||
</body></html>
|
|
||||||
@@ -1,194 +0,0 @@
|
|||||||
"""Complex, multi-step DOM-browser e2e against a small local web app.
|
|
||||||
|
|
||||||
Pages live in ``tests/e2e/pages/`` and are served over a local HTTP server (so
|
|
||||||
localStorage and cross-page navigation work). Scenarios exercise realistic
|
|
||||||
flows: a multi-field form (text + select + radio + checkbox), an add-to-cart +
|
|
||||||
checkout flow across pages, picking one row out of 60, and a login → dashboard
|
|
||||||
redirect. All run in DOM mode (act by element ref), so they are model-agnostic.
|
|
||||||
|
|
||||||
Run:
|
|
||||||
pip install 'nanobot-ai[computer-use]' && playwright install chromium
|
|
||||||
COMPUTER_USE_E2E=1 OPENROUTER_API_KEY=sk-or-... \
|
|
||||||
pytest tests/e2e/test_browser_complex_e2e.py -v -s
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import functools
|
|
||||||
import http.server
|
|
||||||
import os
|
|
||||||
import socketserver
|
|
||||||
import threading
|
|
||||||
from pathlib import Path
|
|
||||||
from types import SimpleNamespace
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
|
|
||||||
from nanobot.agent.runner import AgentRunner, AgentRunSpec
|
|
||||||
from nanobot.agent.tools.browser_tool import BrowserTool
|
|
||||||
from nanobot.agent.tools.registry import ToolRegistry
|
|
||||||
from nanobot.config.schema import AgentDefaults
|
|
||||||
|
|
||||||
pytestmark = pytest.mark.skipif(
|
|
||||||
not (os.getenv("COMPUTER_USE_E2E") and os.getenv("OPENROUTER_API_KEY")),
|
|
||||||
reason="set COMPUTER_USE_E2E=1 and OPENROUTER_API_KEY to run complex browser e2e tests",
|
|
||||||
)
|
|
||||||
|
|
||||||
_DEFAULT_MODELS = [
|
|
||||||
"openai/gpt-5.1",
|
|
||||||
"anthropic/claude-sonnet-4.5",
|
|
||||||
]
|
|
||||||
_MAX_TOOL_RESULT_CHARS = AgentDefaults().max_tool_result_chars
|
|
||||||
_PAGES = Path(__file__).parent / "pages"
|
|
||||||
|
|
||||||
|
|
||||||
def _models() -> list[str]:
|
|
||||||
env = os.getenv("COMPUTER_USE_E2E_MODELS", "").strip()
|
|
||||||
return [m.strip() for m in env.split(",") if m.strip()] or _DEFAULT_MODELS
|
|
||||||
|
|
||||||
|
|
||||||
def _attempts() -> int:
|
|
||||||
try:
|
|
||||||
return max(1, int(os.getenv("COMPUTER_USE_E2E_ATTEMPTS", "2")))
|
|
||||||
except ValueError:
|
|
||||||
return 2
|
|
||||||
|
|
||||||
|
|
||||||
def _make_provider(model: str):
|
|
||||||
from nanobot.providers.openai_compat_provider import OpenAICompatProvider
|
|
||||||
|
|
||||||
provider = OpenAICompatProvider(
|
|
||||||
api_key=os.environ["OPENROUTER_API_KEY"],
|
|
||||||
api_base="https://openrouter.ai/api/v1",
|
|
||||||
default_model=model,
|
|
||||||
)
|
|
||||||
provider.generation = SimpleNamespace(max_tokens=4096, temperature=0.0, reasoning_effort=None)
|
|
||||||
return provider
|
|
||||||
|
|
||||||
|
|
||||||
class _QuietHandler(http.server.SimpleHTTPRequestHandler):
|
|
||||||
def log_message(self, *args): # silence per-request stderr logging
|
|
||||||
pass
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture(scope="module")
|
|
||||||
def base_url() -> str:
|
|
||||||
handler = functools.partial(_QuietHandler, directory=str(_PAGES))
|
|
||||||
httpd = socketserver.TCPServer(("127.0.0.1", 0), handler)
|
|
||||||
port = httpd.server_address[1]
|
|
||||||
thread = threading.Thread(target=httpd.serve_forever, daemon=True)
|
|
||||||
thread.start()
|
|
||||||
yield f"http://127.0.0.1:{port}"
|
|
||||||
httpd.shutdown()
|
|
||||||
httpd.server_close()
|
|
||||||
|
|
||||||
|
|
||||||
def _tool(start_url: str) -> BrowserTool:
|
|
||||||
return BrowserTool(start_url=start_url, headless=True, include_screenshot=False, max_elements=120)
|
|
||||||
|
|
||||||
|
|
||||||
async def _run(provider, tool: BrowserTool, objective: str, max_iterations: int = 24):
|
|
||||||
tools = ToolRegistry()
|
|
||||||
tools.register(tool)
|
|
||||||
return await AgentRunner(provider).run(AgentRunSpec(
|
|
||||||
initial_messages=[{"role": "user", "content": objective}],
|
|
||||||
tools=tools,
|
|
||||||
model=provider.get_default_model(),
|
|
||||||
max_iterations=max_iterations,
|
|
||||||
max_tool_result_chars=_MAX_TOOL_RESULT_CHARS,
|
|
||||||
))
|
|
||||||
|
|
||||||
|
|
||||||
async def _eval(tool: BrowserTool, expr: str):
|
|
||||||
backend = await tool._get_backend()
|
|
||||||
return await backend._page.evaluate(expr)
|
|
||||||
|
|
||||||
|
|
||||||
async def _attempt(model, start_page, objective, check, base_url, max_iterations=24):
|
|
||||||
"""Run up to N attempts; check(tool)->bool decides success. Returns bool."""
|
|
||||||
for _ in range(_attempts()):
|
|
||||||
tool = _tool(f"{base_url}/{start_page}")
|
|
||||||
try:
|
|
||||||
await _run(_make_provider(model), tool, objective, max_iterations)
|
|
||||||
if await check(tool):
|
|
||||||
return True
|
|
||||||
finally:
|
|
||||||
await tool.close()
|
|
||||||
return False
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_multifield_form(model: str, base_url: str):
|
|
||||||
"""Fill a form with text inputs, a <select>, a radio group, and a checkbox."""
|
|
||||||
objective = (
|
|
||||||
"The signup form is open. Using the browser tool, complete it: set Full name to "
|
|
||||||
"'Ada Lovelace', Email to 'ada@example.com', Country to 'United States', choose the "
|
|
||||||
"'Pro' plan, accept the terms, then click 'Create account'."
|
|
||||||
)
|
|
||||||
|
|
||||||
async def check(tool):
|
|
||||||
result = await _eval(tool, "document.getElementById('result').textContent")
|
|
||||||
return bool(result) and all(
|
|
||||||
s in result for s in ("Ada Lovelace", "ada@example.com", "us", "pro", "agreed")
|
|
||||||
)
|
|
||||||
|
|
||||||
assert await _attempt(model, "form.html", objective, check, base_url, max_iterations=28), \
|
|
||||||
f"[{model}] form should be filled and submitted with all fields correct"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_add_to_cart_flow(model: str, base_url: str):
|
|
||||||
"""Add two specific products, navigate to the cart, verify the total."""
|
|
||||||
objective = (
|
|
||||||
"The shop page is already open — do NOT navigate to any URL. Start by calling the "
|
|
||||||
"browser tool's 'snapshot' action, then add BOTH the 'Wireless Mouse' and the "
|
|
||||||
"'Mechanical Keyboard' to the cart (not the monitor), and finally click the "
|
|
||||||
"'View cart' link."
|
|
||||||
)
|
|
||||||
|
|
||||||
async def check(tool):
|
|
||||||
items = await _eval(tool, "document.getElementById('items') ? document.getElementById('items').textContent : ''")
|
|
||||||
total = await _eval(tool, "document.getElementById('total') ? document.getElementById('total').textContent : ''")
|
|
||||||
return ("Wireless Mouse" in items and "Mechanical Keyboard" in items
|
|
||||||
and "Monitor" not in items and "70" in total)
|
|
||||||
|
|
||||||
assert await _attempt(model, "shop.html", objective, check, base_url), \
|
|
||||||
f"[{model}] cart should contain both items with total $70"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_pick_row_among_many(model: str, base_url: str):
|
|
||||||
"""Pick the correct row out of 60 (selecting among many similar elements)."""
|
|
||||||
objective = (
|
|
||||||
"The list page is open. Using the browser tool, find 'Row 42' and click its 'Pick' "
|
|
||||||
"button."
|
|
||||||
)
|
|
||||||
|
|
||||||
async def check(tool):
|
|
||||||
picked = await _eval(tool, "document.getElementById('picked').textContent")
|
|
||||||
return picked == "42"
|
|
||||||
|
|
||||||
assert await _attempt(model, "list.html", objective, check, base_url), \
|
|
||||||
f"[{model}] should pick Row 42"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_login_then_dashboard(model: str, base_url: str):
|
|
||||||
"""Log in (two fields + submit) and follow the redirect to the dashboard."""
|
|
||||||
objective = (
|
|
||||||
"The login page is open. Using the browser tool, sign in with username 'admin' and "
|
|
||||||
"password 'secret', then submit."
|
|
||||||
)
|
|
||||||
|
|
||||||
async def check(tool):
|
|
||||||
welcome = await _eval(
|
|
||||||
tool, "document.getElementById('welcome') ? document.getElementById('welcome').textContent : ''"
|
|
||||||
)
|
|
||||||
return "Welcome admin" in welcome
|
|
||||||
|
|
||||||
assert await _attempt(model, "login.html", objective, check, base_url), \
|
|
||||||
f"[{model}] should log in and reach the dashboard"
|
|
||||||
@@ -1,171 +0,0 @@
|
|||||||
"""End-to-end DOM-browser tests: models that FAIL pixel-based clicking succeed
|
|
||||||
here because they act by element ref instead of by pixel coordinates.
|
|
||||||
|
|
||||||
This is the model-agnostic action path. ``include_screenshot=False`` below proves
|
|
||||||
it needs no vision at all — the model acts purely on the text element list.
|
|
||||||
|
|
||||||
Run:
|
|
||||||
pip install 'nanobot-ai[computer-use]' && playwright install chromium
|
|
||||||
COMPUTER_USE_E2E=1 OPENROUTER_API_KEY=sk-or-... \
|
|
||||||
pytest tests/e2e/test_browser_dom_e2e.py -v -s
|
|
||||||
|
|
||||||
Default models are ones that failed pixel clicking (see test_computer_use_e2e
|
|
||||||
"Findings") to demonstrate the DOM mode fixes them. Override with
|
|
||||||
COMPUTER_USE_E2E_MODELS.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
|
||||||
import tempfile
|
|
||||||
from pathlib import Path
|
|
||||||
from types import SimpleNamespace
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
|
|
||||||
from nanobot.agent.runner import AgentRunner, AgentRunSpec
|
|
||||||
from nanobot.agent.tools.browser_tool import BrowserTool
|
|
||||||
from nanobot.agent.tools.registry import ToolRegistry
|
|
||||||
from nanobot.config.schema import AgentDefaults
|
|
||||||
|
|
||||||
pytestmark = pytest.mark.skipif(
|
|
||||||
not (os.getenv("COMPUTER_USE_E2E") and os.getenv("OPENROUTER_API_KEY")),
|
|
||||||
reason="set COMPUTER_USE_E2E=1 and OPENROUTER_API_KEY to run browser DOM e2e tests",
|
|
||||||
)
|
|
||||||
|
|
||||||
# Default to models that FAILED pixel clicking — DOM mode should fix them.
|
|
||||||
_DEFAULT_MODELS = [
|
|
||||||
"openai/gpt-5.1",
|
|
||||||
"google/gemini-2.5-pro",
|
|
||||||
]
|
|
||||||
_MAX_TOOL_RESULT_CHARS = AgentDefaults().max_tool_result_chars
|
|
||||||
|
|
||||||
_TEST_PAGE = """<!doctype html><html><head><meta charset=utf-8><title>dom e2e</title></head>
|
|
||||||
<body>
|
|
||||||
<div id=number style="font-size:80px">42</div>
|
|
||||||
<button id=submit onclick="document.getElementById('status').textContent='SUBMITTED'">Submit</button>
|
|
||||||
<div id=status>idle</div>
|
|
||||||
<input id=name placeholder="your name">
|
|
||||||
<button id=greet onclick="document.getElementById('greeting').textContent='HELLO '+document.getElementById('name').value">Greet</button>
|
|
||||||
<div id=greeting></div>
|
|
||||||
</body></html>"""
|
|
||||||
|
|
||||||
|
|
||||||
def _models() -> list[str]:
|
|
||||||
env = os.getenv("COMPUTER_USE_E2E_MODELS", "").strip()
|
|
||||||
return [m.strip() for m in env.split(",") if m.strip()] or _DEFAULT_MODELS
|
|
||||||
|
|
||||||
|
|
||||||
def _make_provider(model: str):
|
|
||||||
from nanobot.providers.openai_compat_provider import OpenAICompatProvider
|
|
||||||
|
|
||||||
provider = OpenAICompatProvider(
|
|
||||||
api_key=os.environ["OPENROUTER_API_KEY"],
|
|
||||||
api_base="https://openrouter.ai/api/v1",
|
|
||||||
default_model=model,
|
|
||||||
)
|
|
||||||
provider.generation = SimpleNamespace(max_tokens=2048, temperature=0.0, reasoning_effort=None)
|
|
||||||
return provider
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture(scope="module")
|
|
||||||
def page_url() -> str:
|
|
||||||
fd, path = tempfile.mkstemp(suffix=".html", prefix="nanobot_dom_")
|
|
||||||
Path(path).write_text(_TEST_PAGE, encoding="utf-8")
|
|
||||||
os.close(fd)
|
|
||||||
yield f"file://{path}"
|
|
||||||
try:
|
|
||||||
os.unlink(path)
|
|
||||||
except OSError:
|
|
||||||
pass
|
|
||||||
|
|
||||||
|
|
||||||
def _tool(start_url: str) -> BrowserTool:
|
|
||||||
# No screenshot — pure DOM, to prove vision is not required.
|
|
||||||
return BrowserTool(start_url=start_url, headless=True, include_screenshot=False)
|
|
||||||
|
|
||||||
|
|
||||||
async def _run(provider, tool: BrowserTool, objective: str, max_iterations: int = 12):
|
|
||||||
tools = ToolRegistry()
|
|
||||||
tools.register(tool)
|
|
||||||
return await AgentRunner(provider).run(AgentRunSpec(
|
|
||||||
initial_messages=[{"role": "user", "content": objective}],
|
|
||||||
tools=tools,
|
|
||||||
model=provider.get_default_model(),
|
|
||||||
max_iterations=max_iterations,
|
|
||||||
max_tool_result_chars=_MAX_TOOL_RESULT_CHARS,
|
|
||||||
))
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_dom_click_button(model: str, page_url: str):
|
|
||||||
attempts = max(1, int(os.getenv("COMPUTER_USE_E2E_ATTEMPTS", "2")))
|
|
||||||
ok = False
|
|
||||||
for _ in range(attempts):
|
|
||||||
tool = _tool(page_url)
|
|
||||||
try:
|
|
||||||
await _run(
|
|
||||||
_make_provider(model),
|
|
||||||
tool,
|
|
||||||
"The page is already open. Use the browser tool (start with action=snapshot) "
|
|
||||||
"to find and click the 'Submit' button.",
|
|
||||||
)
|
|
||||||
backend = await tool._get_backend()
|
|
||||||
status = await backend._page.evaluate("document.getElementById('status').textContent")
|
|
||||||
if status == "SUBMITTED":
|
|
||||||
ok = True
|
|
||||||
break
|
|
||||||
finally:
|
|
||||||
await tool.close()
|
|
||||||
assert ok, f"[{model}] DOM click should set #status to SUBMITTED"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_dom_type_and_submit(model: str, page_url: str):
|
|
||||||
attempts = max(1, int(os.getenv("COMPUTER_USE_E2E_ATTEMPTS", "2")))
|
|
||||||
ok = False
|
|
||||||
for _ in range(attempts):
|
|
||||||
tool = _tool(page_url)
|
|
||||||
try:
|
|
||||||
await _run(
|
|
||||||
_make_provider(model),
|
|
||||||
tool,
|
|
||||||
"The page is already open. Use the browser tool: snapshot the page, type "
|
|
||||||
"'Ada' into the name input, then click the 'Greet' button.",
|
|
||||||
max_iterations=16,
|
|
||||||
)
|
|
||||||
backend = await tool._get_backend()
|
|
||||||
greeting = await backend._page.evaluate(
|
|
||||||
"document.getElementById('greeting').textContent"
|
|
||||||
)
|
|
||||||
if greeting and "ADA" in greeting.upper():
|
|
||||||
ok = True
|
|
||||||
break
|
|
||||||
finally:
|
|
||||||
await tool.close()
|
|
||||||
assert ok, f"[{model}] DOM type+greet should produce a greeting with the typed name"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_dom_read_text_no_vision(model: str, page_url: str):
|
|
||||||
"""Pure text path: the model reads the page via read_text (no screenshot at all)."""
|
|
||||||
attempts = max(1, int(os.getenv("COMPUTER_USE_E2E_ATTEMPTS", "2")))
|
|
||||||
ok = False
|
|
||||||
for _ in range(attempts):
|
|
||||||
tool = _tool(page_url)
|
|
||||||
try:
|
|
||||||
result = await _run(
|
|
||||||
_make_provider(model),
|
|
||||||
tool,
|
|
||||||
"Use the browser tool's read_text action to read the page, then tell me the "
|
|
||||||
"number shown. Reply with just the number.",
|
|
||||||
)
|
|
||||||
if result.final_content and "42" in result.final_content:
|
|
||||||
ok = True
|
|
||||||
break
|
|
||||||
finally:
|
|
||||||
await tool.close()
|
|
||||||
assert ok, f"[{model}] should read '42' via read_text (no vision)"
|
|
||||||
@@ -1,205 +0,0 @@
|
|||||||
"""End-to-end computer_use tests: REAL vision models (via OpenRouter) driving the
|
|
||||||
REAL Playwright browser backend through the full AgentRunner loop.
|
|
||||||
|
|
||||||
These are gated and excluded from normal CI. To run:
|
|
||||||
|
|
||||||
pip install 'nanobot-ai[computer-use]'
|
|
||||||
playwright install chromium
|
|
||||||
COMPUTER_USE_E2E=1 OPENROUTER_API_KEY=sk-or-... \
|
|
||||||
pytest tests/e2e/test_computer_use_e2e.py -v -s
|
|
||||||
|
|
||||||
Model selection (must be vision + tool-calling capable on OpenRouter):
|
|
||||||
COMPUTER_USE_E2E_MODELS="anthropic/claude-sonnet-4.5,openai/gpt-4o,google/gemini-2.5-pro"
|
|
||||||
|
|
||||||
Vision models are not deterministic, so each scenario is allowed a few attempts
|
|
||||||
(COMPUTER_USE_E2E_ATTEMPTS, default 2) and passes if any attempt succeeds.
|
|
||||||
|
|
||||||
NOTE: This harness is model-agnostic by design — every model goes through the
|
|
||||||
same OpenAI-compatible OpenRouter endpoint, and screenshots are delivered as
|
|
||||||
follow-up user messages (see runner._split_tool_result_media), so no
|
|
||||||
provider-specific code path is exercised.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
|
||||||
import tempfile
|
|
||||||
from pathlib import Path
|
|
||||||
from types import SimpleNamespace
|
|
||||||
|
|
||||||
import pytest
|
|
||||||
|
|
||||||
from nanobot.agent.runner import AgentRunner, AgentRunSpec
|
|
||||||
from nanobot.agent.tools.computer_use import ComputerUseTool
|
|
||||||
from nanobot.agent.tools.registry import ToolRegistry
|
|
||||||
from nanobot.config.schema import AgentDefaults
|
|
||||||
|
|
||||||
pytestmark = pytest.mark.skipif(
|
|
||||||
not (os.getenv("COMPUTER_USE_E2E") and os.getenv("OPENROUTER_API_KEY")),
|
|
||||||
reason="set COMPUTER_USE_E2E=1 and OPENROUTER_API_KEY to run computer_use e2e tests",
|
|
||||||
)
|
|
||||||
|
|
||||||
# Default to a computer-use-trained model that reliably grounds pixel
|
|
||||||
# coordinates. Pass COMPUTER_USE_E2E_MODELS to test others — but note that
|
|
||||||
# general/GUI VLMs reliably PERCEIVE the screen (test_read_screen) yet usually
|
|
||||||
# miss pixel-precise clicks (see tests/e2e/README.md "Findings").
|
|
||||||
_DEFAULT_MODELS = [
|
|
||||||
"anthropic/claude-sonnet-4.5",
|
|
||||||
]
|
|
||||||
_MAX_TOOL_RESULT_CHARS = AgentDefaults().max_tool_result_chars
|
|
||||||
|
|
||||||
_TEST_PAGE = """<!doctype html>
|
|
||||||
<html><head><meta charset="utf-8"><title>nanobot cu e2e</title>
|
|
||||||
<style>
|
|
||||||
body{font-family:sans-serif;margin:40px;background:#fff;color:#111}
|
|
||||||
#number{font-size:120px;font-weight:bold;color:#0a0}
|
|
||||||
button{font-size:28px;padding:16px 32px;margin-top:24px}
|
|
||||||
#status{font-size:32px;margin-top:24px;color:#c00}
|
|
||||||
input{font-size:28px;padding:8px}
|
|
||||||
</style></head>
|
|
||||||
<body>
|
|
||||||
<div id="number">42</div>
|
|
||||||
<button id="submit" onclick="document.getElementById('status').textContent='SUBMITTED'">Submit</button>
|
|
||||||
<div id="status">idle</div>
|
|
||||||
<hr>
|
|
||||||
<input id="name" placeholder="your name">
|
|
||||||
<button id="greet" onclick="document.getElementById('greeting').textContent='HELLO '+document.getElementById('name').value">Greet</button>
|
|
||||||
<div id="greeting"></div>
|
|
||||||
</body></html>
|
|
||||||
"""
|
|
||||||
|
|
||||||
|
|
||||||
def _models() -> list[str]:
|
|
||||||
env = os.getenv("COMPUTER_USE_E2E_MODELS", "").strip()
|
|
||||||
return [m.strip() for m in env.split(",") if m.strip()] or _DEFAULT_MODELS
|
|
||||||
|
|
||||||
|
|
||||||
def _attempts() -> int:
|
|
||||||
try:
|
|
||||||
return max(1, int(os.getenv("COMPUTER_USE_E2E_ATTEMPTS", "2")))
|
|
||||||
except ValueError:
|
|
||||||
return 2
|
|
||||||
|
|
||||||
|
|
||||||
def _make_provider(model: str):
|
|
||||||
from nanobot.providers.openai_compat_provider import OpenAICompatProvider
|
|
||||||
|
|
||||||
provider = OpenAICompatProvider(
|
|
||||||
api_key=os.environ["OPENROUTER_API_KEY"],
|
|
||||||
api_base="https://openrouter.ai/api/v1",
|
|
||||||
default_model=model,
|
|
||||||
)
|
|
||||||
# The runner reads provider.generation for sampling params (mirrors factory).
|
|
||||||
provider.generation = SimpleNamespace(max_tokens=2048, temperature=0.0, reasoning_effort=None)
|
|
||||||
return provider
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture(scope="module")
|
|
||||||
def page_url() -> str:
|
|
||||||
fd, path = tempfile.mkstemp(suffix=".html", prefix="nanobot_cu_")
|
|
||||||
Path(path).write_text(_TEST_PAGE, encoding="utf-8")
|
|
||||||
os.close(fd)
|
|
||||||
yield f"file://{path}"
|
|
||||||
try:
|
|
||||||
os.unlink(path)
|
|
||||||
except OSError:
|
|
||||||
pass
|
|
||||||
|
|
||||||
|
|
||||||
async def _run(provider, tool: ComputerUseTool, objective: str, max_iterations: int = 12):
|
|
||||||
tools = ToolRegistry()
|
|
||||||
tools.register(tool)
|
|
||||||
runner = AgentRunner(provider)
|
|
||||||
return await runner.run(AgentRunSpec(
|
|
||||||
initial_messages=[{"role": "user", "content": objective}],
|
|
||||||
tools=tools,
|
|
||||||
model=provider.get_default_model(),
|
|
||||||
max_iterations=max_iterations,
|
|
||||||
max_tool_result_chars=_MAX_TOOL_RESULT_CHARS,
|
|
||||||
))
|
|
||||||
|
|
||||||
|
|
||||||
def _browser_tool(start_url: str) -> ComputerUseTool:
|
|
||||||
return ComputerUseTool(
|
|
||||||
backend="browser",
|
|
||||||
target_width=1280,
|
|
||||||
target_height=800,
|
|
||||||
start_url=start_url,
|
|
||||||
headless=True,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_read_screen(model: str, page_url: str):
|
|
||||||
"""The model must read the number rendered on the page from a screenshot."""
|
|
||||||
attempts = _attempts()
|
|
||||||
ok = False
|
|
||||||
for _ in range(attempts):
|
|
||||||
tool = _browser_tool(page_url)
|
|
||||||
try:
|
|
||||||
result = await _run(
|
|
||||||
_make_provider(model),
|
|
||||||
tool,
|
|
||||||
"Use the computer_use tool: take a screenshot of the page and tell me "
|
|
||||||
"the big number shown. Reply with just the number.",
|
|
||||||
)
|
|
||||||
if result.final_content and "42" in result.final_content:
|
|
||||||
ok = True
|
|
||||||
break
|
|
||||||
finally:
|
|
||||||
await tool.close()
|
|
||||||
assert ok, f"[{model}] expected the model to read '42' from the screenshot"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_click_button(model: str, page_url: str):
|
|
||||||
"""The model must locate and click the Submit button (asserted via the DOM)."""
|
|
||||||
attempts = _attempts()
|
|
||||||
ok = False
|
|
||||||
for _ in range(attempts):
|
|
||||||
tool = _browser_tool(page_url)
|
|
||||||
try:
|
|
||||||
await _run(
|
|
||||||
_make_provider(model),
|
|
||||||
tool,
|
|
||||||
"Use the computer_use tool to click the 'Submit' button on the page. "
|
|
||||||
"Take a screenshot first to locate it.",
|
|
||||||
)
|
|
||||||
backend = await tool._get_backend()
|
|
||||||
status = await backend._page.evaluate("document.getElementById('status').textContent")
|
|
||||||
if status == "SUBMITTED":
|
|
||||||
ok = True
|
|
||||||
break
|
|
||||||
finally:
|
|
||||||
await tool.close()
|
|
||||||
assert ok, f"[{model}] expected #status to become 'SUBMITTED' after clicking Submit"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
@pytest.mark.parametrize("model", _models())
|
|
||||||
async def test_type_and_submit(model: str, page_url: str):
|
|
||||||
"""The model must type into a field and click a button (multi-step)."""
|
|
||||||
attempts = _attempts()
|
|
||||||
ok = False
|
|
||||||
for _ in range(attempts):
|
|
||||||
tool = _browser_tool(page_url)
|
|
||||||
try:
|
|
||||||
await _run(
|
|
||||||
_make_provider(model),
|
|
||||||
tool,
|
|
||||||
"Use the computer_use tool: click the name input, type 'Ada', then click "
|
|
||||||
"the 'Greet' button. Take screenshots to guide yourself.",
|
|
||||||
max_iterations=16,
|
|
||||||
)
|
|
||||||
backend = await tool._get_backend()
|
|
||||||
greeting = await backend._page.evaluate(
|
|
||||||
"document.getElementById('greeting').textContent"
|
|
||||||
)
|
|
||||||
if greeting and "ADA" in greeting.upper():
|
|
||||||
ok = True
|
|
||||||
break
|
|
||||||
finally:
|
|
||||||
await tool.close()
|
|
||||||
assert ok, f"[{model}] expected greeting to contain the typed name"
|
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
from nanobot.providers.openai_compat_provider import OpenAICompatProvider
|
||||||
|
|
||||||
|
|
||||||
|
def test_chat_completions_moves_tool_images_after_parallel_results():
|
||||||
|
image = {
|
||||||
|
"type": "image_url",
|
||||||
|
"image_url": {"url": "data:image/png;base64,AAAA"},
|
||||||
|
"_meta": {"path": "screen.png"},
|
||||||
|
}
|
||||||
|
messages = [
|
||||||
|
{"role": "assistant", "content": None, "tool_calls": [{"id": "a"}, {"id": "b"}]},
|
||||||
|
{
|
||||||
|
"role": "tool",
|
||||||
|
"tool_call_id": "a",
|
||||||
|
"content": [image, {"type": "text", "text": "clicked"}],
|
||||||
|
},
|
||||||
|
{"role": "tool", "tool_call_id": "b", "content": "other result"},
|
||||||
|
{"role": "assistant", "content": "done"},
|
||||||
|
]
|
||||||
|
|
||||||
|
result = OpenAICompatProvider._move_tool_images_to_user(messages)
|
||||||
|
|
||||||
|
assert result[1]["content"] == "clicked"
|
||||||
|
assert result[2] == messages[2]
|
||||||
|
assert result[3]["role"] == "user"
|
||||||
|
assert result[3]["content"][0] == {
|
||||||
|
"type": "image_url",
|
||||||
|
"image_url": {"url": "data:image/png;base64,AAAA"},
|
||||||
|
}
|
||||||
|
assert result[4] == messages[3]
|
||||||
|
|
||||||
|
|
||||||
|
def test_chat_completions_merges_tool_images_into_following_user_message():
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "tool",
|
||||||
|
"tool_call_id": "a",
|
||||||
|
"content": [
|
||||||
|
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}},
|
||||||
|
],
|
||||||
|
},
|
||||||
|
{"role": "user", "content": "continue"},
|
||||||
|
]
|
||||||
|
|
||||||
|
result = OpenAICompatProvider._move_tool_images_to_user(messages)
|
||||||
|
|
||||||
|
assert len(result) == 2
|
||||||
|
assert result[0]["content"] == "(image returned)"
|
||||||
|
assert result[1]["role"] == "user"
|
||||||
|
assert result[1]["content"][-1] == {"type": "text", "text": "continue"}
|
||||||
@@ -1,16 +1,16 @@
|
|||||||
"""Tests for the DOM-based browser tool (model-agnostic web actions).
|
"""Tests for DOM-based browser control."""
|
||||||
|
|
||||||
Backend is a duck-typed fake, so no playwright is needed (only Pillow for the
|
|
||||||
optional screenshot path)."""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import io
|
import io
|
||||||
from unittest.mock import MagicMock
|
from types import SimpleNamespace
|
||||||
|
from unittest.mock import AsyncMock, MagicMock
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from nanobot.agent.tools.browser_tool import BrowserTool, BrowserToolConfig
|
from nanobot.agent.tools.browser_tool import BrowserTool, BrowserToolConfig
|
||||||
|
from nanobot.agent.tools.computer_use_backends import browser_playwright
|
||||||
|
from nanobot.agent.tools.computer_use_backends.browser_playwright import BrowserBackend
|
||||||
|
|
||||||
|
|
||||||
class _FakeDomBackend:
|
class _FakeDomBackend:
|
||||||
@@ -65,7 +65,18 @@ class _FakeDomBackend:
|
|||||||
|
|
||||||
def _tool(**kw):
|
def _tool(**kw):
|
||||||
fb = _FakeDomBackend()
|
fb = _FakeDomBackend()
|
||||||
return BrowserTool(backend_impl=fb, **kw), fb
|
return BrowserTool(BrowserToolConfig(**kw), backend_impl=fb), fb
|
||||||
|
|
||||||
|
|
||||||
|
def _route(url: str, *, navigation: bool):
|
||||||
|
return SimpleNamespace(
|
||||||
|
request=SimpleNamespace(
|
||||||
|
url=url,
|
||||||
|
is_navigation_request=MagicMock(return_value=navigation),
|
||||||
|
),
|
||||||
|
abort=AsyncMock(),
|
||||||
|
continue_=AsyncMock(),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class TestConfigAndMetadata:
|
class TestConfigAndMetadata:
|
||||||
@@ -88,7 +99,7 @@ class TestConfigAndMetadata:
|
|||||||
ctx.config.browser = BrowserToolConfig(enable=True, allowed_domains=["example.com"])
|
ctx.config.browser = BrowserToolConfig(enable=True, allowed_domains=["example.com"])
|
||||||
tool = BrowserTool.create(ctx)
|
tool = BrowserTool.create(ctx)
|
||||||
assert isinstance(tool, BrowserTool)
|
assert isinstance(tool, BrowserTool)
|
||||||
assert tool.allowed_domains == ["example.com"]
|
assert tool.config.allowed_domains == ["example.com"]
|
||||||
|
|
||||||
def test_metadata(self):
|
def test_metadata(self):
|
||||||
tool, _ = _tool()
|
tool, _ = _tool()
|
||||||
@@ -117,26 +128,21 @@ class TestDispatch:
|
|||||||
assert '[2] input[text] "your name"' in result
|
assert '[2] input[text] "your name"' in result
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_click_by_ref(self):
|
@pytest.mark.parametrize(
|
||||||
|
("action", "kwargs", "expected"),
|
||||||
|
[
|
||||||
|
("click", {"ref": 1}, ("click", 1)),
|
||||||
|
("type", {"ref": 2, "text": "Ada", "submit": True}, ("fill", 2, "Ada", True)),
|
||||||
|
("select", {"ref": 2, "value": "opt1"}, ("select", 2, "opt1")),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_element_actions(self, action, kwargs, expected):
|
||||||
tool, fb = _tool()
|
tool, fb = _tool()
|
||||||
result = await tool.execute(action="click", ref=1)
|
result = await tool.execute(action=action, **kwargs)
|
||||||
assert ("click", 1) in fb.calls
|
assert expected in fb.calls
|
||||||
assert "Clicked element [1]" in result
|
|
||||||
# fresh snapshot returned so refs stay current
|
|
||||||
assert "Interactive elements" in result
|
assert "Interactive elements" in result
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_type_with_submit(self):
|
|
||||||
tool, fb = _tool()
|
|
||||||
await tool.execute(action="type", ref=2, text="Ada", submit=True)
|
|
||||||
assert ("fill", 2, "Ada", True) in fb.calls
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_select(self):
|
|
||||||
tool, fb = _tool()
|
|
||||||
await tool.execute(action="select", ref=2, value="opt1")
|
|
||||||
assert ("select", 2, "opt1") in fb.calls
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_scroll_and_key_and_back(self):
|
async def test_scroll_and_key_and_back(self):
|
||||||
tool, fb = _tool()
|
tool, fb = _tool()
|
||||||
@@ -167,27 +173,115 @@ class TestDispatch:
|
|||||||
|
|
||||||
|
|
||||||
class TestErrorsAndPolicy:
|
class TestErrorsAndPolicy:
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
("kwargs", "error"),
|
||||||
|
[
|
||||||
|
({"action": "teleport"}, "unknown action"),
|
||||||
|
({"action": "click"}, "requires an element 'ref'"),
|
||||||
|
],
|
||||||
|
)
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_unknown_action(self):
|
async def test_tool_errors_are_returned_to_model(self, kwargs, error):
|
||||||
tool, _ = _tool()
|
tool, _ = _tool()
|
||||||
result = await tool.execute(action="teleport")
|
result = await tool.execute(**kwargs)
|
||||||
assert isinstance(result, str) and "unknown action" in result
|
assert isinstance(result, str) and error in result
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_click_requires_ref(self):
|
async def test_backend_blocks_disallowed_navigation(self):
|
||||||
tool, _ = _tool()
|
backend = BrowserBackend(allowed_domains=["example.com"])
|
||||||
result = await tool.execute(action="click")
|
page = SimpleNamespace(goto=AsyncMock())
|
||||||
assert isinstance(result, str) and "requires an element 'ref'" in result
|
backend._page = page
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="allowed_domains"):
|
||||||
|
await backend.navigate("https://evil.test/")
|
||||||
|
|
||||||
|
page.goto.assert_not_awaited()
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_navigate_blocked_by_allowlist(self):
|
async def test_backend_allows_subdomain_navigation(self, monkeypatch: pytest.MonkeyPatch):
|
||||||
tool, fb = _tool(allowed_domains=["example.com"])
|
check = MagicMock(return_value=(True, ""))
|
||||||
result = await tool.execute(action="navigate", url="https://evil.test/")
|
monkeypatch.setattr(browser_playwright, "validate_url_target", check)
|
||||||
assert "blocked by the allowed_domains" in result
|
backend = BrowserBackend(allowed_domains=["example.com"])
|
||||||
assert not any(c[0] == "navigate" for c in fb.calls)
|
page = SimpleNamespace(goto=AsyncMock())
|
||||||
|
backend._page = page
|
||||||
|
|
||||||
|
await backend.navigate("https://app.example.com/x")
|
||||||
|
|
||||||
|
page.goto.assert_awaited_once_with("https://app.example.com/x")
|
||||||
|
check.assert_called_once()
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"url",
|
||||||
|
[
|
||||||
|
"file:///etc/passwd",
|
||||||
|
"http://127.0.0.1/",
|
||||||
|
"http://169.254.169.254/latest/meta-data/",
|
||||||
|
"ws://localhost/socket",
|
||||||
|
],
|
||||||
|
)
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_browser_network_policy_blocks_local_targets(self, url: str):
|
||||||
|
backend = BrowserBackend()
|
||||||
|
with pytest.raises(ValueError, match="blocked"):
|
||||||
|
await backend.navigate(url)
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_navigate_allowed_subdomain(self):
|
async def test_backend_intercepts_blocked_navigation(self):
|
||||||
tool, fb = _tool(allowed_domains=["example.com"])
|
backend = BrowserBackend(allowed_domains=["example.com"])
|
||||||
await tool.execute(action="navigate", url="https://app.example.com/x")
|
route = _route("https://evil.test/", navigation=True)
|
||||||
assert ("navigate", "https://app.example.com/x") in fb.calls
|
|
||||||
|
await backend._route_request(route)
|
||||||
|
|
||||||
|
route.abort.assert_awaited_once_with("blockedbyclient")
|
||||||
|
route.continue_.assert_not_awaited()
|
||||||
|
assert "allowed_domains" in (backend.pop_blocked_navigation() or "")
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_backend_intercepts_private_subresource(self):
|
||||||
|
backend = BrowserBackend()
|
||||||
|
route = _route(
|
||||||
|
"http://169.254.169.254/latest/meta-data/",
|
||||||
|
navigation=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
await backend._route_request(route)
|
||||||
|
|
||||||
|
route.abort.assert_awaited_once_with("blockedbyclient")
|
||||||
|
assert backend.pop_blocked_navigation() is None
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_backend_does_not_apply_navigation_allowlist_to_subresources(
|
||||||
|
self, monkeypatch: pytest.MonkeyPatch
|
||||||
|
):
|
||||||
|
monkeypatch.setattr(
|
||||||
|
browser_playwright,
|
||||||
|
"validate_url_target",
|
||||||
|
MagicMock(return_value=(True, "")),
|
||||||
|
)
|
||||||
|
backend = BrowserBackend(allowed_domains=["example.com"])
|
||||||
|
route = _route("https://cdn.other.test/app.js", navigation=False)
|
||||||
|
|
||||||
|
await backend._route_request(route)
|
||||||
|
|
||||||
|
route.continue_.assert_awaited_once()
|
||||||
|
route.abort.assert_not_awaited()
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_backend_intercepts_private_websocket(self):
|
||||||
|
backend = BrowserBackend()
|
||||||
|
web_socket = SimpleNamespace(
|
||||||
|
url="ws://127.0.0.1/socket",
|
||||||
|
close=AsyncMock(),
|
||||||
|
connect_to_server=AsyncMock(),
|
||||||
|
)
|
||||||
|
|
||||||
|
await backend._route_web_socket(web_socket)
|
||||||
|
|
||||||
|
web_socket.close.assert_awaited_once()
|
||||||
|
web_socket.connect_to_server.assert_not_awaited()
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_backend_rejects_file_start_url_before_launch(self):
|
||||||
|
backend = BrowserBackend(start_url="file:///etc/passwd")
|
||||||
|
with pytest.raises(ValueError, match="start_url is blocked"):
|
||||||
|
await backend.dimensions()
|
||||||
|
|||||||
@@ -1,19 +1,19 @@
|
|||||||
"""Tests for the computer_use tool, its config wiring, and coordinate scaling.
|
"""Tests for screenshot-based computer control."""
|
||||||
|
|
||||||
Backend actuation is exercised through an injected fake backend, so these tests
|
|
||||||
need neither pyautogui nor playwright (only Pillow, for screenshot downscaling).
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import io
|
import io
|
||||||
|
import sys
|
||||||
|
from types import SimpleNamespace
|
||||||
from unittest.mock import MagicMock
|
from unittest.mock import MagicMock
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from nanobot.agent.tools.computer_use import ComputerUseTool, ComputerUseToolConfig
|
from nanobot.agent.tools.computer_use import ComputerUseTool, ComputerUseToolConfig
|
||||||
from nanobot.agent.tools.computer_use_backends.base import ComputerBackend
|
from nanobot.agent.tools.computer_use_backends.base import ComputerBackend, SessionBackendPool
|
||||||
from nanobot.utils.screen_scale import ScreenScaler, fit_target_size
|
from nanobot.agent.tools.computer_use_backends.desktop_pyautogui import DesktopBackend
|
||||||
|
from nanobot.agent.tools.context import RequestContext, request_context
|
||||||
|
from nanobot.config.schema import ToolsConfig
|
||||||
|
|
||||||
|
|
||||||
class _FakeBackend(ComputerBackend):
|
class _FakeBackend(ComputerBackend):
|
||||||
@@ -24,6 +24,7 @@ class _FakeBackend(ComputerBackend):
|
|||||||
def __init__(self, width: int = 2560, height: int = 1600):
|
def __init__(self, width: int = 2560, height: int = 1600):
|
||||||
self.calls: list[tuple] = []
|
self.calls: list[tuple] = []
|
||||||
self._w, self._h = width, height
|
self._w, self._h = width, height
|
||||||
|
self.closed = False
|
||||||
|
|
||||||
async def dimensions(self) -> tuple[int, int]:
|
async def dimensions(self) -> tuple[int, int]:
|
||||||
return (self._w, self._h)
|
return (self._w, self._h)
|
||||||
@@ -52,6 +53,9 @@ class _FakeBackend(ComputerBackend):
|
|||||||
|
|
||||||
async def key(self, combo):
|
async def key(self, combo):
|
||||||
self.calls.append(("key", combo))
|
self.calls.append(("key", combo))
|
||||||
|
|
||||||
|
async def close(self):
|
||||||
|
self.closed = True
|
||||||
# navigate() inherited -> raises NotImplementedError (desktop has no navigate)
|
# navigate() inherited -> raises NotImplementedError (desktop has no navigate)
|
||||||
|
|
||||||
|
|
||||||
@@ -64,35 +68,11 @@ def _split(result):
|
|||||||
|
|
||||||
def _tool(**kw):
|
def _tool(**kw):
|
||||||
fb = _FakeBackend(width=kw.pop("w", 2560), height=kw.pop("h", 1600))
|
fb = _FakeBackend(width=kw.pop("w", 2560), height=kw.pop("h", 1600))
|
||||||
tool = ComputerUseTool(backend_impl=fb, target_width=1280, target_height=800, **kw)
|
config = ComputerUseToolConfig(target_width=1280, target_height=800, **kw)
|
||||||
|
tool = ComputerUseTool(config, backend_impl=fb)
|
||||||
return tool, fb
|
return tool, fb
|
||||||
|
|
||||||
|
|
||||||
# --------------------------- coordinate scaling (pure) ---------------------------
|
|
||||||
|
|
||||||
class TestScreenScale:
|
|
||||||
def test_fit_never_upscales(self):
|
|
||||||
assert fit_target_size(800, 600, 1280, 800) == (800, 600)
|
|
||||||
|
|
||||||
def test_fit_preserves_aspect(self):
|
|
||||||
# 2560x1600 (16:10) into 1280x800 box -> exactly halved.
|
|
||||||
assert fit_target_size(2560, 1600, 1280, 800) == (1280, 800)
|
|
||||||
|
|
||||||
def test_fit_landscape_into_box(self):
|
|
||||||
# 3000x1000 into 1280x800 -> width-bound: scale 1280/3000.
|
|
||||||
w, h = fit_target_size(3000, 1000, 1280, 800)
|
|
||||||
assert w == 1280 and h == round(1000 * 1280 / 3000)
|
|
||||||
|
|
||||||
def test_to_real_scales_up(self):
|
|
||||||
sc = ScreenScaler.for_screen(2560, 1600, 1280, 800)
|
|
||||||
assert sc.to_real(100, 50) == (200, 100)
|
|
||||||
|
|
||||||
def test_to_real_clamps_in_bounds(self):
|
|
||||||
sc = ScreenScaler.for_screen(1000, 1000, 1000, 1000)
|
|
||||||
assert sc.to_real(5000, 5000) == (999, 999)
|
|
||||||
assert sc.to_real(-10, -10) == (0, 0)
|
|
||||||
|
|
||||||
|
|
||||||
# --------------------------- config + metadata ---------------------------
|
# --------------------------- config + metadata ---------------------------
|
||||||
|
|
||||||
class TestConfigAndMetadata:
|
class TestConfigAndMetadata:
|
||||||
@@ -101,7 +81,19 @@ class TestConfigAndMetadata:
|
|||||||
assert cfg.enable is False
|
assert cfg.enable is False
|
||||||
assert cfg.backend == "desktop"
|
assert cfg.backend == "desktop"
|
||||||
assert (cfg.target_width, cfg.target_height) == (1280, 800)
|
assert (cfg.target_width, cfg.target_height) == (1280, 800)
|
||||||
assert cfg.require_approval is True
|
assert "require_approval" not in type(cfg).model_fields
|
||||||
|
|
||||||
|
def test_tools_config_accepts_camel_case(self):
|
||||||
|
cfg = ToolsConfig.model_validate({
|
||||||
|
"browser": {"enable": True},
|
||||||
|
"computerUse": {"enable": True, "backend": "browser"},
|
||||||
|
})
|
||||||
|
|
||||||
|
assert cfg.browser.enable is True
|
||||||
|
assert cfg.computer_use.enable is True
|
||||||
|
assert cfg.computer_use.backend == "browser"
|
||||||
|
dumped = cfg.model_dump(by_alias=True)
|
||||||
|
assert "computerUse" in dumped
|
||||||
|
|
||||||
def test_enabled_reads_config(self):
|
def test_enabled_reads_config(self):
|
||||||
ctx = MagicMock()
|
ctx = MagicMock()
|
||||||
@@ -117,8 +109,8 @@ class TestConfigAndMetadata:
|
|||||||
)
|
)
|
||||||
tool = ComputerUseTool.create(ctx)
|
tool = ComputerUseTool.create(ctx)
|
||||||
assert isinstance(tool, ComputerUseTool)
|
assert isinstance(tool, ComputerUseTool)
|
||||||
assert tool.backend_name == "browser"
|
assert tool.config.backend == "browser"
|
||||||
assert (tool.target_width, tool.target_height) == (1024, 768)
|
assert (tool.config.target_width, tool.config.target_height) == (1024, 768)
|
||||||
|
|
||||||
def test_tool_metadata(self):
|
def test_tool_metadata(self):
|
||||||
tool, _ = _tool()
|
tool, _ = _tool()
|
||||||
@@ -159,49 +151,35 @@ class TestExecute:
|
|||||||
assert "left_click at (200, 100)" in texts[-1]["text"]
|
assert "left_click at (200, 100)" in texts[-1]["text"]
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_click_accepts_coordinate_array(self):
|
async def test_click_clamps_coordinates_to_screen(self):
|
||||||
# Some models emit a combined [x, y] array (Anthropic's native convention).
|
|
||||||
tool, fb = _tool()
|
tool, fb = _tool()
|
||||||
await tool.execute(action="left_click", x=[100, 50])
|
await tool.execute(action="left_click", x=5000, y=-10)
|
||||||
assert fb.calls == [("click", 200, 100, "left", 1)]
|
assert fb.calls == [("click", 2559, 0, "left", 1)]
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_double_and_triple_click_counts(self):
|
@pytest.mark.parametrize(
|
||||||
tool, fb = _tool()
|
("action", "kwargs", "expected"),
|
||||||
await tool.execute(action="double_click", x=10, y=10)
|
[
|
||||||
await tool.execute(action="triple_click", x=10, y=10)
|
("double_click", {"x": 10, "y": 10}, ("click", 20, 20, "left", 2)),
|
||||||
assert fb.calls[0] == ("click", 20, 20, "left", 2)
|
("triple_click", {"x": 10, "y": 10}, ("click", 20, 20, "left", 3)),
|
||||||
assert fb.calls[1] == ("click", 20, 20, "left", 3)
|
("right_click", {"x": 5, "y": 5}, ("click", 10, 10, "right", 1)),
|
||||||
|
("middle_click", {"x": 5, "y": 5}, ("click", 10, 10, "middle", 1)),
|
||||||
|
(
|
||||||
|
"scroll",
|
||||||
|
{"x": 100, "y": 100, "scroll_direction": "down", "scroll_amount": 5},
|
||||||
|
("scroll", 200, 200, "down", 5),
|
||||||
|
),
|
||||||
|
("type", {"text": "hello"}, ("type", "hello")),
|
||||||
|
("key", {"text": "ctrl+s"}, ("key", "ctrl+s")),
|
||||||
|
("mouse_move", {"x": 10, "y": 10}, ("move", 20, 20)),
|
||||||
|
("left_click_drag", {"x": 20, "y": 30}, ("drag", 40, 60)),
|
||||||
|
],
|
||||||
|
)
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_right_and_middle_click_buttons(self):
|
async def test_actions_dispatch_to_backend(self, action, kwargs, expected):
|
||||||
tool, fb = _tool()
|
tool, fb = _tool()
|
||||||
await tool.execute(action="right_click", x=5, y=5)
|
await tool.execute(action=action, **kwargs)
|
||||||
await tool.execute(action="middle_click", x=5, y=5)
|
assert fb.calls == [expected]
|
||||||
assert fb.calls[0][3] == "right"
|
|
||||||
assert fb.calls[1][3] == "middle"
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_scroll_defaults_and_args(self):
|
|
||||||
tool, fb = _tool()
|
|
||||||
await tool.execute(action="scroll", x=100, y=100, scroll_direction="down", scroll_amount=5)
|
|
||||||
assert fb.calls == [("scroll", 200, 200, "down", 5)]
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_type_and_key(self):
|
|
||||||
tool, fb = _tool()
|
|
||||||
await tool.execute(action="type", text="hello")
|
|
||||||
await tool.execute(action="key", text="ctrl+s")
|
|
||||||
assert ("type", "hello") in fb.calls
|
|
||||||
assert ("key", "ctrl+s") in fb.calls
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_drag_and_move(self):
|
|
||||||
tool, fb = _tool()
|
|
||||||
await tool.execute(action="mouse_move", x=10, y=10)
|
|
||||||
await tool.execute(action="left_click_drag", x=20, y=30)
|
|
||||||
assert ("move", 20, 20) in fb.calls
|
|
||||||
assert ("drag", 40, 60) in fb.calls
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_wait(self):
|
async def test_wait(self):
|
||||||
@@ -210,67 +188,73 @@ class TestExecute:
|
|||||||
_, texts = _split(result)
|
_, texts = _split(result)
|
||||||
assert "Waited" in texts[-1]["text"]
|
assert "Waited" in texts[-1]["text"]
|
||||||
|
|
||||||
# ---- error paths return a plain string (model can self-correct) ----
|
@pytest.mark.parametrize(
|
||||||
|
("kwargs", "error"),
|
||||||
|
[
|
||||||
|
({"action": "frobnicate"}, "unknown action"),
|
||||||
|
({"action": "left_click"}, "requires"),
|
||||||
|
({"action": "navigate", "url": "https://example.com"}, "Error"),
|
||||||
|
],
|
||||||
|
)
|
||||||
@pytest.mark.asyncio
|
@pytest.mark.asyncio
|
||||||
async def test_unknown_action_errors(self):
|
async def test_errors_are_returned_to_model(self, kwargs, error):
|
||||||
tool, _ = _tool()
|
tool, _ = _tool()
|
||||||
result = await tool.execute(action="frobnicate")
|
result = await tool.execute(**kwargs)
|
||||||
assert isinstance(result, str) and "unknown action" in result
|
assert isinstance(result, str) and error in result
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_click_requires_coordinates(self):
|
|
||||||
tool, _ = _tool()
|
|
||||||
result = await tool.execute(action="left_click")
|
|
||||||
assert isinstance(result, str) and "requires" in result
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_navigate_unsupported_on_desktop(self):
|
|
||||||
tool, _ = _tool()
|
|
||||||
result = await tool.execute(action="navigate", url="https://example.com")
|
|
||||||
assert isinstance(result, str) and "Error" in result
|
|
||||||
|
|
||||||
|
|
||||||
class _BrowserFakeBackend(_FakeBackend):
|
@pytest.mark.asyncio
|
||||||
environment = "browser"
|
async def test_backend_pool_isolates_sessions_and_closes_all():
|
||||||
|
created: list[_FakeBackend] = []
|
||||||
|
finalized: list[bool] = []
|
||||||
|
|
||||||
async def navigate(self, url):
|
def factory():
|
||||||
self.calls.append(("navigate", url))
|
backend = _FakeBackend()
|
||||||
|
created.append(backend)
|
||||||
|
return backend
|
||||||
|
|
||||||
|
async def finalize():
|
||||||
|
finalized.append(all(backend.closed for backend in created))
|
||||||
|
|
||||||
|
pool = SessionBackendPool(factory, finalizer=finalize)
|
||||||
|
with request_context(RequestContext(channel="test", chat_id="a", session_key="test:a")):
|
||||||
|
first = pool.get()
|
||||||
|
assert pool.get() is first
|
||||||
|
with request_context(RequestContext(channel="test", chat_id="b", session_key="test:b")):
|
||||||
|
second = pool.get()
|
||||||
|
|
||||||
|
assert first is not second
|
||||||
|
await pool.close()
|
||||||
|
assert len(created) == 2
|
||||||
|
assert all(backend.closed for backend in created)
|
||||||
|
assert finalized == [True]
|
||||||
|
|
||||||
|
await pool.close()
|
||||||
|
assert finalized == [True]
|
||||||
|
|
||||||
|
|
||||||
class TestAllowedDomainsPolicy:
|
@pytest.mark.asyncio
|
||||||
def _browser_tool(self, allowed):
|
async def test_desktop_backend_uses_safe_pyautogui_calls():
|
||||||
fb = _BrowserFakeBackend(width=1280, height=800)
|
pg = MagicMock()
|
||||||
tool = ComputerUseTool(
|
pg.easeInOutQuad = object()
|
||||||
backend_impl=fb,
|
backend = DesktopBackend()
|
||||||
backend="browser",
|
backend._pg = pg
|
||||||
target_width=1280,
|
|
||||||
target_height=800,
|
|
||||||
allowed_domains=allowed,
|
|
||||||
)
|
|
||||||
return tool, fb
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
await backend.drag(10, 20)
|
||||||
async def test_empty_allowlist_allows_all(self):
|
await backend.scroll(10, 20, "down", 3)
|
||||||
tool, fb = self._browser_tool([])
|
|
||||||
await tool.execute(action="navigate", url="https://anything.example/")
|
|
||||||
assert ("navigate", "https://anything.example/") in fb.calls
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
assert pg.dragTo.call_args.kwargs == {
|
||||||
async def test_exact_domain_allowed(self):
|
"duration": 0.3,
|
||||||
tool, fb = self._browser_tool(["example.com"])
|
"tween": pg.easeInOutQuad,
|
||||||
await tool.execute(action="navigate", url="https://example.com/path")
|
"button": "left",
|
||||||
assert any(c[0] == "navigate" for c in fb.calls)
|
}
|
||||||
|
pg.scroll.assert_called_once_with(-3)
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
|
||||||
async def test_subdomain_allowed(self):
|
|
||||||
tool, fb = self._browser_tool(["example.com"])
|
|
||||||
await tool.execute(action="navigate", url="https://app.example.com/")
|
|
||||||
assert any(c[0] == "navigate" for c in fb.calls)
|
|
||||||
|
|
||||||
@pytest.mark.asyncio
|
def test_desktop_backend_preserves_pyautogui_failsafe(monkeypatch):
|
||||||
async def test_disallowed_domain_blocked(self):
|
pg = SimpleNamespace(FAILSAFE=True)
|
||||||
tool, fb = self._browser_tool(["example.com"])
|
monkeypatch.setitem(sys.modules, "pyautogui", pg)
|
||||||
result = await tool.execute(action="navigate", url="https://evil.test/")
|
|
||||||
assert isinstance(result, str) and "blocked by the allowed_domains" in result
|
backend = DesktopBackend()
|
||||||
assert not any(c[0] == "navigate" for c in fb.calls)
|
assert backend._ensure() is pg
|
||||||
|
assert pg.FAILSAFE is True
|
||||||
|
|||||||
@@ -28,6 +28,18 @@ class _FakeTool(Tool):
|
|||||||
async def execute(self, **kwargs: Any) -> Any:
|
async def execute(self, **kwargs: Any) -> Any:
|
||||||
return kwargs
|
return kwargs
|
||||||
|
|
||||||
|
|
||||||
|
class _ClosableTool(_FakeTool):
|
||||||
|
def __init__(self, name: str, *, error: BaseException | None = None):
|
||||||
|
super().__init__(name)
|
||||||
|
self.closed = False
|
||||||
|
self.error = error
|
||||||
|
|
||||||
|
async def close(self) -> None:
|
||||||
|
self.closed = True
|
||||||
|
if self.error is not None:
|
||||||
|
raise self.error
|
||||||
|
|
||||||
def _tool_names(definitions: list[dict[str, Any]]) -> list[str]:
|
def _tool_names(definitions: list[dict[str, Any]]) -> list[str]:
|
||||||
names: list[str] = []
|
names: list[str] = []
|
||||||
for definition in definitions:
|
for definition in definitions:
|
||||||
@@ -58,6 +70,24 @@ def test_get_definitions_orders_builtins_then_mcp_tools() -> None:
|
|||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_close_attempts_every_registered_tool() -> None:
|
||||||
|
registry = ToolRegistry()
|
||||||
|
broken = _ClosableTool("broken", error=RuntimeError("close failed"))
|
||||||
|
healthy = _ClosableTool("healthy")
|
||||||
|
registry.register(broken)
|
||||||
|
registry.register(healthy)
|
||||||
|
|
||||||
|
try:
|
||||||
|
await registry.close()
|
||||||
|
except RuntimeError as exc:
|
||||||
|
assert str(exc) == "close failed"
|
||||||
|
else:
|
||||||
|
raise AssertionError("expected close failure")
|
||||||
|
|
||||||
|
assert broken.closed is True
|
||||||
|
assert healthy.closed is True
|
||||||
|
|
||||||
|
|
||||||
def test_prepare_call_rejects_near_miss_tool_name_with_suggestion() -> None:
|
def test_prepare_call_rejects_near_miss_tool_name_with_suggestion() -> None:
|
||||||
registry = ToolRegistry()
|
registry = ToolRegistry()
|
||||||
registry.register(_FakeTool("read_file"))
|
registry.register(_FakeTool("read_file"))
|
||||||
|
|||||||
Reference in New Issue
Block a user