Merge remote-tracking branch 'origin/main' into pr5636

This commit is contained in:
Magnus Müller
2026-09-04 17:47:55 -07:00
32 changed files with 1299 additions and 427 deletions
+4 -3
View File
@@ -35,7 +35,7 @@ jobs:
if: github.event_name == 'workflow_dispatch' && inputs.create_tag
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- name: Create pre-release tag
run: |
git fetch --tags
@@ -107,9 +107,10 @@ jobs:
exit 1
fi
echo "::notice::Environment '${ENV_NAME}' is protected with prevent_self_review. OK."
- uses: actions/checkout@v4
- uses: astral-sh/setup-uv@v6
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
- uses: astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e # v6
with:
version: "0.12.9"
enable-cache: true
activate-environment: true
- run: uv sync
+13 -18
View File
@@ -521,6 +521,18 @@ class AgentHistory(BaseModel):
return redact_sensitive_string(value, sensitive_values)
def _filter_sensitive_data_from_value(self, value: Any, sensitive_data: dict[str, str | dict[str, str]] | None) -> Any:
"""Recursively filter sensitive data from any supported container or string value"""
if isinstance(value, str):
return self._filter_sensitive_data_from_string(value, sensitive_data)
if isinstance(value, dict):
return {key: self._filter_sensitive_data_from_value(item, sensitive_data) for key, item in value.items()}
if isinstance(value, list):
return [self._filter_sensitive_data_from_value(item, sensitive_data) for item in value]
if isinstance(value, tuple):
return tuple(self._filter_sensitive_data_from_value(item, sensitive_data) for item in value)
return value
def _filter_sensitive_data_from_dict(
self, data: dict[str, Any], sensitive_data: dict[str, str | dict[str, str]] | None
) -> dict[str, Any]:
@@ -528,24 +540,7 @@ class AgentHistory(BaseModel):
if not sensitive_data:
return data
filtered_data = {}
for key, value in data.items():
if isinstance(value, str):
filtered_data[key] = self._filter_sensitive_data_from_string(value, sensitive_data)
elif isinstance(value, dict):
filtered_data[key] = self._filter_sensitive_data_from_dict(value, sensitive_data)
elif isinstance(value, list):
filtered_data[key] = [
self._filter_sensitive_data_from_string(item, sensitive_data)
if isinstance(item, str)
else self._filter_sensitive_data_from_dict(item, sensitive_data)
if isinstance(item, dict)
else item
for item in value
]
else:
filtered_data[key] = value
return filtered_data
return {key: self._filter_sensitive_data_from_value(value, sensitive_data) for key, value in data.items()}
def model_dump(self, sensitive_data: dict[str, str | dict[str, str]] | None = None, **kwargs) -> dict[str, Any]:
"""Custom serialization handling circular references and filtering sensitive data"""
+7 -1
View File
@@ -690,11 +690,15 @@ class BrowserSession(BaseModel):
self.logger.info('✅ Browser session reset complete')
def model_post_init(self, __context) -> None:
"""Register event handlers after model initialization."""
"""Initialize runtime state and register event handlers."""
self._connection_lock = asyncio.Lock()
# Initialize reconnect event as set (no reconnection pending)
self._reconnect_event = asyncio.Event()
self._reconnect_event.set()
self._register_session_event_handlers()
def _register_session_event_handlers(self) -> None:
"""Register BrowserSession handlers on the current event bus."""
# Check if handlers are already registered to prevent duplicates
from browser_use.browser.watchdog_base import BaseWatchdog
@@ -746,6 +750,7 @@ class BrowserSession(BaseModel):
await self.reset()
# Create fresh event bus
self.event_bus = ResilientEventBus()
self._register_session_event_handlers()
async def stop(self) -> None:
"""Stop the browser session without killing the browser process.
@@ -771,6 +776,7 @@ class BrowserSession(BaseModel):
await self.reset()
# Create fresh event bus
self.event_bus = ResilientEventBus()
self._register_session_event_handlers()
async def close(self) -> None:
"""Alias for stop()."""
@@ -2504,31 +2504,34 @@ class DefaultActionWatchdog(BaseWatchdog):
'end': 'End',
}
# Parse and normalize the key string
keys = event.keys
if '+' in keys:
# Handle key combinations like "ctrl+a"
parts = keys.split('+')
normalized_parts = []
for part in parts:
part_lower = part.strip().lower()
normalized = key_aliases.get(part_lower, part)
normalized_parts.append(normalized)
normalized_keys = '+'.join(normalized_parts)
else:
# Single key
keys_lower = keys.strip().lower()
normalized_keys = key_aliases.get(keys_lower, keys)
# Handle key combinations like "Control+A"
if '+' in normalized_keys:
parts = normalized_keys.split('+')
modifiers = parts[:-1]
main_key = parts[-1]
modifier_map = {'Alt': 1, 'Control': 2, 'Meta': 4, 'Shift': 8}
is_combination = False
modifiers = []
main_key = None
if '+' in keys and keys != '+':
if keys.endswith('++'):
prefix = keys[:-2]
raw_modifiers = prefix.split('+')
if all(part.strip() for part in raw_modifiers):
normalized_modifiers = [key_aliases.get(part.strip().lower(), part) for part in raw_modifiers]
if all(modifier in modifier_map for modifier in normalized_modifiers):
is_combination = True
modifiers = normalized_modifiers
main_key = '+'
else:
prefix, suffix = keys.rsplit('+', 1)
raw_modifiers = prefix.split('+')
if suffix.strip() and all(part.strip() for part in raw_modifiers):
normalized_modifiers = [key_aliases.get(part.strip().lower(), part) for part in raw_modifiers]
if all(modifier in modifier_map for modifier in normalized_modifiers):
is_combination = True
modifiers = normalized_modifiers
main_key = key_aliases.get(suffix.strip().lower(), suffix)
if is_combination and main_key is not None:
# Calculate modifier bitmask
modifier_value = 0
modifier_map = {'Alt': 1, 'Control': 2, 'Meta': 4, 'Shift': 8}
for mod in modifiers:
modifier_value |= modifier_map.get(mod, 0)
@@ -2545,6 +2548,9 @@ class DefaultActionWatchdog(BaseWatchdog):
for mod in reversed(modifiers):
await self._dispatch_key_event(cdp_session, 'keyUp', mod)
else:
keys_lower = keys.strip().lower()
normalized_keys = key_aliases.get(keys_lower, keys)
# Check if this is a text string or special key
special_keys = {
'Enter',
@@ -157,7 +157,7 @@ class LocalBrowserWatchdog(BaseWatchdog):
process = psutil.Process(subprocess.pid)
# Wait for CDP to be ready and get the URL
cdp_url = await self._wait_for_cdp_url(debug_port)
cdp_url = await self._wait_for_cdp_url(debug_port, process=process)
# Success! Clean up only the temp dirs we created but didn't use
currently_used_dir = str(profile.user_data_dir)
@@ -405,13 +405,42 @@ class LocalBrowserWatchdog(BaseWatchdog):
return port
@staticmethod
async def _wait_for_cdp_url(port: int, timeout: float = 30) -> str:
"""Wait for the browser to start and return the CDP URL."""
async def _wait_for_cdp_url(port: int, timeout: float = 30, process: psutil.Process | None = None) -> str:
"""Wait for the browser to start and return the CDP URL.
Args:
port: The local port Chrome is listening on for CDP.
timeout: Maximum seconds to wait before raising TimeoutError.
process: Optional psutil.Process for the browser subprocess. If provided,
the loop will fail fast with a descriptive error if the process
exits before CDP becomes available (e.g. due to missing sandbox
capabilities or a missing display on headless Linux).
"""
import aiohttp
start_time = asyncio.get_event_loop().time()
start_time = asyncio.get_running_loop().time()
while asyncio.get_running_loop().time() - start_time < timeout:
# Fail fast if the browser process has already exited
if process is not None:
try:
if not process.is_running():
raise RuntimeError(
f'Browser process (PID {process.pid}) exited before CDP became available on port {port}. '
'This usually means Chrome failed to start — check that it is properly installed and '
'all required system dependencies are present (e.g. --no-sandbox may be needed in '
'Docker/headless environments, or a virtual display such as Xvfb on headless Linux).'
)
except psutil.NoSuchProcess:
raise RuntimeError(
f'Browser process (PID {process.pid}) exited before CDP became available on port {port}. '
'This usually means Chrome failed to start — check that it is properly installed and '
'all required system dependencies are present (e.g. --no-sandbox may be needed in '
'Docker/headless environments, or a virtual display such as Xvfb on headless Linux).'
)
except psutil.AccessDenied:
pass # Cannot check process status; continue polling CDP
while asyncio.get_event_loop().time() - start_time < timeout:
try:
async with aiohttp.ClientSession() as session:
async with session.get(f'http://127.0.0.1:{port}/json/version') as resp:
+38
View File
@@ -43,6 +43,30 @@ def _parse_computed_styles(strings: list[str], style_indices: list[int]) -> dict
return styles
# Live values of these fields never leave the snapshot: they would otherwise be
# serialized by EnhancedDOMTreeNode.__json__ and could reach logs or the LLM.
_SENSITIVE_INPUT_TYPES = frozenset({'password', 'file', 'hidden'})
_SENSITIVE_AUTOCOMPLETE_PREFIXES = ('cc-', 'one-time-code')
def _is_sensitive_input(strings: list[str], nodes: NodeTreeSnapshot, snapshot_index: int) -> bool:
"""True for password/file/hidden inputs and payment or one-time-code autocomplete fields."""
attribute_lists = nodes.get('attributes')
if not attribute_lists or snapshot_index >= len(attribute_lists):
return False
indices = attribute_lists[snapshot_index]
for name_index, value_index in zip(indices[0::2], indices[1::2]):
if not (0 <= name_index < len(strings) and 0 <= value_index < len(strings)):
continue
name = strings[name_index].lower()
value = strings[value_index].lower()
if name == 'type' and value in _SENSITIVE_INPUT_TYPES:
return True
if name == 'autocomplete' and value.startswith(_SENSITIVE_AUTOCOMPLETE_PREFIXES):
return True
return False
def build_snapshot_lookup(
snapshot: CaptureSnapshotReturns,
device_pixel_ratio: float = 1.0,
@@ -91,6 +115,18 @@ def build_snapshot_lookup(
has_clickable_data = 'isClickable' in nodes
is_clickable_set: set[int] = set(nodes['isClickable']['index']) if has_clickable_data else set()
# Live form values live in the snapshot, not in the DOM attributes. Map
# snapshot index -> string once so each node lookup stays O(1).
input_value_by_index: dict[int, str] = {}
for key in ('inputValue', 'textValue'):
rare = nodes.get(key)
if rare:
for idx, string_index in zip(rare.get('index', []), rare.get('value', [])):
if 0 <= string_index < len(strings) and not _is_sensitive_input(strings, nodes, idx):
input_value_by_index[idx] = strings[string_index]
input_checked_set: set[int] = set(nodes['inputChecked']['index']) if 'inputChecked' in nodes else set()
has_checked_data = 'inputChecked' in nodes
# Build snapshot lookup for each backend node id
for backend_node_id, snapshot_index in backend_node_to_snapshot_index.items():
is_clickable = None
@@ -173,6 +209,8 @@ def build_snapshot_lookup(
computed_styles=computed_styles if computed_styles else None,
paint_order=paint_order,
stacking_contexts=stacking_contexts,
input_value=input_value_by_index.get(snapshot_index),
input_checked=(snapshot_index in input_checked_set) if has_checked_data else None,
)
# Count how many have bounds (are actually visible/laid out)
+35 -22
View File
@@ -16,6 +16,7 @@ from browser_use.dom.views import (
)
DISABLED_ELEMENTS = {'style', 'script', 'head', 'meta', 'link', 'title'}
URL_C0_CONTROL_OR_SPACE = ''.join(chr(codepoint) for codepoint in range(0x21))
# SVG child elements to skip (decorative only, no interaction value)
SVG_ELEMENTS = {
@@ -57,6 +58,7 @@ class DOMTreeSerializer:
DEFAULT_CONTAINMENT_THRESHOLD = 0.99 # 99% containment by default
MAX_CHILD_IMAGE_CONTEXTS = 3
MAX_CHILD_IMAGE_DESCENDANTS = 100
MAX_IMAGE_CONTEXT_ATTRIBUTE_LENGTH = 4096
def __init__(
self,
@@ -922,17 +924,47 @@ class DOMTreeSerializer:
@staticmethod
def _get_child_image_context(node: SimplifiedNode) -> str:
"""Extract compact context from image descendants of an interactive element."""
"""Extract compact context from an interactive image and its descendants."""
image_context: list[str] = []
def normalize_src(src: str) -> str:
clean_src = src.strip()
if len(src) > DOMTreeSerializer.MAX_IMAGE_CONTEXT_ATTRIBUTE_LENGTH:
return ''
clean_src = src.strip(URL_C0_CONTROL_OR_SPACE).replace('\t', '').replace('\n', '').replace('\r', '')
if clean_src.lower().startswith('data:'):
return ''
path_without_query = clean_src.split('?', 1)[0].split('#', 1)[0].rstrip('/')
return path_without_query.rsplit('/', 1)[-1]
def add_image_context(original_node: EnhancedDOMTreeNode) -> None:
if original_node.node_type != NodeType.ELEMENT_NODE or original_node.tag_name != 'img':
return
attributes = original_node.attributes or {}
parts = []
for attr_name, output_name in (
('alt', 'image_alt'),
('title', 'image_title'),
('aria-label', 'image_label'),
):
raw_attr_value = str(attributes.get(attr_name) or '')
if len(raw_attr_value) > DOMTreeSerializer.MAX_IMAGE_CONTEXT_ATTRIBUTE_LENGTH:
continue
attr_value = raw_attr_value.strip()
if attr_value:
parts.append(f'{output_name}={cap_text_length(attr_value, 100)}')
src = normalize_src(str(attributes.get('src') or ''))
if src:
parts.append(f'image_src={cap_text_length(src, 100)}')
if parts:
image_context.append(' '.join(parts))
add_image_context(node.original_node)
child_iterators = [iter(node.children)]
visited_descendants = 0
while (
@@ -946,26 +978,7 @@ class DOMTreeSerializer:
child_iterators.pop()
continue
visited_descendants += 1
original_node = current.original_node
if original_node.node_type == NodeType.ELEMENT_NODE and original_node.tag_name == 'img':
attributes = original_node.attributes or {}
parts = []
for attr_name, output_name in (
('alt', 'image_alt'),
('title', 'image_title'),
('aria-label', 'image_label'),
):
attr_value = str(attributes.get(attr_name) or '').strip()
if attr_value:
parts.append(f'{output_name}={cap_text_length(attr_value, 100)}')
src = normalize_src(str(attributes.get('src') or ''))
if src:
parts.append(f'image_src={cap_text_length(src, 100)}')
if parts:
image_context.append(' '.join(parts))
add_image_context(current.original_node)
if current.children:
child_iterators.append(iter(current.children))
+14
View File
@@ -811,6 +811,20 @@ class DomService:
# Get snapshot data and calculate absolute position
snapshot_data = snapshot_lookup.get(node['backendNodeId'], None)
# Show the value a field currently holds, not only the static attribute.
# JS, autofill, and framework bindings set the property without touching
# the attribute, so the agent used to see pre-filled fields as empty (#5647).
if snapshot_data and node['nodeName'].upper() in ('INPUT', 'TEXTAREA'):
if snapshot_data.input_value is not None:
attributes = dict(attributes or {})
attributes['value'] = snapshot_data.input_value
if snapshot_data.input_checked is not None:
attributes = dict(attributes or {})
if snapshot_data.input_checked:
attributes['checked'] = 'true'
else:
attributes.pop('checked', None)
# DIAGNOSTIC: Log when interactive elements don't have snapshot data
if not snapshot_data and node['nodeName'].upper() in ['INPUT', 'BUTTON', 'SELECT', 'TEXTAREA', 'A']:
parent_has_shadow = False
+4
View File
@@ -353,6 +353,10 @@ class EnhancedSnapshotNode:
"""Paint order from the layout tree"""
stacking_contexts: int | None
"""Stacking contexts from the layout tree"""
input_value: str | None = None
"""Live value of an <input> or <textarea> (DOMSnapshot inputValue/textValue), which the value attribute misses when JS, autofill, or a framework set it."""
input_checked: bool | None = None
"""Live checked state of a checkbox or radio input (DOMSnapshot inputChecked)."""
# @dataclass(slots=True)
+122 -13
View File
@@ -14,13 +14,8 @@ from typing import Any
from pydantic import BaseModel, Field
UNSUPPORTED_BINARY_EXTENSIONS = {
'png',
'jpg',
'jpeg',
'gif',
'bmp',
'svg',
'webp',
'ico',
'mp3',
'mp4',
@@ -394,6 +389,109 @@ class XmlFile(BaseFile):
return 'xml'
class Base64BinaryFile(BaseFile):
"""Small binary file the agent authors as base64 text.
``content`` holds the base64 string; the decoded bytes are written to disk so the
file is a real, uploadable image. ``read()`` returns a short stub instead of the
base64 so it never bloats or confuses the agent prompt (describe() calls read()
every step). Intended for tiny fixtures (e.g. a 1x1 PNG) for upload-validation
flows, not for arbitrary large binaries.
"""
# Leading magic bytes that identify a real file of this type. Keyed by extension so
# base64 that decodes but isn't actually an image (e.g. 'aGVsbG8=' -> b'hello') is
# rejected instead of written as a corrupt upload.
_MAGIC: dict[str, tuple[bytes, ...]] = {
'png': (b'\x89PNG\r\n\x1a\n',),
'gif': (b'GIF87a', b'GIF89a'),
'jpg': (b'\xff\xd8\xff',),
'jpeg': (b'\xff\xd8\xff',),
'webp': (b'RIFF',), # RIFF container; 'WEBP' tag checked below
}
def _decoded(self) -> bytes:
# Strip all whitespace (the write_file action appends a trailing newline) then
# decode strictly so non-base64 text is rejected rather than silently corrupted.
return base64.b64decode(''.join(self.content.split()), validate=True)
def _validate(self, content: str) -> None:
"""Decode and confirm the bytes actually are an image of this extension. Raises FileSystemError."""
try:
data = base64.b64decode(''.join(content.split()), validate=True)
except Exception as e:
raise FileSystemError(
f"Error: content for '{self.full_name}' is not valid base64. "
f'For images, provide the base64 of a valid {self.extension} file. ({e})'
)
magic = self._MAGIC.get(self.extension, ())
if magic and not any(data.startswith(m) for m in magic):
raise FileSystemError(
f"Error: content for '{self.full_name}' is valid base64 but not a {self.extension} image "
f'(wrong magic bytes). Provide the base64 of a real {self.extension} file.'
)
if self.extension == 'webp' and not (data[:4] == b'RIFF' and data[8:12] == b'WEBP'):
raise FileSystemError(f"Error: content for '{self.full_name}' is not a valid WEBP file.")
def write_file_content(self, content: str) -> None:
self._validate(content)
self.update_content(content)
def append_file_content(self, content: str) -> None:
raise FileSystemError(f"Error: cannot append to binary file '{self.full_name}'. Overwrite it instead.")
def sync_to_disk_sync(self, path: Path) -> None:
(path / self.full_name).write_bytes(self._decoded())
async def sync_to_disk(self, path: Path) -> None:
with ThreadPoolExecutor() as executor:
await asyncio.get_event_loop().run_in_executor(executor, lambda: self.sync_to_disk_sync(path))
def read(self) -> str:
try:
n = len(self._decoded())
except Exception:
return '[binary file: content is not valid base64]'
return f'[binary {self.extension} file, {n} bytes]'
@property
def get_size(self) -> int:
try:
return len(self._decoded())
except Exception:
return 0
class PngFile(Base64BinaryFile):
@property
def extension(self) -> str:
return 'png'
class GifFile(Base64BinaryFile):
@property
def extension(self) -> str:
return 'gif'
class JpgFile(Base64BinaryFile):
@property
def extension(self) -> str:
return 'jpg'
class JpegFile(Base64BinaryFile):
@property
def extension(self) -> str:
return 'jpeg'
class WebpFile(Base64BinaryFile):
@property
def extension(self) -> str:
return 'webp'
class FileSystemState(BaseModel):
"""Serializable state of the file system"""
@@ -427,6 +525,11 @@ class FileSystem:
'docx': DocxFile,
'html': HtmlFile,
'xml': XmlFile,
'png': PngFile,
'gif': GifFile,
'jpg': JpgFile,
'jpeg': JpegFile,
'webp': WebpFile,
}
self.files = {}
@@ -576,7 +679,7 @@ class FileSystem:
return result
# Text-based extensions: derive from _file_types, excluding those with special readers
_special_extensions = {'docx', 'pdf', 'jpg', 'jpeg', 'png'}
_special_extensions = {'docx', 'pdf', 'jpg', 'jpeg', 'png', 'gif', 'webp'}
text_extensions = [ext for ext in self._file_types if ext not in _special_extensions]
if extension in text_extensions:
@@ -711,7 +814,7 @@ class FileSystem:
)
return result
elif extension in ['jpg', 'jpeg', 'png']:
elif extension in ['jpg', 'jpeg', 'png', 'gif', 'webp']:
import anyio
# Read image file and convert to base64
@@ -786,15 +889,16 @@ class FileSystem:
if not file_class:
raise ValueError(f"Error: Invalid file extension '{extension}' for file '{full_filename}'.")
# Create or get existing file using full filename as key
if full_filename in self.files:
file_obj = self.files[full_filename]
else:
file_obj = file_class(name=name_without_ext)
self.files[full_filename] = file_obj # Use full filename as key
# Create or get existing file using full filename as key. A NEW file is only
# registered after a successful write, so a failed write (e.g. invalid base64
# for an image) leaves no ghost entry in self.files / state.
is_new = full_filename not in self.files
file_obj = self.files[full_filename] if not is_new else file_class(name=name_without_ext)
# Use file-specific write method
await file_obj.write(content, self.data_dir)
if is_new:
self.files[full_filename] = file_obj
sanitize_note = f" (auto-corrected from '{original_filename}')" if was_sanitized else ''
return f'Data written to file {full_filename} successfully.{sanitize_note}'
except FileSystemError as e:
@@ -980,6 +1084,11 @@ class FileSystem:
'DocxFile': DocxFile,
'HtmlFile': HtmlFile,
'XmlFile': XmlFile,
'PngFile': PngFile,
'GifFile': GifFile,
'JpgFile': JpgFile,
'JpegFile': JpegFile,
'WebpFile': WebpFile,
}
file_class = file_type_map.get(file_type)
+28 -7
View File
@@ -1,10 +1,11 @@
import os
from collections.abc import Mapping
from dataclasses import dataclass
from typing import Any, TypeVar, overload
import httpx
from openai import APIConnectionError, APIStatusError, AsyncOpenAI, RateLimitError
from openai.types.chat.chat_completion import ChatCompletion
from openai.types.chat.chat_completion import ChatCompletion, Choice
from openai.types.shared_params.response_format_json_schema import (
JSONSchema,
ResponseFormatJSONSchema,
@@ -57,17 +58,23 @@ class ChatOpenRouter(BaseChatModel):
def _get_client_params(self) -> dict[str, Any]:
"""Prepare client parameters dictionary."""
api_key = self.api_key or os.getenv('OPENROUTER_API_KEY')
if not api_key:
raise ModelProviderError(
message='Missing OpenRouter API key. Set OPENROUTER_API_KEY or pass api_key.',
status_code=401,
model=self.name,
)
# Define base client params
base_params = {
'api_key': self.api_key,
'api_key': api_key,
'base_url': self.base_url,
'timeout': self.timeout,
'max_retries': self.max_retries,
'default_headers': self.default_headers,
'default_query': self.default_query,
'_strict_response_validation': self._strict_response_validation,
'top_p': self.top_p,
'seed': self.seed,
}
# Create client_params dict with non-None values
@@ -91,6 +98,15 @@ class ChatOpenRouter(BaseChatModel):
self._client = AsyncOpenAI(**client_params)
return self._client
def _get_first_choice(self, response: ChatCompletion) -> Choice:
if response.choices:
return response.choices[0]
raise ModelProviderError(
message='Invalid OpenRouter response: missing or empty `choices`.',
status_code=502,
model=self.name,
)
@property
def name(self) -> str:
return str(self.model)
@@ -154,9 +170,10 @@ class ChatOpenRouter(BaseChatModel):
**(self.extra_body or {}),
)
choice = self._get_first_choice(response)
usage = self._get_usage(response)
return ChatInvokeCompletion(
completion=response.choices[0].message.content or '',
completion=choice.message.content or '',
usage=usage,
)
@@ -185,7 +202,8 @@ class ChatOpenRouter(BaseChatModel):
**(self.extra_body or {}),
)
if response.choices[0].message.content is None:
choice = self._get_first_choice(response)
if choice.message.content is None:
raise ModelProviderError(
message='Failed to parse structured output from model response',
status_code=500,
@@ -193,13 +211,16 @@ class ChatOpenRouter(BaseChatModel):
)
usage = self._get_usage(response)
parsed = output_format.model_validate_json(response.choices[0].message.content)
parsed = output_format.model_validate_json(choice.message.content)
return ChatInvokeCompletion(
completion=parsed,
usage=usage,
)
except ModelProviderError:
raise
except RateLimitError as e:
raise ModelRateLimitError(message=e.message, model=self.name) from e
+9
View File
@@ -348,6 +348,12 @@ class ChatVercel(BaseChatModel):
def _get_client_params(self) -> dict[str, Any]:
"""Prepare client parameters dictionary."""
api_key = self.api_key or os.getenv('AI_GATEWAY_API_KEY') or os.getenv('VERCEL_OIDC_TOKEN')
if not api_key:
raise ModelProviderError(
message='Missing Vercel AI Gateway API key. Set AI_GATEWAY_API_KEY or VERCEL_OIDC_TOKEN, or pass api_key.',
status_code=401,
model=self.name,
)
base_params = {
'api_key': api_key,
@@ -661,6 +667,9 @@ class ChatVercel(BaseChatModel):
stop_reason=response.choices[0].finish_reason if response.choices else None,
)
except ModelProviderError:
raise
except RateLimitError as e:
raise ModelRateLimitError(message=e.message, model=self.name) from e
+32 -25
View File
@@ -12,8 +12,7 @@ from typing import Any
import mcp.server.stdio
import mcp.types as types
from mcp.server import NotificationOptions, Server
from mcp.server.models import InitializationOptions
from mcp.server import Server
from browser_use.utils import get_browser_use_version
@@ -34,7 +33,11 @@ class CLIMCPServer:
"""Stateful stdio MCP server wrapping the browser-harness exec model."""
def __init__(self):
self.server: Server = Server('browser-use')
self.server: Server = Server(
'browser-use',
version=get_browser_use_version(),
instructions=self._instructions(),
)
self._namespace: dict[str, Any] | None = None
self._exec_lock = asyncio.Lock()
self._register_handlers()
@@ -50,7 +53,7 @@ class CLIMCPServer:
'persists across calls. Returns whatever the code prints. First navigation '
'should be new_tab(url).'
),
inputSchema={
input_schema={
'type': 'object',
'properties': {
'code': {'type': 'string', 'description': 'Python code to execute'},
@@ -61,7 +64,7 @@ class CLIMCPServer:
types.Tool(
name='browser_screenshot',
description='Capture the current page and return it as an image. Prefer this over capture_screenshot() in browser_exec.',
inputSchema={
input_schema={
'type': 'object',
'properties': {
'full': {'type': 'boolean', 'description': 'Capture beyond the viewport (full page)', 'default': False},
@@ -76,28 +79,40 @@ class CLIMCPServer:
]
def _register_handlers(self):
@self.server.list_tools()
async def handle_list_tools() -> list[types.Tool]:
return self._tool_definitions()
async def handle_list_tools(_context: Any, _params: types.PaginatedRequestParams) -> types.ListToolsResult:
return types.ListToolsResult(tools=self._tool_definitions())
@self.server.call_tool()
async def handle_call_tool(name: str, arguments: dict[str, Any] | None) -> list[types.TextContent | types.ImageContent]:
arguments = arguments or {}
async def handle_call_tool(_context: Any, params: types.CallToolRequestParams) -> types.CallToolResult:
name = params.name
arguments = params.arguments or {}
if name == 'browser_exec':
code = arguments.get('code')
if not isinstance(code, str) or not code.strip():
return [types.TextContent(type='text', text="Error: 'code' must be a non-empty string")]
return types.CallToolResult(
content=[types.TextContent(type='text', text="Error: 'code' must be a non-empty string")],
is_error=True,
)
async with self._exec_lock:
output = await asyncio.to_thread(self._execute, code)
return [types.TextContent(type='text', text=output or '(no output)')]
return types.CallToolResult(content=[types.TextContent(type='text', text=output or '(no output)')])
if name == 'browser_screenshot':
max_dim = arguments.get('max_dim')
if max_dim is not None and (isinstance(max_dim, bool) or not isinstance(max_dim, int) or max_dim < 1):
return [types.TextContent(type='text', text="Error: 'max_dim' must be a positive integer")]
return types.CallToolResult(
content=[types.TextContent(type='text', text="Error: 'max_dim' must be a positive integer")],
is_error=True,
)
async with self._exec_lock:
png = await asyncio.to_thread(self._screenshot, bool(arguments.get('full', False)), max_dim)
return [types.ImageContent(type='image', data=png, mimeType='image/png')]
return [types.TextContent(type='text', text=f'Unknown tool: {name}')]
content: list[types.ContentBlock] = [types.ImageContent(type='image', data=png, mime_type='image/png')]
return types.CallToolResult(content=content)
return types.CallToolResult(
content=[types.TextContent(type='text', text=f'Unknown tool: {name}')],
is_error=True,
)
self.server.add_request_handler('tools/list', types.PaginatedRequestParams, handle_list_tools)
self.server.add_request_handler('tools/call', types.CallToolRequestParams, handle_call_tool)
def _ensure_namespace(self) -> dict[str, Any]:
if self._namespace is None:
@@ -151,15 +166,7 @@ class CLIMCPServer:
await self.server.run(
read_stream,
write_stream,
InitializationOptions(
server_name='browser-use',
server_version=get_browser_use_version(),
instructions=self._instructions(),
capabilities=self.server.get_capabilities(
notification_options=NotificationOptions(),
experimental_capabilities={},
),
),
self.server.create_initialization_options(),
)
except BrokenPipeError:
pass
+5 -5
View File
@@ -256,10 +256,10 @@ class MCPClient:
# Parse tool parameters to create Pydantic model
param_fields = {}
if tool.inputSchema:
if tool.input_schema:
# MCP tools use JSON Schema for parameters
properties = tool.inputSchema.get('properties', {})
required = set(tool.inputSchema.get('required', []))
properties = tool.input_schema.get('properties', {})
required = set(tool.input_schema.get('required', []))
for param_name, param_schema in properties.items():
# Convert JSON Schema type to Python type
@@ -326,7 +326,7 @@ class MCPClient:
# Convert MCP result to ActionResult
extracted_content = self._format_mcp_result(result)
if getattr(result, 'isError', False):
if result.is_error:
error_msg = f"MCP tool '{tool.name}' reported an error: {extracted_content}"
return ActionResult(error=error_msg, success=False)
@@ -374,7 +374,7 @@ class MCPClient:
# Convert MCP result to ActionResult
extracted_content = self._format_mcp_result(result)
if getattr(result, 'isError', False):
if result.is_error:
error_msg = f"MCP tool '{tool.name}' reported an error: {extracted_content}"
return ActionResult(error=error_msg, success=False)
+3 -3
View File
@@ -101,10 +101,10 @@ class MCPToolWrapper:
# Parse tool parameters to create Pydantic model
param_fields = {}
if tool.inputSchema:
if tool.input_schema:
# MCP tools use JSON Schema for parameters
properties = tool.inputSchema.get('properties', {})
required = set(tool.inputSchema.get('required', []))
properties = tool.input_schema.get('properties', {})
required = set(tool.input_schema.get('required', []))
for param_name, param_schema in properties.items():
# Convert JSON Schema type to Python type
+253 -253
View File
@@ -133,8 +133,7 @@ _ensure_all_loggers_use_stderr()
try:
import mcp.server.stdio
import mcp.types as types
from mcp.server import NotificationOptions, Server
from mcp.server.models import InitializationOptions
from mcp.server import Server
MCP_AVAILABLE = True
@@ -191,7 +190,7 @@ class BrowserUseServer:
# Ensure all logging goes to stderr (in case new loggers were created)
_ensure_all_loggers_use_stderr()
self.server = Server('browser-use')
self.server = Server('browser-use', version=get_browser_use_version())
self.config = load_browser_use_config()
self.agent: Agent | None = None
self.browser_session: BrowserSession | None = None
@@ -212,271 +211,276 @@ class BrowserUseServer:
def _setup_handlers(self):
"""Setup MCP server handlers."""
@self.server.list_tools()
async def handle_list_tools() -> list[types.Tool]:
async def handle_list_tools(_context: Any, _params: types.PaginatedRequestParams) -> types.ListToolsResult:
"""List all available browser-use tools."""
return [
# Agent tools
# Direct browser control tools
types.Tool(
name='browser_navigate',
description='Navigate to a URL in the browser',
inputSchema={
'type': 'object',
'properties': {
'url': {'type': 'string', 'description': 'The URL to navigate to'},
'new_tab': {'type': 'boolean', 'description': 'Whether to open in a new tab', 'default': False},
return types.ListToolsResult(
tools=[
# Agent tools
# Direct browser control tools
types.Tool(
name='browser_navigate',
description='Navigate to a URL in the browser',
input_schema={
'type': 'object',
'properties': {
'url': {'type': 'string', 'description': 'The URL to navigate to'},
'new_tab': {'type': 'boolean', 'description': 'Whether to open in a new tab', 'default': False},
},
'required': ['url'],
},
'required': ['url'],
},
),
types.Tool(
name='browser_click',
description='Click an element by index or at specific viewport coordinates. Use index for elements from browser_get_state, or coordinate_x/coordinate_y for pixel-precise clicking.',
inputSchema={
'type': 'object',
'properties': {
'index': {
'type': 'integer',
'description': 'The index of the element to click (from browser_get_state). Provide this OR coordinate_x+coordinate_y.',
},
'coordinate_x': {
'type': 'integer',
'description': 'X coordinate in pixels from the left edge of the viewport. Must be used together with coordinate_y. Provide this OR index.',
},
'coordinate_y': {
'type': 'integer',
'description': 'Y coordinate in pixels from the top edge of the viewport. Must be used together with coordinate_x. Provide this OR index.',
},
'new_tab': {
'type': 'boolean',
'description': 'Whether to open any resulting navigation in a new tab',
'default': False,
),
types.Tool(
name='browser_click',
description='Click an element by index or at specific viewport coordinates. Use index for elements from browser_get_state, or coordinate_x/coordinate_y for pixel-precise clicking.',
input_schema={
'type': 'object',
'properties': {
'index': {
'type': 'integer',
'description': 'The index of the element to click (from browser_get_state). Provide this OR coordinate_x+coordinate_y.',
},
'coordinate_x': {
'type': 'integer',
'description': 'X coordinate in pixels from the left edge of the viewport. Must be used together with coordinate_y. Provide this OR index.',
},
'coordinate_y': {
'type': 'integer',
'description': 'Y coordinate in pixels from the top edge of the viewport. Must be used together with coordinate_x. Provide this OR index.',
},
'new_tab': {
'type': 'boolean',
'description': 'Whether to open any resulting navigation in a new tab',
'default': False,
},
},
},
},
),
types.Tool(
name='browser_type',
description='Type text into an input field. Clears existing text by default; pass text="" to clear only.',
inputSchema={
'type': 'object',
'properties': {
'index': {
'type': 'integer',
'description': 'The index of the input element (from browser_get_state)',
),
types.Tool(
name='browser_type',
description='Type text into an input field. Clears existing text by default; pass text="" to clear only.',
input_schema={
'type': 'object',
'properties': {
'index': {
'type': 'integer',
'description': 'The index of the input element (from browser_get_state)',
},
'text': {
'type': 'string',
'description': 'The text to type. Pass an empty string ("") to clear the field without typing.',
},
},
'text': {
'type': 'string',
'description': 'The text to type. Pass an empty string ("") to clear the field without typing.',
'required': ['index', 'text'],
},
),
types.Tool(
name='browser_get_state',
description='Get the current state of the page including all interactive elements',
input_schema={
'type': 'object',
'properties': {
'include_screenshot': {
'type': 'boolean',
'description': 'Whether to include a screenshot of the current page',
'default': False,
}
},
},
'required': ['index', 'text'],
},
),
types.Tool(
name='browser_get_state',
description='Get the current state of the page including all interactive elements',
inputSchema={
'type': 'object',
'properties': {
'include_screenshot': {
'type': 'boolean',
'description': 'Whether to include a screenshot of the current page',
'default': False,
}
annotations=types.ToolAnnotations(read_only_hint=True),
),
types.Tool(
name='browser_extract_content',
description='Extract structured content from the current page based on a query',
input_schema={
'type': 'object',
'properties': {
'query': {'type': 'string', 'description': 'What information to extract from the page'},
'extract_links': {
'type': 'boolean',
'description': 'Whether to include links in the extraction',
'default': False,
},
},
'required': ['query'],
},
},
annotations=types.ToolAnnotations(readOnlyHint=True),
),
types.Tool(
name='browser_extract_content',
description='Extract structured content from the current page based on a query',
inputSchema={
'type': 'object',
'properties': {
'query': {'type': 'string', 'description': 'What information to extract from the page'},
'extract_links': {
'type': 'boolean',
'description': 'Whether to include links in the extraction',
'default': False,
),
types.Tool(
name='browser_get_html',
description='Get the raw HTML of the current page or a specific element by CSS selector',
input_schema={
'type': 'object',
'properties': {
'selector': {
'type': 'string',
'description': 'Optional CSS selector to get HTML of a specific element. If omitted, returns full page HTML.',
},
},
},
'required': ['query'],
},
),
types.Tool(
name='browser_get_html',
description='Get the raw HTML of the current page or a specific element by CSS selector',
inputSchema={
'type': 'object',
'properties': {
'selector': {
'type': 'string',
'description': 'Optional CSS selector to get HTML of a specific element. If omitted, returns full page HTML.',
annotations=types.ToolAnnotations(read_only_hint=True),
),
types.Tool(
name='browser_screenshot',
description='Take a screenshot of the current page. Returns viewport metadata as text and the screenshot as an image.',
input_schema={
'type': 'object',
'properties': {
'full_page': {
'type': 'boolean',
'description': 'Whether to capture the full scrollable page or just the visible viewport',
'default': False,
},
},
},
},
annotations=types.ToolAnnotations(readOnlyHint=True),
),
types.Tool(
name='browser_screenshot',
description='Take a screenshot of the current page. Returns viewport metadata as text and the screenshot as an image.',
inputSchema={
'type': 'object',
'properties': {
'full_page': {
'type': 'boolean',
'description': 'Whether to capture the full scrollable page or just the visible viewport',
'default': False,
annotations=types.ToolAnnotations(read_only_hint=True),
),
types.Tool(
name='browser_scroll',
description='Scroll the page',
input_schema={
'type': 'object',
'properties': {
'direction': {
'type': 'string',
'enum': ['up', 'down'],
'description': 'Direction to scroll',
'default': 'down',
}
},
},
},
annotations=types.ToolAnnotations(readOnlyHint=True),
),
types.Tool(
name='browser_scroll',
description='Scroll the page',
inputSchema={
'type': 'object',
'properties': {
'direction': {
'type': 'string',
'enum': ['up', 'down'],
'description': 'Direction to scroll',
'default': 'down',
}
),
types.Tool(
name='browser_go_back',
description='Go back to the previous page',
input_schema={'type': 'object', 'properties': {}},
),
# Tab management
types.Tool(
name='browser_list_tabs',
description='List all open tabs',
input_schema={'type': 'object', 'properties': {}},
annotations=types.ToolAnnotations(read_only_hint=True),
),
types.Tool(
name='browser_switch_tab',
description='Switch to a different tab',
input_schema={
'type': 'object',
'properties': {
'tab_id': {'type': 'string', 'description': '4 Character Tab ID of the tab to switch to'}
},
'required': ['tab_id'],
},
},
),
types.Tool(
name='browser_go_back',
description='Go back to the previous page',
inputSchema={'type': 'object', 'properties': {}},
),
# Tab management
types.Tool(
name='browser_list_tabs',
description='List all open tabs',
inputSchema={'type': 'object', 'properties': {}},
annotations=types.ToolAnnotations(readOnlyHint=True),
),
types.Tool(
name='browser_switch_tab',
description='Switch to a different tab',
inputSchema={
'type': 'object',
'properties': {'tab_id': {'type': 'string', 'description': '4 Character Tab ID of the tab to switch to'}},
'required': ['tab_id'],
},
),
types.Tool(
name='browser_close_tab',
description='Close a tab',
inputSchema={
'type': 'object',
'properties': {'tab_id': {'type': 'string', 'description': '4 Character Tab ID of the tab to close'}},
'required': ['tab_id'],
},
),
# types.Tool(
# name="browser_close",
# description="Close the browser session",
# inputSchema={
# "type": "object",
# "properties": {}
# }
# ),
types.Tool(
name='retry_with_browser_use_agent',
description='Retry a task using the browser-use agent. Only use this as a last resort if you fail to interact with a page multiple times.',
inputSchema={
'type': 'object',
'properties': {
'task': {
'type': 'string',
'description': 'The high-level goal and detailed step-by-step description of the task the AI browser agent needs to attempt, along with any relevant data needed to complete the task and info about previous attempts.',
},
'max_steps': {
'type': 'integer',
'description': 'Maximum number of steps an agent can take.',
'default': 100,
},
'model': {
'type': 'string',
'description': 'LLM model to use (e.g., gpt-4o, claude-3-opus-20240229). Defaults to the configured model.',
},
'allowed_domains': {
'type': 'array',
'items': {'type': 'string'},
'description': (
'List of domains the agent is allowed to visit (security feature). '
'Omit to use the server-configured profile defaults. '
'An empty list is treated the same as omitting the argument and '
'will NOT disable server-configured restrictions.'
),
},
'use_vision': {
'type': 'boolean',
'description': 'Whether to use vision capabilities (screenshots) for the agent',
'default': True,
},
),
types.Tool(
name='browser_close_tab',
description='Close a tab',
input_schema={
'type': 'object',
'properties': {'tab_id': {'type': 'string', 'description': '4 Character Tab ID of the tab to close'}},
'required': ['tab_id'],
},
'required': ['task'],
},
),
# Browser session management tools
types.Tool(
name='browser_list_sessions',
description='List all active browser sessions with their details and last activity time',
inputSchema={'type': 'object', 'properties': {}},
annotations=types.ToolAnnotations(readOnlyHint=True),
),
types.Tool(
name='browser_close_session',
description='Close a specific browser session by its ID',
inputSchema={
'type': 'object',
'properties': {
'session_id': {
'type': 'string',
'description': 'The browser session ID to close (get from browser_list_sessions)',
}
),
# types.Tool(
# name="browser_close",
# description="Close the browser session",
# input_schema={
# "type": "object",
# "properties": {}
# }
# ),
types.Tool(
name='retry_with_browser_use_agent',
description='Retry a task using the browser-use agent. Only use this as a last resort if you fail to interact with a page multiple times.',
input_schema={
'type': 'object',
'properties': {
'task': {
'type': 'string',
'description': 'The high-level goal and detailed step-by-step description of the task the AI browser agent needs to attempt, along with any relevant data needed to complete the task and info about previous attempts.',
},
'max_steps': {
'type': 'integer',
'description': 'Maximum number of steps an agent can take.',
'default': 100,
},
'model': {
'type': 'string',
'description': 'LLM model to use (e.g., gpt-4o, claude-3-opus-20240229). Defaults to the configured model.',
},
'allowed_domains': {
'type': 'array',
'items': {'type': 'string'},
'description': (
'List of domains the agent is allowed to visit (security feature). '
'Omit to use the server-configured profile defaults. '
'An empty list is treated the same as omitting the argument and '
'will NOT disable server-configured restrictions.'
),
},
'use_vision': {
'type': 'boolean',
'description': 'Whether to use vision capabilities (screenshots) for the agent',
'default': True,
},
},
'required': ['task'],
},
'required': ['session_id'],
},
),
types.Tool(
name='browser_close_all',
description='Close all active browser sessions and clean up resources',
inputSchema={'type': 'object', 'properties': {}},
),
]
),
# Browser session management tools
types.Tool(
name='browser_list_sessions',
description='List all active browser sessions with their details and last activity time',
input_schema={'type': 'object', 'properties': {}},
annotations=types.ToolAnnotations(read_only_hint=True),
),
types.Tool(
name='browser_close_session',
description='Close a specific browser session by its ID',
input_schema={
'type': 'object',
'properties': {
'session_id': {
'type': 'string',
'description': 'The browser session ID to close (get from browser_list_sessions)',
}
},
'required': ['session_id'],
},
),
types.Tool(
name='browser_close_all',
description='Close all active browser sessions and clean up resources',
input_schema={'type': 'object', 'properties': {}},
),
]
)
@self.server.list_resources()
async def handle_list_resources() -> list[types.Resource]:
async def handle_list_resources(_context: Any, _params: types.PaginatedRequestParams) -> types.ListResourcesResult:
"""List available resources (none for browser-use)."""
return []
return types.ListResourcesResult(resources=[])
@self.server.list_prompts()
async def handle_list_prompts() -> list[types.Prompt]:
async def handle_list_prompts(_context: Any, _params: types.PaginatedRequestParams) -> types.ListPromptsResult:
"""List available prompts (none for browser-use)."""
return []
return types.ListPromptsResult(prompts=[])
@self.server.call_tool()
async def handle_call_tool(name: str, arguments: dict[str, Any] | None) -> list[types.TextContent | types.ImageContent]:
async def handle_call_tool(_context: Any, params: types.CallToolRequestParams) -> types.CallToolResult:
"""Handle tool execution."""
name = params.name
arguments = params.arguments
start_time = time.time()
error_msg = None
try:
result = await self._execute_tool(name, arguments or {})
if isinstance(result, list):
return result
return [types.TextContent(type='text', text=result)]
return types.CallToolResult(content=result)
return types.CallToolResult(content=[types.TextContent(type='text', text=result)])
except Exception as e:
error_msg = str(e)
logger.error(f'Tool execution failed: {e}', exc_info=True)
return [types.TextContent(type='text', text=f'Error: {str(e)}')]
return types.CallToolResult(
content=[types.TextContent(type='text', text=f'Error: {str(e)}')],
is_error=True,
)
finally:
# Capture telemetry for tool calls
duration = time.time() - start_time
@@ -490,9 +494,12 @@ class BrowserUseServer:
)
)
async def _execute_tool(
self, tool_name: str, arguments: dict[str, Any]
) -> str | list[types.TextContent | types.ImageContent]:
self.server.add_request_handler('tools/list', types.PaginatedRequestParams, handle_list_tools)
self.server.add_request_handler('resources/list', types.PaginatedRequestParams, handle_list_resources)
self.server.add_request_handler('prompts/list', types.PaginatedRequestParams, handle_list_prompts)
self.server.add_request_handler('tools/call', types.CallToolRequestParams, handle_call_tool)
async def _execute_tool(self, tool_name: str, arguments: dict[str, Any]) -> str | list[types.ContentBlock]:
"""Execute a browser-use tool. Returns str for most tools, or a content list for tools with image output."""
# Agent-based tools
@@ -537,9 +544,9 @@ class BrowserUseServer:
elif tool_name == 'browser_get_state':
state_json, screenshot_b64 = await self._get_browser_state(arguments.get('include_screenshot', False))
content: list[types.TextContent | types.ImageContent] = [types.TextContent(type='text', text=state_json)]
content: list[types.ContentBlock] = [types.TextContent(type='text', text=state_json)]
if screenshot_b64:
content.append(types.ImageContent(type='image', data=screenshot_b64, mimeType='image/png'))
content.append(types.ImageContent(type='image', data=screenshot_b64, mime_type='image/png'))
return content
elif tool_name == 'browser_get_html':
@@ -547,9 +554,9 @@ class BrowserUseServer:
elif tool_name == 'browser_screenshot':
meta_json, screenshot_b64 = await self._screenshot(arguments.get('full_page', False))
content: list[types.TextContent | types.ImageContent] = [types.TextContent(type='text', text=meta_json)]
content: list[types.ContentBlock] = [types.TextContent(type='text', text=meta_json)]
if screenshot_b64:
content.append(types.ImageContent(type='image', data=screenshot_b64, mimeType='image/png'))
content.append(types.ImageContent(type='image', data=screenshot_b64, mime_type='image/png'))
return content
elif tool_name == 'browser_extract_content':
@@ -573,7 +580,7 @@ class BrowserUseServer:
elif tool_name == 'browser_close_tab':
return await self._close_tab(arguments['tab_id'])
return f'Unknown tool: {tool_name}'
raise ValueError(f'Unknown tool: {tool_name}')
async def _init_browser_session(self, allowed_domains: list[str] | None = None, **kwargs):
"""Initialize browser session using config"""
@@ -1248,14 +1255,7 @@ class BrowserUseServer:
await self.server.run(
read_stream,
write_stream,
InitializationOptions(
server_name='browser-use',
server_version='0.1.0',
capabilities=self.server.get_capabilities(
notification_options=NotificationOptions(),
experimental_capabilities={},
),
),
self.server.create_initialization_options(),
)
except BrokenPipeError:
logger.warning('MCP client disconnected while writing to stdio; shutting down server cleanly.')
+16 -5
View File
@@ -54,6 +54,8 @@ PY
changing Chrome's visible tab. Screenshots and normal CDP input work in the
background; call `activate_tab(target)` only when the user explicitly asks
or a page demonstrably pauses rendering while hidden.
- Set `BH_TAB_MARKER=0` before starting the daemon to leave page titles unchanged.
The horse marker remains enabled by default.
- A timed-out `scroll(...)` on an attached background tab is evidence that the
page needs to be visible. Call `activate_tab(current_tab())`, retry the same
scroll once, then re-read the scroll position. This visibly switches tabs,
@@ -77,14 +79,22 @@ If Chrome is running but remote debugging is not enabled, the harness opens:
chrome://inspect/#remote-debugging
```
On macOS, when Chrome asks for remote-debugging permission, run:
On macOS, when local Chrome asks for remote-debugging permission, keep the
original browser command running and call `mac-approve` in another shell/tool
call. Preserve the exact daemon name: if the waiting command used
`BU_NAME=r7k2`, run:
```text
browser-use mac-approve
BU_NAME=r7k2 browser-use mac-approve
```
Continue browser work when it returns `ready`; otherwise follow its printed
instruction.
For the default daemon, omit the `BU_NAME` prefix. The original command resumes
when the helper returns `ready`; do not rerun it. If the helper reports
`accessibility-required`, ask the user once to grant the app launching
browser-use (for example Terminal, iTerm, or Codex) access in System
Settings > Privacy & Security > Accessibility, then call `mac-approve` once
again. This is only for local Chrome; do not call it for `BU_CDP_URL`,
`BU_CDP_WS`, or Browser Use Cloud.
## Remote Browsers
@@ -136,6 +146,7 @@ Cloud profile cookie sync reference: https://github.com/browser-use/browser-harn
- After navigation, call `wait_for_load()`.
- If the current tab is stale or internal, call `ensure_real_tab()`.
- Use `js(...)` for DOM inspection or extraction when coordinates are the wrong tool.
- When entering unusually long text, avoid slow per-character typing: find a faster page-appropriate input method, then verify the page kept the exact value.
- Login walls: stop and ask. Exception: use available SSO automatically when Chrome is already signed in; still stop for passwords, MFA, consent, or ambiguous account choice.
- Raw CDP is available with `cdp("Domain.method", ...)`.
@@ -207,7 +218,7 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro
## Gotchas
- `chrome://inspect/#remote-debugging` must be enabled for local Chrome control.
- On macOS, if Chrome shows an "Allow remote debugging?" popup, run `browser-use mac-approve`. Do not poll in a loop — the daemon holds one connection.
- On macOS, if local Chrome shows an "Allow remote debugging?" popup, call `mac-approve` once with the same `BU_NAME` while the original browser command waits. Do not poll or rerun the browser command; remote and cloud browsers do not use this helper.
- Omnibox popups are not real work tabs.
- CDP target order is not Chrome's visible tab-strip order.
- `BU_CDP_URL` is an HTTP DevTools endpoint; the daemon resolves it to WebSocket.
+23 -13
View File
@@ -1013,19 +1013,27 @@ class Tools(Generic[Context]):
event = browser_session.event_bus.dispatch(SwitchTabEvent(target_id=target_id))
await event
new_target_id = await event.event_result(raise_if_any=False, raise_if_none=False) # Don't raise on errors
if new_target_id:
memory = f'Switched to tab #{new_target_id[-4:]}'
else:
memory = f'Switched to tab #{params.tab_id}'
logger.info(f'🔄 {memory}')
return ActionResult(extracted_content=memory, long_term_memory=memory)
# raise_if_any=True so a handler failure surfaces its real cause here instead
# of silently becoming a "produced no result" below.
new_target_id = await event.event_result(raise_if_any=True, raise_if_none=False)
except Exception as e:
logger.warning(f'Tab switch may have failed: {e}')
memory = f'Attempted to switch to tab #{params.tab_id}'
return ActionResult(extracted_content=memory, long_term_memory=memory)
logger.warning(f'Tab switch failed: {e}')
# Preserve the concrete cause (e.g. a stale tab_id) instead of a generic
# message, so the agent gets actionable failure info in both memories.
memory = f'Failed to switch to tab #{params.tab_id}: {e}'
raise BrowserError(memory, short_term_memory=memory, long_term_memory=memory)
# on_SwitchTabEvent returns the newly focused TargetID on every success path, so a
# missing result (with no exception raised) means the handler still failed rather
# than having quietly succeeded.
if not new_target_id:
memory = f'Failed to switch to tab #{params.tab_id}: tab switch produced no result'
logger.warning(memory)
raise BrowserError(memory, short_term_memory=memory, long_term_memory=memory)
memory = f'Switched to tab #{new_target_id[-4:]}'
logger.info(f'🔄 {memory}')
return ActionResult(extracted_content=memory, long_term_memory=memory)
@self.registry.action(
'Close a tab by tab_id. Tab IDs are shown in browser state tabs list (last 4 chars of target_id). Use to clean up tabs you no longer need.',
@@ -1747,7 +1755,9 @@ You will be given a query and the markdown of a webpage that has been filtered t
'Write content to a file. By default this OVERWRITES the entire file - use append=true to add to an existing file, or use replace_file for targeted edits within a file. '
'FILENAME RULES: Use only letters, numbers, underscores, hyphens, dots, parentheses. Spaces are auto-converted to hyphens. '
'SUPPORTED EXTENSIONS: .txt, .md, .json, .jsonl, .csv, .html, .xml, .pdf, .docx. '
'CANNOT write binary/image files (.png, .jpg, .mp4, etc.) - do not attempt to save screenshots as files. '
'For small images (.png, .gif, .jpg, .jpeg, .webp) — e.g. a tiny file to upload — set content to the '
'base64 of a valid image (a 1x1 PNG is ~92 base64 chars); the bytes are decoded and written for you. '
'CANNOT write other binary files (.mp4, .zip, etc.) and do not attempt to save screenshots as files. '
'For PDF files, write content in markdown format and it will be auto-converted to PDF.'
)
async def write_file(
+9 -8
View File
@@ -2,7 +2,7 @@
name = "browser-use"
description = "Make websites accessible for AI agents"
authors = [{ name = "Gregor Zunic" }]
version = "0.13.8"
version = "0.13.10"
readme = "README.md"
requires-python = ">=3.11,<4.0"
classifiers = [
@@ -21,7 +21,8 @@ dependencies = [
"httpx==0.28.1",
"posthog==7.7.0",
"psutil==7.2.2",
"pydantic>=2.12.5,<2.14",
"pydantic==2.13.5",
"pydantic-settings==2.15.0",
"pyobjc==12.1; platform_system == 'darwin'",
"python-dotenv==1.2.2",
"requests==2.33.0",
@@ -36,8 +37,8 @@ dependencies = [
"google-api-python-client==2.188.0",
"google-auth==2.48.0",
"google-auth-oauthlib==1.2.4",
"mcp==1.28.1",
"pypdf==6.15.0",
"mcp==2.1.1",
"pypdf==6.16.2",
"reportlab==4.4.9",
"cdp-use==1.4.5",
"pyotp==2.9.0",
@@ -46,7 +47,7 @@ dependencies = [
"markdownify==1.2.2",
"python-docx==1.2.0",
"browser-use-sdk==3.4.2",
"browser-harness==0.1.10",
"browser-harness==0.1.13",
]
# google-api-core: only used for Google LLM APIs
# pyperclip: only used for examples that use copy/paste
@@ -108,12 +109,12 @@ browser = "browser_use.cli:main" # Alias for browser-use
browser-use-tui = "browser_use.cli:browser_use_tui_main" # Deprecated alias for browser-use
[build-system]
requires = ["hatchling"]
requires = ["hatchling==1.32.0"]
build-backend = "hatchling.build"
[tool.codespell]
ignore-words-list = "bu,wit,dont,cant,wont,re-use,re-used,re-using,re-usable,thats,doesnt,doubleclick,finaly,finalY"
ignore-words-list = "bu,wit,dont,cant,wont,re-use,re-used,re-using,re-usable,thats,doesnt,doubleclick,finaly,finalY,iterm"
skip = "*.json"
[tool.ruff]
@@ -242,5 +243,5 @@ dev-dependencies = [
"lmnr[all]==0.7.42",
# "pytest-playwright-asyncio>=0.7.0", # not actually needed I think
"pytest-timeout==2.4.0",
"pydantic_settings==2.12.0",
"pydantic_settings==2.15.0",
]
+16 -5
View File
@@ -54,6 +54,8 @@ PY
changing Chrome's visible tab. Screenshots and normal CDP input work in the
background; call `activate_tab(target)` only when the user explicitly asks
or a page demonstrably pauses rendering while hidden.
- Set `BH_TAB_MARKER=0` before starting the daemon to leave page titles unchanged.
The horse marker remains enabled by default.
- A timed-out `scroll(...)` on an attached background tab is evidence that the
page needs to be visible. Call `activate_tab(current_tab())`, retry the same
scroll once, then re-read the scroll position. This visibly switches tabs,
@@ -77,14 +79,22 @@ If Chrome is running but remote debugging is not enabled, the harness opens:
chrome://inspect/#remote-debugging
```
On macOS, when Chrome asks for remote-debugging permission, run:
On macOS, when local Chrome asks for remote-debugging permission, keep the
original browser command running and call `mac-approve` in another shell/tool
call. Preserve the exact daemon name: if the waiting command used
`BU_NAME=r7k2`, run:
```text
browser-use mac-approve
BU_NAME=r7k2 browser-use mac-approve
```
Continue browser work when it returns `ready`; otherwise follow its printed
instruction.
For the default daemon, omit the `BU_NAME` prefix. The original command resumes
when the helper returns `ready`; do not rerun it. If the helper reports
`accessibility-required`, ask the user once to grant the app launching
browser-use (for example Terminal, iTerm, or Codex) access in System
Settings > Privacy & Security > Accessibility, then call `mac-approve` once
again. This is only for local Chrome; do not call it for `BU_CDP_URL`,
`BU_CDP_WS`, or Browser Use Cloud.
## Remote Browsers
@@ -136,6 +146,7 @@ Cloud profile cookie sync reference: https://github.com/browser-use/browser-harn
- After navigation, call `wait_for_load()`.
- If the current tab is stale or internal, call `ensure_real_tab()`.
- Use `js(...)` for DOM inspection or extraction when coordinates are the wrong tool.
- When entering unusually long text, avoid slow per-character typing: find a faster page-appropriate input method, then verify the page kept the exact value.
- Login walls: stop and ask. Exception: use available SSO automatically when Chrome is already signed in; still stop for passwords, MFA, consent, or ambiguous account choice.
- Raw CDP is available with `cdp("Domain.method", ...)`.
@@ -207,7 +218,7 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro
## Gotchas
- `chrome://inspect/#remote-debugging` must be enabled for local Chrome control.
- On macOS, if Chrome shows an "Allow remote debugging?" popup, run `browser-use mac-approve`. Do not poll in a loop — the daemon holds one connection.
- On macOS, if local Chrome shows an "Allow remote debugging?" popup, call `mac-approve` once with the same `BU_NAME` while the original browser command waits. Do not poll or rerun the browser command; remote and cloud browsers do not use this helper.
- Omnibox popups are not real work tabs.
- CDP target order is not Chrome's visible tab-strip order.
- `BU_CDP_URL` is an HTTP DevTools endpoint; the daemon resolves it to WebSocket.
@@ -0,0 +1,89 @@
import asyncio
from types import SimpleNamespace
from typing import cast
from browser_use.browser.events import SendKeysEvent
from browser_use.browser.watchdogs.default_action_watchdog import DefaultActionWatchdog
def make_watchdog(recorded_params, dispatched_keys) -> DefaultActionWatchdog:
class Input:
async def dispatchKeyEvent(self, params=None, session_id=None):
recorded_params.append(params or {})
cdp_session = SimpleNamespace(
cdp_client=SimpleNamespace(send=SimpleNamespace(Input=Input())),
session_id='session-1',
)
class BrowserSession:
async def get_or_create_cdp_session(self, focus=False):
return cdp_session
async def dispatch_key_event(_session, event_type, key, modifiers=0):
dispatched_keys.append((event_type, key, modifiers))
watchdog = SimpleNamespace(
browser_session=BrowserSession(),
logger=SimpleNamespace(info=lambda *args, **kwargs: None),
_dispatch_key_event=dispatch_key_event,
)
watchdog._get_char_modifiers_and_vk = DefaultActionWatchdog._get_char_modifiers_and_vk.__get__(watchdog)
watchdog._get_key_code_for_char = DefaultActionWatchdog._get_key_code_for_char.__get__(watchdog)
return cast(DefaultActionWatchdog, watchdog)
def test_send_keys_literal_plus_dispatches_char_event():
recorded_params = []
dispatched_keys = []
watchdog = make_watchdog(recorded_params, dispatched_keys)
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='+')))
char_events = [params for params in recorded_params if params.get('type') == 'char']
assert [(params.get('text'), params.get('key')) for params in char_events] == [('+', '+')]
assert all(params.get('key') for params in recorded_params)
def test_send_keys_text_with_plus_dispatches_all_characters():
recorded_params = []
dispatched_keys = []
watchdog = make_watchdog(recorded_params, dispatched_keys)
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='C++')))
assert [params['text'] for params in recorded_params if params.get('type') == 'char'] == ['C', '+', '+']
def test_send_keys_control_plus_keeps_plus_as_main_key():
recorded_params = []
dispatched_keys = []
watchdog = make_watchdog(recorded_params, dispatched_keys)
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='Control++')))
assert dispatched_keys == [
('keyDown', 'Control', 0),
('keyDown', '+', 2),
('keyUp', '+', 2),
('keyUp', 'Control', 0),
]
def test_send_keys_existing_shortcut_and_special_key_still_work():
recorded_params = []
dispatched_keys = []
watchdog = make_watchdog(recorded_params, dispatched_keys)
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='Control+a')))
assert dispatched_keys == [
('keyDown', 'Control', 0),
('keyDown', 'a', 2),
('keyUp', 'a', 2),
('keyUp', 'Control', 0),
]
recorded_params.clear()
dispatched_keys.clear()
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='Enter')))
assert dispatched_keys == [('keyDown', 'Enter', 0), ('keyUp', 'Enter', 0)]
+20
View File
@@ -246,6 +246,26 @@ class TestBrowserSessionEventSystem:
assert browser_session.event_bus.name.startswith('EventBus_')
# Event bus name format may vary, just check it exists
@pytest.mark.parametrize('reset_method', ['stop', 'kill'])
async def test_session_handlers_registered_after_event_bus_reset(self, browser_session: BrowserSession, reset_method: str):
"""Session handlers must be restored when stop() or kill() replaces the event bus."""
initial_bus = browser_session.event_bus
initial_handlers = {
event_name: [getattr(handler, '__name__', str(handler)) for handler in handlers]
for event_name, handlers in initial_bus.handlers.items()
}
assert initial_handlers
assert 'BrowserStartEvent' in initial_handlers
assert 'BrowserStopEvent' in initial_handlers
await getattr(browser_session, reset_method)()
assert browser_session.event_bus is not initial_bus
assert {
event_name: [getattr(handler, '__name__', str(handler)) for handler in handlers]
for event_name, handlers in browser_session.event_bus.handlers.items()
} == initial_handlers
async def test_event_handlers_registration(self, browser_session: BrowserSession):
"""Test that event handlers are properly registered."""
# Attach all watchdogs to register their handlers
+94
View File
@@ -19,13 +19,17 @@ Usage:
import asyncio
import time
from typing import Any
import pytest
from pytest_httpserver import HTTPServer
from browser_use.agent.service import Agent
from browser_use.agent.views import ActionModel
from browser_use.browser import BrowserSession
from browser_use.browser.events import SwitchTabEvent
from browser_use.browser.profile import BrowserProfile
from browser_use.tools.service import Tools
from tests.ci.conftest import create_mock_llm
@@ -669,3 +673,93 @@ class TestMultiTabOperations:
assert 'Successfully' in final_result, 'Agent should report success'
except TimeoutError:
pytest.fail('Test timed out after 2 minutes - agent hung during multiple tab operations')
class _TabActionModel(ActionModel):
"""ActionModel with explicit slots for the tab actions driven directly via tools.act().
registry.create_action_model() builds its fields at runtime, so a statically declared
subclass is what keeps pyright able to check these call sites. act() dispatches on the
key returned by model_dump(exclude_unset=True), so the real registered actions still run.
"""
switch: dict[str, Any] | None = None
navigate: dict[str, Any] | None = None
class TestSwitchTabFailureReporting:
"""A failed `switch` must surface as ActionResult.error rather than a fake success.
Previously both failure paths (a stale/unknown tab_id, and a missing SwitchTabEvent
result) returned a non-error ActionResult claiming the switch succeeded. The false
claim was written into long_term_memory, so subsequent steps reasoned from a tab the
agent never actually reached.
"""
async def test_switch_to_nonexistent_tab_reports_error(self, browser_session):
tools = Tools()
tabs_before = await browser_session.get_tabs()
live_ids = {tab.target_id[-4:] for tab in tabs_before}
bogus_tab_id = 'zzzz'
assert bogus_tab_id not in live_ids
result = await tools.act(_TabActionModel(switch={'tab_id': bogus_tab_id}), browser_session=browser_session)
assert result.error is not None, 'a failed tab switch must set ActionResult.error'
assert bogus_tab_id in result.error
assert 'Switched to tab' not in (result.extracted_content or '')
assert 'Switched to tab' not in (result.long_term_memory or '')
tabs_after = await browser_session.get_tabs()
assert {tab.target_id[-4:] for tab in tabs_after} == live_ids, 'no tab switch should have happened'
async def test_switch_to_open_tab_still_succeeds(self, browser_session, base_url):
tools = Tools()
original_tab_id = (await browser_session.get_tabs())[0].target_id[-4:]
open_result = await tools.act(
_TabActionModel(navigate={'url': f'{base_url}/page1', 'new_tab': True}),
browser_session=browser_session,
)
assert open_result.error is None, f'opening a new tab should not error: {open_result.error}'
result = await tools.act(_TabActionModel(switch={'tab_id': original_tab_id}), browser_session=browser_session)
assert result.error is None, f'switching to a live tab must not error: {result.error}'
assert result.long_term_memory is not None
assert 'Switched to tab' in result.long_term_memory
async def test_switch_reports_error_when_event_yields_no_result(self, browser_session, monkeypatch):
"""A handler that completes without raising but yields no TargetID is still a failure.
on_SwitchTabEvent returns a TargetID on every success path, so a missing result is
never a quiet success - this must be reported as an error even though nothing raised.
"""
tools = Tools()
tab_id = (await browser_session.get_tabs())[0].target_id[-4:]
class NoResultEvent:
async def _wait(self):
return self
def __await__(self):
return self._wait().__await__()
async def event_result(self, **_kwargs):
return None
original_dispatch = browser_session.event_bus.dispatch
monkeypatch.setattr(
browser_session.event_bus,
'dispatch',
lambda event: NoResultEvent() if isinstance(event, SwitchTabEvent) else original_dispatch(event),
)
result = await tools.act(_TabActionModel(switch={'tab_id': tab_id}), browser_session=browser_session)
assert result.error is not None, 'a switch that yields no result must set ActionResult.error'
assert 'produced no result' in result.error
assert 'Switched to tab' not in (result.extracted_content or '')
assert 'Switched to tab' not in (result.long_term_memory or '')
+28 -8
View File
@@ -465,8 +465,16 @@ class TestFileSystem:
assert fs._is_valid_filename('.json') is False # no name
assert fs._is_valid_filename('.jsonl') is False # no name
assert fs._is_valid_filename('.csv') is False # no name
assert fs._is_valid_filename('screenshot.png') is False # binary extension
assert fs._is_valid_filename('image.jpg') is False # binary extension
# Small image extensions are now supported (base64 content -> real bytes, for upload flows)
assert fs._is_valid_filename('screenshot.png') is True
assert fs._is_valid_filename('image.jpg') is True
assert fs._is_valid_filename('pic.gif') is True
assert fs._is_valid_filename('photo.webp') is True
# Other binary types remain unsupported
assert fs._is_valid_filename('clip.mp4') is False # binary extension
assert fs._is_valid_filename('archive.zip') is False # binary extension
assert fs._is_valid_filename('icon.svg') is False # binary extension
def test_filename_parsing(self, temp_filesystem):
"""Test filename parsing into name and extension."""
@@ -1185,17 +1193,29 @@ class TestFilenameSanitization:
fs.nuke()
async def test_write_file_binary_extension_error(self):
"""Test that writing to binary extensions gives a clear error."""
"""Unsupported binary extensions give a clear error; small images accept base64."""
with tempfile.TemporaryDirectory() as tmp_dir:
fs = FileSystem(base_dir=tmp_dir, create_default_files=False)
result = await fs.write_file('screenshot.png', 'content')
# Non-image binaries are still rejected outright
result = await fs.write_file('clip.mp4', 'content')
assert 'binary/image' in result.lower() or 'Cannot write' in result
assert 'screenshot.png' not in fs.list_files()
assert 'clip.mp4' not in fs.list_files()
result = await fs.write_file('photo.jpg', 'content')
result = await fs.write_file('archive.zip', 'content')
assert 'binary/image' in result.lower() or 'Cannot write' in result
# Small images are supported: non-base64 content is rejected (no corrupt file),
# valid base64 is written as real bytes.
result = await fs.write_file('screenshot.png', 'not base64!!!')
assert 'Error' in result
assert not (fs.get_dir() / 'screenshot.png').exists()
png_1x1 = 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAAC0lEQVR4nGNgAAIAAAUAAXpeqz8AAAAASUVORK5CYII='
result = await fs.write_file('logo.png', png_1x1)
assert 'successfully' in result
assert (fs.get_dir() / 'logo.png').read_bytes()[:8] == b'\x89PNG\r\n\x1a\n'
fs.nuke()
async def test_write_file_unsupported_extension_error(self):
@@ -1315,8 +1335,8 @@ class TestFilenameSanitization:
result = await fs.read_file('noextension')
assert 'no extension' in result.lower()
# Binary extension - specific error
result = await fs.write_file('image.png', 'data')
# Unsupported binary extension - specific error
result = await fs.write_file('clip.mp4', 'data')
assert 'binary' in result.lower() or 'Cannot write' in result
fs.nuke()
+78
View File
@@ -0,0 +1,78 @@
"""Regression tests for OpenRouter client setup and response handling."""
from unittest.mock import AsyncMock, patch
import pytest
from openai.types.chat import ChatCompletion, ChatCompletionMessage
from openai.types.chat.chat_completion import Choice
from pydantic import BaseModel
from browser_use.llm.exceptions import ModelProviderError
from browser_use.llm.messages import UserMessage
from browser_use.llm.openrouter.chat import ChatOpenRouter
class Answer(BaseModel):
answer: str
def _completion(*, content: str | None = 'ok', choices: bool = True) -> ChatCompletion:
return ChatCompletion(
id='chatcmpl-test',
choices=[Choice(finish_reason='stop', index=0, message=ChatCompletionMessage(role='assistant', content=content))]
if choices
else [],
created=0,
model='openai/gpt-4o',
object='chat.completion',
)
async def test_request_params_reach_completion_not_client():
llm = ChatOpenRouter(model='openai/gpt-4o', api_key='test-key', top_p=0.9, seed=42)
client = llm.get_client()
assert client.api_key == 'test-key'
assert 'top_p' not in llm._get_client_params()
assert 'seed' not in llm._get_client_params()
create = AsyncMock(return_value=_completion())
with patch.object(type(client.chat.completions), 'create', create):
await llm.ainvoke([UserMessage(content='question')])
request_kwargs = create.await_args_list[0].kwargs
assert request_kwargs['top_p'] == 0.9
assert request_kwargs['seed'] == 42
def test_provider_key_does_not_fall_back_to_openai_key(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv('OPENAI_API_KEY', 'wrong-provider-key')
monkeypatch.setenv('OPENROUTER_API_KEY', 'openrouter-key')
assert ChatOpenRouter(model='openai/gpt-4o').get_client().api_key == 'openrouter-key'
monkeypatch.delenv('OPENROUTER_API_KEY')
with pytest.raises(ModelProviderError, match='Missing OpenRouter API key') as exc_info:
ChatOpenRouter(model='openai/gpt-4o').get_client()
assert exc_info.value.status_code == 401
async def test_empty_choices_raise_provider_error():
llm = ChatOpenRouter(model='openai/gpt-4o', api_key='test-key')
create = AsyncMock(return_value=_completion(choices=False))
with patch.object(type(llm.get_client().chat.completions), 'create', create):
with pytest.raises(ModelProviderError, match='missing or empty `choices`') as exc_info:
await llm.ainvoke([UserMessage(content='question')])
assert exc_info.value.status_code == 502
async def test_structured_provider_error_keeps_status_code():
llm = ChatOpenRouter(model='openai/gpt-4o', api_key='test-key')
create = AsyncMock(return_value=_completion(content=None))
with patch.object(type(llm.get_client().chat.completions), 'create', create):
with pytest.raises(ModelProviderError, match='Failed to parse structured output') as exc_info:
await llm.ainvoke([UserMessage(content='question')], Answer)
assert exc_info.value.status_code == 500
+18
View File
@@ -0,0 +1,18 @@
"""Regression tests for Vercel AI Gateway client setup."""
import pytest
from browser_use.llm.exceptions import ModelProviderError
from browser_use.llm.vercel.chat import ChatVercel
async def test_provider_key_does_not_fall_back_to_openai_key(monkeypatch: pytest.MonkeyPatch):
monkeypatch.setenv('OPENAI_API_KEY', 'wrong-provider-key')
monkeypatch.setenv('AI_GATEWAY_API_KEY', 'gateway-key')
assert ChatVercel(model='openai/gpt-4o').get_client().api_key == 'gateway-key'
monkeypatch.delenv('AI_GATEWAY_API_KEY')
monkeypatch.delenv('VERCEL_OIDC_TOKEN', raising=False)
with pytest.raises(ModelProviderError, match='Missing Vercel AI Gateway API key') as exc_info:
await ChatVercel(model='openai/gpt-4o').ainvoke([])
assert exc_info.value.status_code == 401
+47
View File
@@ -623,3 +623,50 @@ def test_password_field_without_type_attribute():
attrs_str = DOMTreeSerializer._build_attributes_string(node, list(DEFAULT_INCLUDE_ATTRIBUTES), '')
assert value in attrs_str, 'Input without type attribute should preserve its value'
def test_history_filters_sensitive_data_inside_nested_lists(tmp_path):
"""
Saved history must not leak sensitive values that sit below a nested-list
boundary in an action parameter (e.g. a list of rows, each row a list).
"""
from typing import Any
from pydantic import create_model
from browser_use.agent.views import AgentHistory, AgentHistoryList, AgentOutput
from browser_use.browser.views import BrowserStateHistory
from browser_use.tools.registry.views import ActionModel
class NestedInputAction(BaseModel):
rows: list[list[str]]
lookup: dict[str, list[dict[str, str]]]
InputActionModel = create_model('InputActionModel', __base__=ActionModel, input=(NestedInputAction | None, None))
OutputModel = AgentOutput.type_with_custom_actions(InputActionModel)
# built via model_validate because create_model's field is invisible to static analysis
action = InputActionModel.model_validate(
{
'input': {
'rows': [['token-123']],
'lookup': {'headers': [{'authorization': 'token-123'}]},
}
}
)
history = AgentHistoryList[Any](
history=[
AgentHistory(
model_output=OutputModel(memory='', action=[action]),
result=[],
state=BrowserStateHistory(url='https://example.test', title='t', tabs=[], interacted_element=[None]),
)
]
)
filepath = tmp_path / 'history.json'
history.save_to_file(filepath, sensitive_data={'api_key': 'token-123'})
saved = filepath.read_text(encoding='utf-8')
assert 'token-123' not in saved, 'Sensitive value leaked into the saved history file'
assert saved.count('<secret>api_key</secret>') == 2
+68
View File
@@ -0,0 +1,68 @@
"""The DOM the agent sees must show the value a form field currently holds.
Regression test for #5647: values set by JavaScript, autofill, or a framework
live in the element's `value` property, not in the `value` attribute, so the
agent used to see pre-filled fields as empty.
"""
import pytest
from pytest_httpserver import HTTPServer
from browser_use.browser.events import NavigateToUrlEvent
PAGE = """<!DOCTYPE html>
<html><head><title>Prefilled form</title></head>
<body>
<label for="name">Name</label>
<input id="name" type="text" placeholder="Your name">
<label for="notes">Notes</label>
<textarea id="notes"></textarea>
<input id="secret" type="password">
<input id="otp" type="text" autocomplete="one-time-code">
<input id="card" type="text" autocomplete="cc-number">
<input id="agree" type="checkbox" checked>
<input id="news" type="checkbox">
<script>
document.getElementById('name').value = 'Ada Lovelace';
document.getElementById('notes').value = 'call back tuesday';
document.getElementById('secret').value = 'hunter2';
document.getElementById('otp').value = '493021';
document.getElementById('card').value = '4242424242424242';
document.getElementById('agree').checked = false;
document.getElementById('news').checked = true;
</script>
</body></html>"""
@pytest.fixture(scope='module')
def http_server():
server = HTTPServer()
server.start()
server.expect_request('/prefilled').respond_with_data(PAGE, content_type='text/html')
yield server
server.stop()
async def test_live_input_values_reach_the_agent(browser_session, http_server):
event = browser_session.event_bus.dispatch(NavigateToUrlEvent(url=http_server.url_for('/prefilled')))
await event
await event.event_result(raise_if_any=True, raise_if_none=False)
state = await browser_session.get_browser_state_summary()
by_id = {node.attributes.get('id'): node for node in state.dom_state.selector_map.values() if node.attributes}
assert by_id['name'].attributes.get('value') == 'Ada Lovelace'
assert by_id['notes'].attributes.get('value') == 'call back tuesday'
assert by_id['secret'].attributes.get('value') is None, 'password values must not be exposed'
assert by_id['otp'].attributes.get('value') is None, 'one-time codes must not be exposed'
assert by_id['card'].attributes.get('value') is None, 'card numbers must not be exposed'
assert by_id['secret'].snapshot_node is not None and by_id['secret'].snapshot_node.input_value is None
assert by_id['agree'].attributes.get('checked') is None, 'live unchecked state wins over the checked attribute'
assert by_id['news'].attributes.get('checked') == 'true'
llm_view = state.dom_state.llm_representation()
assert 'Ada Lovelace' in llm_view
assert 'call back tuesday' in llm_view
assert 'hunter2' not in llm_view
assert '493021' not in llm_view
assert '4242424242424242' not in llm_view
+65 -1
View File
@@ -7,7 +7,7 @@ from pathlib import Path
import pytest
from PIL import Image
from browser_use.filesystem.file_system import FileSystem
from browser_use.filesystem.file_system import Base64BinaryFile, FileSystem
class TestImageFiles:
@@ -247,3 +247,67 @@ class TestActionResultImages:
if __name__ == '__main__':
pytest.main([__file__, '-v'])
class TestBinaryFileCreation:
"""The agent can author a tiny binary file (base64) that becomes a real, uploadable image."""
PNG_1X1 = 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAAC0lEQVR4nGNgAAIAAAUAAXpeqz8AAAAASUVORK5CYII='
async def test_write_png_produces_real_bytes(self, tmp_path: Path):
fs = FileSystem(str(tmp_path))
# write_file action appends a trailing newline; that must be tolerated
result = await fs.write_file('logo.png', self.PNG_1X1 + '\n')
assert 'successfully' in result
raw = (fs.get_dir() / 'logo.png').read_bytes()
assert raw[:8] == b'\x89PNG\r\n\x1a\n' # real PNG magic, not base64 text
assert len(raw) > 0
async def test_binary_file_is_uploadable_by_basename(self, tmp_path: Path):
fs = FileSystem(str(tmp_path))
await fs.write_file('pic.gif', 'R0lGODlhAQABAIAAAAAAAP///yH5BAEAAAAALAAAAAABAAEAAAIBRAA7')
fobj = fs.get_file('pic.gif') # the exact resolution upload_file uses
assert fobj is not None
assert fobj.full_name == 'pic.gif'
async def test_describe_does_not_leak_base64(self, tmp_path: Path):
fs = FileSystem(str(tmp_path))
await fs.write_file('logo.png', self.PNG_1X1)
desc = fs.describe()
assert self.PNG_1X1[:24] not in desc # base64 never enters the prompt
assert '[binary png file' in desc
async def test_invalid_base64_is_rejected_no_corrupt_file(self, tmp_path: Path):
fs = FileSystem(str(tmp_path))
result = await fs.write_file('bad.png', 'this is definitely not base64 @@@')
assert 'Error' in result
assert not (fs.get_dir() / 'bad.png').exists() # no 0-byte / corrupt file left for upload
# No ghost entry in the in-memory filesystem / state either
assert 'bad.png' not in fs.list_files()
assert 'bad.png' not in [f for f in fs.get_state().model_dump().get('files', {})]
async def test_valid_base64_but_not_an_image_is_rejected(self, tmp_path: Path):
"""'aGVsbG8=' is valid base64 (-> b'hello') but not a PNG: must be rejected, not written."""
fs = FileSystem(str(tmp_path))
result = await fs.write_file('fake.png', 'aGVsbG8=')
assert 'Error' in result
assert not (fs.get_dir() / 'fake.png').exists()
assert 'fake.png' not in fs.list_files()
async def test_wrong_magic_for_extension_is_rejected(self, tmp_path: Path):
"""PNG bytes written under a .gif name are rejected (magic mismatch)."""
fs = FileSystem(str(tmp_path))
result = await fs.write_file('mislabeled.gif', self.PNG_1X1)
assert 'Error' in result
assert 'mislabeled.gif' not in fs.list_files()
async def test_state_round_trip_preserves_binary(self, tmp_path: Path):
fs = FileSystem(str(tmp_path))
await fs.write_file('logo.png', self.PNG_1X1)
original_bytes = (fs.get_dir() / 'logo.png').read_bytes()
fs2 = FileSystem.from_state(fs.get_state())
restored = fs2.get_file('logo.png')
assert isinstance(restored, Base64BinaryFile)
# Content integrity, not just existence: decoded bytes must round-trip and be a real PNG
assert restored._decoded() == original_bytes
assert restored._decoded()[:8] == b'\x89PNG\r\n\x1a\n'
@@ -81,6 +81,79 @@ def test_image_only_interactive_parent_includes_child_image_context_in_llm_dom()
assert 'acme-bank-primary-card.png' in llm_dom
def test_direct_interactive_image_includes_own_context_in_llm_dom():
"""A directly clickable image should expose its own sanitized source context."""
image = _make_element_node(
221,
'img',
{'src': 'https://cdn.example.test/logos/acme-card.png?token=must-not-leak#preview'},
x=18,
y=18,
)
llm_dom = DOMTreeSerializer.serialize_tree(
SimplifiedNode(
original_node=image,
children=[],
is_interactive=True,
selector_index=221,
),
DEFAULT_INCLUDE_ATTRIBUTES,
)
assert '[221]<img' in llm_dom
assert 'image_src=acme-card.png' in llm_dom
assert 'must-not-leak' not in llm_dom
def test_direct_interactive_image_rejects_browser_normalized_data_src():
"""Data URLs remain hidden when URL parsing ignores control characters in the scheme."""
image = _make_element_node(
222,
'img',
{'src': '\x00Da\nTa:image/png;base64,must-not-leak'},
x=18,
y=18,
)
llm_dom = DOMTreeSerializer.serialize_tree(
SimplifiedNode(
original_node=image,
children=[],
is_interactive=True,
selector_index=222,
),
DEFAULT_INCLUDE_ATTRIBUTES,
)
assert llm_dom == '[222]<img />'
assert 'must-not-leak' not in llm_dom
def test_direct_interactive_image_omits_oversized_src_context():
"""Image context does not scan or serialize oversized raw source attributes."""
image = _make_element_node(
223,
'img',
{'src': f'/{"a" * DOMTreeSerializer.MAX_IMAGE_CONTEXT_ATTRIBUTE_LENGTH}/must-not-leak.png'},
x=18,
y=18,
)
llm_dom = DOMTreeSerializer.serialize_tree(
SimplifiedNode(
original_node=image,
children=[],
is_interactive=True,
selector_index=223,
),
DEFAULT_INCLUDE_ATTRIBUTES,
)
assert llm_dom == '[223]<img />'
assert 'must-not-leak' not in llm_dom
def _simplified_image(backend_node_id: int, attributes: dict[str, str]) -> SimplifiedNode:
image = _make_element_node(backend_node_id, 'img', attributes, x=18, y=18)
return SimplifiedNode(original_node=image, children=[])
+6 -6
View File
@@ -30,12 +30,12 @@ async def test_mcp_tool_isError_true_is_surfaced_as_action_result_error():
client.session.call_tool = AsyncMock( # type: ignore[union-attr]
return_value=types.CallToolResult(
content=[types.TextContent(type='text', text='File not found: /tmp/does-not-exist.txt')],
isError=True,
is_error=True,
)
)
tools = Tools()
tool = types.Tool(name='read_file', description='Read a file', inputSchema={'type': 'object', 'properties': {}})
tool = types.Tool(name='read_file', description='Read a file', input_schema={'type': 'object', 'properties': {}})
client._register_tool_as_action(tools.registry, 'read_file', tool)
result = await tools.registry.execute_action('read_file', {})
@@ -52,7 +52,7 @@ async def test_parameterized_mcp_tool_isError_true_is_surfaced_as_action_result_
client.session.call_tool = AsyncMock( # type: ignore[union-attr]
return_value=types.CallToolResult(
content=[types.TextContent(type='text', text='File not found: /tmp/does-not-exist.txt')],
isError=True,
is_error=True,
)
)
@@ -60,7 +60,7 @@ async def test_parameterized_mcp_tool_isError_true_is_surfaced_as_action_result_
tool = types.Tool(
name='read_file',
description='Read a file',
inputSchema={
input_schema={
'type': 'object',
'properties': {'path': {'type': 'string'}},
'required': ['path'],
@@ -83,12 +83,12 @@ async def test_mcp_tool_isError_false_still_succeeds():
client.session.call_tool = AsyncMock( # type: ignore[union-attr]
return_value=types.CallToolResult(
content=[types.TextContent(type='text', text='ok')],
isError=False,
is_error=False,
)
)
tools = Tools()
tool = types.Tool(name='read_file', description='Read a file', inputSchema={'type': 'object', 'properties': {}})
tool = types.Tool(name='read_file', description='Read a file', input_schema={'type': 'object', 'properties': {}})
client._register_tool_as_action(tools.registry, 'read_file', tool)
result = await tools.registry.execute_action('read_file', {})
+21 -5
View File
@@ -35,14 +35,13 @@ def server() -> BrowserUseServer:
def _is_read_only(tool: types.Tool) -> bool:
return tool.annotations is not None and tool.annotations.readOnlyHint is True
return tool.annotations is not None and tool.annotations.read_only_hint is True
async def _list_tools(server: BrowserUseServer) -> list[types.Tool]:
handler = server.server.request_handlers[types.ListToolsRequest]
result = await handler(types.ListToolsRequest(method='tools/list'))
assert isinstance(result, types.ServerResult), f'expected ServerResult, got {type(result).__name__}'
list_result = result.root
handler = server.server.get_request_handler('tools/list')
assert handler is not None, 'tools/list handler is not registered'
list_result = await handler.handler(None, types.PaginatedRequestParams()) # type: ignore[arg-type]
assert isinstance(list_result, types.ListToolsResult), f'expected ListToolsResult, got {type(list_result).__name__}'
assert len(list_result.tools) > 0, 'tools/list returned an empty catalogue'
return list_result.tools
@@ -73,3 +72,20 @@ async def test_mutating_tools_never_advertise_read_only_hint(server: BrowserUseS
mislabeled = sorted(tool.name for tool in tools if tool.name not in EXPECTED_READ_ONLY_TOOLS and _is_read_only(tool))
assert not mislabeled, f'state-changing tools wrongly advertise readOnlyHint=True: {mislabeled}'
async def test_unknown_tool_is_reported_as_mcp_error(server: BrowserUseServer) -> None:
"""Unknown tool calls must not be returned as successful MCP results."""
handler = server.server.get_request_handler('tools/call')
assert handler is not None, 'tools/call handler is not registered'
result = await handler.handler(
None, # type: ignore[arg-type]
types.CallToolRequestParams(name='does_not_exist', arguments={}),
)
assert isinstance(result, types.CallToolResult)
assert result.is_error is True
assert len(result.content) == 1
assert isinstance(result.content[0], types.TextContent)
assert 'Unknown tool: does_not_exist' in result.content[0].text