mirror of
https://github.com/browser-use/browser-use.git
synced 2026-10-02 04:04:36 +08:00
Merge remote-tracking branch 'origin/main' into pr5636
This commit is contained in:
@@ -35,7 +35,7 @@ jobs:
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.create_tag
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
- name: Create pre-release tag
|
||||
run: |
|
||||
git fetch --tags
|
||||
@@ -107,9 +107,10 @@ jobs:
|
||||
exit 1
|
||||
fi
|
||||
echo "::notice::Environment '${ENV_NAME}' is protected with prevent_self_review. OK."
|
||||
- uses: actions/checkout@v4
|
||||
- uses: astral-sh/setup-uv@v6
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
- uses: astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e # v6
|
||||
with:
|
||||
version: "0.12.9"
|
||||
enable-cache: true
|
||||
activate-environment: true
|
||||
- run: uv sync
|
||||
|
||||
+13
-18
@@ -521,6 +521,18 @@ class AgentHistory(BaseModel):
|
||||
|
||||
return redact_sensitive_string(value, sensitive_values)
|
||||
|
||||
def _filter_sensitive_data_from_value(self, value: Any, sensitive_data: dict[str, str | dict[str, str]] | None) -> Any:
|
||||
"""Recursively filter sensitive data from any supported container or string value"""
|
||||
if isinstance(value, str):
|
||||
return self._filter_sensitive_data_from_string(value, sensitive_data)
|
||||
if isinstance(value, dict):
|
||||
return {key: self._filter_sensitive_data_from_value(item, sensitive_data) for key, item in value.items()}
|
||||
if isinstance(value, list):
|
||||
return [self._filter_sensitive_data_from_value(item, sensitive_data) for item in value]
|
||||
if isinstance(value, tuple):
|
||||
return tuple(self._filter_sensitive_data_from_value(item, sensitive_data) for item in value)
|
||||
return value
|
||||
|
||||
def _filter_sensitive_data_from_dict(
|
||||
self, data: dict[str, Any], sensitive_data: dict[str, str | dict[str, str]] | None
|
||||
) -> dict[str, Any]:
|
||||
@@ -528,24 +540,7 @@ class AgentHistory(BaseModel):
|
||||
if not sensitive_data:
|
||||
return data
|
||||
|
||||
filtered_data = {}
|
||||
for key, value in data.items():
|
||||
if isinstance(value, str):
|
||||
filtered_data[key] = self._filter_sensitive_data_from_string(value, sensitive_data)
|
||||
elif isinstance(value, dict):
|
||||
filtered_data[key] = self._filter_sensitive_data_from_dict(value, sensitive_data)
|
||||
elif isinstance(value, list):
|
||||
filtered_data[key] = [
|
||||
self._filter_sensitive_data_from_string(item, sensitive_data)
|
||||
if isinstance(item, str)
|
||||
else self._filter_sensitive_data_from_dict(item, sensitive_data)
|
||||
if isinstance(item, dict)
|
||||
else item
|
||||
for item in value
|
||||
]
|
||||
else:
|
||||
filtered_data[key] = value
|
||||
return filtered_data
|
||||
return {key: self._filter_sensitive_data_from_value(value, sensitive_data) for key, value in data.items()}
|
||||
|
||||
def model_dump(self, sensitive_data: dict[str, str | dict[str, str]] | None = None, **kwargs) -> dict[str, Any]:
|
||||
"""Custom serialization handling circular references and filtering sensitive data"""
|
||||
|
||||
@@ -690,11 +690,15 @@ class BrowserSession(BaseModel):
|
||||
self.logger.info('✅ Browser session reset complete')
|
||||
|
||||
def model_post_init(self, __context) -> None:
|
||||
"""Register event handlers after model initialization."""
|
||||
"""Initialize runtime state and register event handlers."""
|
||||
self._connection_lock = asyncio.Lock()
|
||||
# Initialize reconnect event as set (no reconnection pending)
|
||||
self._reconnect_event = asyncio.Event()
|
||||
self._reconnect_event.set()
|
||||
self._register_session_event_handlers()
|
||||
|
||||
def _register_session_event_handlers(self) -> None:
|
||||
"""Register BrowserSession handlers on the current event bus."""
|
||||
|
||||
# Check if handlers are already registered to prevent duplicates
|
||||
from browser_use.browser.watchdog_base import BaseWatchdog
|
||||
@@ -746,6 +750,7 @@ class BrowserSession(BaseModel):
|
||||
await self.reset()
|
||||
# Create fresh event bus
|
||||
self.event_bus = ResilientEventBus()
|
||||
self._register_session_event_handlers()
|
||||
|
||||
async def stop(self) -> None:
|
||||
"""Stop the browser session without killing the browser process.
|
||||
@@ -771,6 +776,7 @@ class BrowserSession(BaseModel):
|
||||
await self.reset()
|
||||
# Create fresh event bus
|
||||
self.event_bus = ResilientEventBus()
|
||||
self._register_session_event_handlers()
|
||||
|
||||
async def close(self) -> None:
|
||||
"""Alias for stop()."""
|
||||
|
||||
@@ -2504,31 +2504,34 @@ class DefaultActionWatchdog(BaseWatchdog):
|
||||
'end': 'End',
|
||||
}
|
||||
|
||||
# Parse and normalize the key string
|
||||
keys = event.keys
|
||||
if '+' in keys:
|
||||
# Handle key combinations like "ctrl+a"
|
||||
parts = keys.split('+')
|
||||
normalized_parts = []
|
||||
for part in parts:
|
||||
part_lower = part.strip().lower()
|
||||
normalized = key_aliases.get(part_lower, part)
|
||||
normalized_parts.append(normalized)
|
||||
normalized_keys = '+'.join(normalized_parts)
|
||||
else:
|
||||
# Single key
|
||||
keys_lower = keys.strip().lower()
|
||||
normalized_keys = key_aliases.get(keys_lower, keys)
|
||||
|
||||
# Handle key combinations like "Control+A"
|
||||
if '+' in normalized_keys:
|
||||
parts = normalized_keys.split('+')
|
||||
modifiers = parts[:-1]
|
||||
main_key = parts[-1]
|
||||
modifier_map = {'Alt': 1, 'Control': 2, 'Meta': 4, 'Shift': 8}
|
||||
is_combination = False
|
||||
modifiers = []
|
||||
main_key = None
|
||||
if '+' in keys and keys != '+':
|
||||
if keys.endswith('++'):
|
||||
prefix = keys[:-2]
|
||||
raw_modifiers = prefix.split('+')
|
||||
if all(part.strip() for part in raw_modifiers):
|
||||
normalized_modifiers = [key_aliases.get(part.strip().lower(), part) for part in raw_modifiers]
|
||||
if all(modifier in modifier_map for modifier in normalized_modifiers):
|
||||
is_combination = True
|
||||
modifiers = normalized_modifiers
|
||||
main_key = '+'
|
||||
else:
|
||||
prefix, suffix = keys.rsplit('+', 1)
|
||||
raw_modifiers = prefix.split('+')
|
||||
if suffix.strip() and all(part.strip() for part in raw_modifiers):
|
||||
normalized_modifiers = [key_aliases.get(part.strip().lower(), part) for part in raw_modifiers]
|
||||
if all(modifier in modifier_map for modifier in normalized_modifiers):
|
||||
is_combination = True
|
||||
modifiers = normalized_modifiers
|
||||
main_key = key_aliases.get(suffix.strip().lower(), suffix)
|
||||
|
||||
if is_combination and main_key is not None:
|
||||
# Calculate modifier bitmask
|
||||
modifier_value = 0
|
||||
modifier_map = {'Alt': 1, 'Control': 2, 'Meta': 4, 'Shift': 8}
|
||||
for mod in modifiers:
|
||||
modifier_value |= modifier_map.get(mod, 0)
|
||||
|
||||
@@ -2545,6 +2548,9 @@ class DefaultActionWatchdog(BaseWatchdog):
|
||||
for mod in reversed(modifiers):
|
||||
await self._dispatch_key_event(cdp_session, 'keyUp', mod)
|
||||
else:
|
||||
keys_lower = keys.strip().lower()
|
||||
normalized_keys = key_aliases.get(keys_lower, keys)
|
||||
|
||||
# Check if this is a text string or special key
|
||||
special_keys = {
|
||||
'Enter',
|
||||
|
||||
@@ -157,7 +157,7 @@ class LocalBrowserWatchdog(BaseWatchdog):
|
||||
process = psutil.Process(subprocess.pid)
|
||||
|
||||
# Wait for CDP to be ready and get the URL
|
||||
cdp_url = await self._wait_for_cdp_url(debug_port)
|
||||
cdp_url = await self._wait_for_cdp_url(debug_port, process=process)
|
||||
|
||||
# Success! Clean up only the temp dirs we created but didn't use
|
||||
currently_used_dir = str(profile.user_data_dir)
|
||||
@@ -405,13 +405,42 @@ class LocalBrowserWatchdog(BaseWatchdog):
|
||||
return port
|
||||
|
||||
@staticmethod
|
||||
async def _wait_for_cdp_url(port: int, timeout: float = 30) -> str:
|
||||
"""Wait for the browser to start and return the CDP URL."""
|
||||
async def _wait_for_cdp_url(port: int, timeout: float = 30, process: psutil.Process | None = None) -> str:
|
||||
"""Wait for the browser to start and return the CDP URL.
|
||||
|
||||
Args:
|
||||
port: The local port Chrome is listening on for CDP.
|
||||
timeout: Maximum seconds to wait before raising TimeoutError.
|
||||
process: Optional psutil.Process for the browser subprocess. If provided,
|
||||
the loop will fail fast with a descriptive error if the process
|
||||
exits before CDP becomes available (e.g. due to missing sandbox
|
||||
capabilities or a missing display on headless Linux).
|
||||
"""
|
||||
import aiohttp
|
||||
|
||||
start_time = asyncio.get_event_loop().time()
|
||||
start_time = asyncio.get_running_loop().time()
|
||||
|
||||
while asyncio.get_running_loop().time() - start_time < timeout:
|
||||
# Fail fast if the browser process has already exited
|
||||
if process is not None:
|
||||
try:
|
||||
if not process.is_running():
|
||||
raise RuntimeError(
|
||||
f'Browser process (PID {process.pid}) exited before CDP became available on port {port}. '
|
||||
'This usually means Chrome failed to start — check that it is properly installed and '
|
||||
'all required system dependencies are present (e.g. --no-sandbox may be needed in '
|
||||
'Docker/headless environments, or a virtual display such as Xvfb on headless Linux).'
|
||||
)
|
||||
except psutil.NoSuchProcess:
|
||||
raise RuntimeError(
|
||||
f'Browser process (PID {process.pid}) exited before CDP became available on port {port}. '
|
||||
'This usually means Chrome failed to start — check that it is properly installed and '
|
||||
'all required system dependencies are present (e.g. --no-sandbox may be needed in '
|
||||
'Docker/headless environments, or a virtual display such as Xvfb on headless Linux).'
|
||||
)
|
||||
except psutil.AccessDenied:
|
||||
pass # Cannot check process status; continue polling CDP
|
||||
|
||||
while asyncio.get_event_loop().time() - start_time < timeout:
|
||||
try:
|
||||
async with aiohttp.ClientSession() as session:
|
||||
async with session.get(f'http://127.0.0.1:{port}/json/version') as resp:
|
||||
|
||||
@@ -43,6 +43,30 @@ def _parse_computed_styles(strings: list[str], style_indices: list[int]) -> dict
|
||||
return styles
|
||||
|
||||
|
||||
# Live values of these fields never leave the snapshot: they would otherwise be
|
||||
# serialized by EnhancedDOMTreeNode.__json__ and could reach logs or the LLM.
|
||||
_SENSITIVE_INPUT_TYPES = frozenset({'password', 'file', 'hidden'})
|
||||
_SENSITIVE_AUTOCOMPLETE_PREFIXES = ('cc-', 'one-time-code')
|
||||
|
||||
|
||||
def _is_sensitive_input(strings: list[str], nodes: NodeTreeSnapshot, snapshot_index: int) -> bool:
|
||||
"""True for password/file/hidden inputs and payment or one-time-code autocomplete fields."""
|
||||
attribute_lists = nodes.get('attributes')
|
||||
if not attribute_lists or snapshot_index >= len(attribute_lists):
|
||||
return False
|
||||
indices = attribute_lists[snapshot_index]
|
||||
for name_index, value_index in zip(indices[0::2], indices[1::2]):
|
||||
if not (0 <= name_index < len(strings) and 0 <= value_index < len(strings)):
|
||||
continue
|
||||
name = strings[name_index].lower()
|
||||
value = strings[value_index].lower()
|
||||
if name == 'type' and value in _SENSITIVE_INPUT_TYPES:
|
||||
return True
|
||||
if name == 'autocomplete' and value.startswith(_SENSITIVE_AUTOCOMPLETE_PREFIXES):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def build_snapshot_lookup(
|
||||
snapshot: CaptureSnapshotReturns,
|
||||
device_pixel_ratio: float = 1.0,
|
||||
@@ -91,6 +115,18 @@ def build_snapshot_lookup(
|
||||
has_clickable_data = 'isClickable' in nodes
|
||||
is_clickable_set: set[int] = set(nodes['isClickable']['index']) if has_clickable_data else set()
|
||||
|
||||
# Live form values live in the snapshot, not in the DOM attributes. Map
|
||||
# snapshot index -> string once so each node lookup stays O(1).
|
||||
input_value_by_index: dict[int, str] = {}
|
||||
for key in ('inputValue', 'textValue'):
|
||||
rare = nodes.get(key)
|
||||
if rare:
|
||||
for idx, string_index in zip(rare.get('index', []), rare.get('value', [])):
|
||||
if 0 <= string_index < len(strings) and not _is_sensitive_input(strings, nodes, idx):
|
||||
input_value_by_index[idx] = strings[string_index]
|
||||
input_checked_set: set[int] = set(nodes['inputChecked']['index']) if 'inputChecked' in nodes else set()
|
||||
has_checked_data = 'inputChecked' in nodes
|
||||
|
||||
# Build snapshot lookup for each backend node id
|
||||
for backend_node_id, snapshot_index in backend_node_to_snapshot_index.items():
|
||||
is_clickable = None
|
||||
@@ -173,6 +209,8 @@ def build_snapshot_lookup(
|
||||
computed_styles=computed_styles if computed_styles else None,
|
||||
paint_order=paint_order,
|
||||
stacking_contexts=stacking_contexts,
|
||||
input_value=input_value_by_index.get(snapshot_index),
|
||||
input_checked=(snapshot_index in input_checked_set) if has_checked_data else None,
|
||||
)
|
||||
|
||||
# Count how many have bounds (are actually visible/laid out)
|
||||
|
||||
@@ -16,6 +16,7 @@ from browser_use.dom.views import (
|
||||
)
|
||||
|
||||
DISABLED_ELEMENTS = {'style', 'script', 'head', 'meta', 'link', 'title'}
|
||||
URL_C0_CONTROL_OR_SPACE = ''.join(chr(codepoint) for codepoint in range(0x21))
|
||||
|
||||
# SVG child elements to skip (decorative only, no interaction value)
|
||||
SVG_ELEMENTS = {
|
||||
@@ -57,6 +58,7 @@ class DOMTreeSerializer:
|
||||
DEFAULT_CONTAINMENT_THRESHOLD = 0.99 # 99% containment by default
|
||||
MAX_CHILD_IMAGE_CONTEXTS = 3
|
||||
MAX_CHILD_IMAGE_DESCENDANTS = 100
|
||||
MAX_IMAGE_CONTEXT_ATTRIBUTE_LENGTH = 4096
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
@@ -922,17 +924,47 @@ class DOMTreeSerializer:
|
||||
|
||||
@staticmethod
|
||||
def _get_child_image_context(node: SimplifiedNode) -> str:
|
||||
"""Extract compact context from image descendants of an interactive element."""
|
||||
"""Extract compact context from an interactive image and its descendants."""
|
||||
|
||||
image_context: list[str] = []
|
||||
|
||||
def normalize_src(src: str) -> str:
|
||||
clean_src = src.strip()
|
||||
if len(src) > DOMTreeSerializer.MAX_IMAGE_CONTEXT_ATTRIBUTE_LENGTH:
|
||||
return ''
|
||||
clean_src = src.strip(URL_C0_CONTROL_OR_SPACE).replace('\t', '').replace('\n', '').replace('\r', '')
|
||||
if clean_src.lower().startswith('data:'):
|
||||
return ''
|
||||
path_without_query = clean_src.split('?', 1)[0].split('#', 1)[0].rstrip('/')
|
||||
return path_without_query.rsplit('/', 1)[-1]
|
||||
|
||||
def add_image_context(original_node: EnhancedDOMTreeNode) -> None:
|
||||
if original_node.node_type != NodeType.ELEMENT_NODE or original_node.tag_name != 'img':
|
||||
return
|
||||
|
||||
attributes = original_node.attributes or {}
|
||||
parts = []
|
||||
|
||||
for attr_name, output_name in (
|
||||
('alt', 'image_alt'),
|
||||
('title', 'image_title'),
|
||||
('aria-label', 'image_label'),
|
||||
):
|
||||
raw_attr_value = str(attributes.get(attr_name) or '')
|
||||
if len(raw_attr_value) > DOMTreeSerializer.MAX_IMAGE_CONTEXT_ATTRIBUTE_LENGTH:
|
||||
continue
|
||||
attr_value = raw_attr_value.strip()
|
||||
if attr_value:
|
||||
parts.append(f'{output_name}={cap_text_length(attr_value, 100)}')
|
||||
|
||||
src = normalize_src(str(attributes.get('src') or ''))
|
||||
if src:
|
||||
parts.append(f'image_src={cap_text_length(src, 100)}')
|
||||
|
||||
if parts:
|
||||
image_context.append(' '.join(parts))
|
||||
|
||||
add_image_context(node.original_node)
|
||||
|
||||
child_iterators = [iter(node.children)]
|
||||
visited_descendants = 0
|
||||
while (
|
||||
@@ -946,26 +978,7 @@ class DOMTreeSerializer:
|
||||
child_iterators.pop()
|
||||
continue
|
||||
visited_descendants += 1
|
||||
original_node = current.original_node
|
||||
if original_node.node_type == NodeType.ELEMENT_NODE and original_node.tag_name == 'img':
|
||||
attributes = original_node.attributes or {}
|
||||
parts = []
|
||||
|
||||
for attr_name, output_name in (
|
||||
('alt', 'image_alt'),
|
||||
('title', 'image_title'),
|
||||
('aria-label', 'image_label'),
|
||||
):
|
||||
attr_value = str(attributes.get(attr_name) or '').strip()
|
||||
if attr_value:
|
||||
parts.append(f'{output_name}={cap_text_length(attr_value, 100)}')
|
||||
|
||||
src = normalize_src(str(attributes.get('src') or ''))
|
||||
if src:
|
||||
parts.append(f'image_src={cap_text_length(src, 100)}')
|
||||
|
||||
if parts:
|
||||
image_context.append(' '.join(parts))
|
||||
add_image_context(current.original_node)
|
||||
|
||||
if current.children:
|
||||
child_iterators.append(iter(current.children))
|
||||
|
||||
@@ -811,6 +811,20 @@ class DomService:
|
||||
# Get snapshot data and calculate absolute position
|
||||
snapshot_data = snapshot_lookup.get(node['backendNodeId'], None)
|
||||
|
||||
# Show the value a field currently holds, not only the static attribute.
|
||||
# JS, autofill, and framework bindings set the property without touching
|
||||
# the attribute, so the agent used to see pre-filled fields as empty (#5647).
|
||||
if snapshot_data and node['nodeName'].upper() in ('INPUT', 'TEXTAREA'):
|
||||
if snapshot_data.input_value is not None:
|
||||
attributes = dict(attributes or {})
|
||||
attributes['value'] = snapshot_data.input_value
|
||||
if snapshot_data.input_checked is not None:
|
||||
attributes = dict(attributes or {})
|
||||
if snapshot_data.input_checked:
|
||||
attributes['checked'] = 'true'
|
||||
else:
|
||||
attributes.pop('checked', None)
|
||||
|
||||
# DIAGNOSTIC: Log when interactive elements don't have snapshot data
|
||||
if not snapshot_data and node['nodeName'].upper() in ['INPUT', 'BUTTON', 'SELECT', 'TEXTAREA', 'A']:
|
||||
parent_has_shadow = False
|
||||
|
||||
@@ -353,6 +353,10 @@ class EnhancedSnapshotNode:
|
||||
"""Paint order from the layout tree"""
|
||||
stacking_contexts: int | None
|
||||
"""Stacking contexts from the layout tree"""
|
||||
input_value: str | None = None
|
||||
"""Live value of an <input> or <textarea> (DOMSnapshot inputValue/textValue), which the value attribute misses when JS, autofill, or a framework set it."""
|
||||
input_checked: bool | None = None
|
||||
"""Live checked state of a checkbox or radio input (DOMSnapshot inputChecked)."""
|
||||
|
||||
|
||||
# @dataclass(slots=True)
|
||||
|
||||
@@ -14,13 +14,8 @@ from typing import Any
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
UNSUPPORTED_BINARY_EXTENSIONS = {
|
||||
'png',
|
||||
'jpg',
|
||||
'jpeg',
|
||||
'gif',
|
||||
'bmp',
|
||||
'svg',
|
||||
'webp',
|
||||
'ico',
|
||||
'mp3',
|
||||
'mp4',
|
||||
@@ -394,6 +389,109 @@ class XmlFile(BaseFile):
|
||||
return 'xml'
|
||||
|
||||
|
||||
class Base64BinaryFile(BaseFile):
|
||||
"""Small binary file the agent authors as base64 text.
|
||||
|
||||
``content`` holds the base64 string; the decoded bytes are written to disk so the
|
||||
file is a real, uploadable image. ``read()`` returns a short stub instead of the
|
||||
base64 so it never bloats or confuses the agent prompt (describe() calls read()
|
||||
every step). Intended for tiny fixtures (e.g. a 1x1 PNG) for upload-validation
|
||||
flows, not for arbitrary large binaries.
|
||||
"""
|
||||
|
||||
# Leading magic bytes that identify a real file of this type. Keyed by extension so
|
||||
# base64 that decodes but isn't actually an image (e.g. 'aGVsbG8=' -> b'hello') is
|
||||
# rejected instead of written as a corrupt upload.
|
||||
_MAGIC: dict[str, tuple[bytes, ...]] = {
|
||||
'png': (b'\x89PNG\r\n\x1a\n',),
|
||||
'gif': (b'GIF87a', b'GIF89a'),
|
||||
'jpg': (b'\xff\xd8\xff',),
|
||||
'jpeg': (b'\xff\xd8\xff',),
|
||||
'webp': (b'RIFF',), # RIFF container; 'WEBP' tag checked below
|
||||
}
|
||||
|
||||
def _decoded(self) -> bytes:
|
||||
# Strip all whitespace (the write_file action appends a trailing newline) then
|
||||
# decode strictly so non-base64 text is rejected rather than silently corrupted.
|
||||
return base64.b64decode(''.join(self.content.split()), validate=True)
|
||||
|
||||
def _validate(self, content: str) -> None:
|
||||
"""Decode and confirm the bytes actually are an image of this extension. Raises FileSystemError."""
|
||||
try:
|
||||
data = base64.b64decode(''.join(content.split()), validate=True)
|
||||
except Exception as e:
|
||||
raise FileSystemError(
|
||||
f"Error: content for '{self.full_name}' is not valid base64. "
|
||||
f'For images, provide the base64 of a valid {self.extension} file. ({e})'
|
||||
)
|
||||
magic = self._MAGIC.get(self.extension, ())
|
||||
if magic and not any(data.startswith(m) for m in magic):
|
||||
raise FileSystemError(
|
||||
f"Error: content for '{self.full_name}' is valid base64 but not a {self.extension} image "
|
||||
f'(wrong magic bytes). Provide the base64 of a real {self.extension} file.'
|
||||
)
|
||||
if self.extension == 'webp' and not (data[:4] == b'RIFF' and data[8:12] == b'WEBP'):
|
||||
raise FileSystemError(f"Error: content for '{self.full_name}' is not a valid WEBP file.")
|
||||
|
||||
def write_file_content(self, content: str) -> None:
|
||||
self._validate(content)
|
||||
self.update_content(content)
|
||||
|
||||
def append_file_content(self, content: str) -> None:
|
||||
raise FileSystemError(f"Error: cannot append to binary file '{self.full_name}'. Overwrite it instead.")
|
||||
|
||||
def sync_to_disk_sync(self, path: Path) -> None:
|
||||
(path / self.full_name).write_bytes(self._decoded())
|
||||
|
||||
async def sync_to_disk(self, path: Path) -> None:
|
||||
with ThreadPoolExecutor() as executor:
|
||||
await asyncio.get_event_loop().run_in_executor(executor, lambda: self.sync_to_disk_sync(path))
|
||||
|
||||
def read(self) -> str:
|
||||
try:
|
||||
n = len(self._decoded())
|
||||
except Exception:
|
||||
return '[binary file: content is not valid base64]'
|
||||
return f'[binary {self.extension} file, {n} bytes]'
|
||||
|
||||
@property
|
||||
def get_size(self) -> int:
|
||||
try:
|
||||
return len(self._decoded())
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
|
||||
class PngFile(Base64BinaryFile):
|
||||
@property
|
||||
def extension(self) -> str:
|
||||
return 'png'
|
||||
|
||||
|
||||
class GifFile(Base64BinaryFile):
|
||||
@property
|
||||
def extension(self) -> str:
|
||||
return 'gif'
|
||||
|
||||
|
||||
class JpgFile(Base64BinaryFile):
|
||||
@property
|
||||
def extension(self) -> str:
|
||||
return 'jpg'
|
||||
|
||||
|
||||
class JpegFile(Base64BinaryFile):
|
||||
@property
|
||||
def extension(self) -> str:
|
||||
return 'jpeg'
|
||||
|
||||
|
||||
class WebpFile(Base64BinaryFile):
|
||||
@property
|
||||
def extension(self) -> str:
|
||||
return 'webp'
|
||||
|
||||
|
||||
class FileSystemState(BaseModel):
|
||||
"""Serializable state of the file system"""
|
||||
|
||||
@@ -427,6 +525,11 @@ class FileSystem:
|
||||
'docx': DocxFile,
|
||||
'html': HtmlFile,
|
||||
'xml': XmlFile,
|
||||
'png': PngFile,
|
||||
'gif': GifFile,
|
||||
'jpg': JpgFile,
|
||||
'jpeg': JpegFile,
|
||||
'webp': WebpFile,
|
||||
}
|
||||
|
||||
self.files = {}
|
||||
@@ -576,7 +679,7 @@ class FileSystem:
|
||||
return result
|
||||
|
||||
# Text-based extensions: derive from _file_types, excluding those with special readers
|
||||
_special_extensions = {'docx', 'pdf', 'jpg', 'jpeg', 'png'}
|
||||
_special_extensions = {'docx', 'pdf', 'jpg', 'jpeg', 'png', 'gif', 'webp'}
|
||||
text_extensions = [ext for ext in self._file_types if ext not in _special_extensions]
|
||||
|
||||
if extension in text_extensions:
|
||||
@@ -711,7 +814,7 @@ class FileSystem:
|
||||
)
|
||||
return result
|
||||
|
||||
elif extension in ['jpg', 'jpeg', 'png']:
|
||||
elif extension in ['jpg', 'jpeg', 'png', 'gif', 'webp']:
|
||||
import anyio
|
||||
|
||||
# Read image file and convert to base64
|
||||
@@ -786,15 +889,16 @@ class FileSystem:
|
||||
if not file_class:
|
||||
raise ValueError(f"Error: Invalid file extension '{extension}' for file '{full_filename}'.")
|
||||
|
||||
# Create or get existing file using full filename as key
|
||||
if full_filename in self.files:
|
||||
file_obj = self.files[full_filename]
|
||||
else:
|
||||
file_obj = file_class(name=name_without_ext)
|
||||
self.files[full_filename] = file_obj # Use full filename as key
|
||||
# Create or get existing file using full filename as key. A NEW file is only
|
||||
# registered after a successful write, so a failed write (e.g. invalid base64
|
||||
# for an image) leaves no ghost entry in self.files / state.
|
||||
is_new = full_filename not in self.files
|
||||
file_obj = self.files[full_filename] if not is_new else file_class(name=name_without_ext)
|
||||
|
||||
# Use file-specific write method
|
||||
await file_obj.write(content, self.data_dir)
|
||||
if is_new:
|
||||
self.files[full_filename] = file_obj
|
||||
sanitize_note = f" (auto-corrected from '{original_filename}')" if was_sanitized else ''
|
||||
return f'Data written to file {full_filename} successfully.{sanitize_note}'
|
||||
except FileSystemError as e:
|
||||
@@ -980,6 +1084,11 @@ class FileSystem:
|
||||
'DocxFile': DocxFile,
|
||||
'HtmlFile': HtmlFile,
|
||||
'XmlFile': XmlFile,
|
||||
'PngFile': PngFile,
|
||||
'GifFile': GifFile,
|
||||
'JpgFile': JpgFile,
|
||||
'JpegFile': JpegFile,
|
||||
'WebpFile': WebpFile,
|
||||
}
|
||||
|
||||
file_class = file_type_map.get(file_type)
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
import os
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, TypeVar, overload
|
||||
|
||||
import httpx
|
||||
from openai import APIConnectionError, APIStatusError, AsyncOpenAI, RateLimitError
|
||||
from openai.types.chat.chat_completion import ChatCompletion
|
||||
from openai.types.chat.chat_completion import ChatCompletion, Choice
|
||||
from openai.types.shared_params.response_format_json_schema import (
|
||||
JSONSchema,
|
||||
ResponseFormatJSONSchema,
|
||||
@@ -57,17 +58,23 @@ class ChatOpenRouter(BaseChatModel):
|
||||
|
||||
def _get_client_params(self) -> dict[str, Any]:
|
||||
"""Prepare client parameters dictionary."""
|
||||
api_key = self.api_key or os.getenv('OPENROUTER_API_KEY')
|
||||
if not api_key:
|
||||
raise ModelProviderError(
|
||||
message='Missing OpenRouter API key. Set OPENROUTER_API_KEY or pass api_key.',
|
||||
status_code=401,
|
||||
model=self.name,
|
||||
)
|
||||
|
||||
# Define base client params
|
||||
base_params = {
|
||||
'api_key': self.api_key,
|
||||
'api_key': api_key,
|
||||
'base_url': self.base_url,
|
||||
'timeout': self.timeout,
|
||||
'max_retries': self.max_retries,
|
||||
'default_headers': self.default_headers,
|
||||
'default_query': self.default_query,
|
||||
'_strict_response_validation': self._strict_response_validation,
|
||||
'top_p': self.top_p,
|
||||
'seed': self.seed,
|
||||
}
|
||||
|
||||
# Create client_params dict with non-None values
|
||||
@@ -91,6 +98,15 @@ class ChatOpenRouter(BaseChatModel):
|
||||
self._client = AsyncOpenAI(**client_params)
|
||||
return self._client
|
||||
|
||||
def _get_first_choice(self, response: ChatCompletion) -> Choice:
|
||||
if response.choices:
|
||||
return response.choices[0]
|
||||
raise ModelProviderError(
|
||||
message='Invalid OpenRouter response: missing or empty `choices`.',
|
||||
status_code=502,
|
||||
model=self.name,
|
||||
)
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
return str(self.model)
|
||||
@@ -154,9 +170,10 @@ class ChatOpenRouter(BaseChatModel):
|
||||
**(self.extra_body or {}),
|
||||
)
|
||||
|
||||
choice = self._get_first_choice(response)
|
||||
usage = self._get_usage(response)
|
||||
return ChatInvokeCompletion(
|
||||
completion=response.choices[0].message.content or '',
|
||||
completion=choice.message.content or '',
|
||||
usage=usage,
|
||||
)
|
||||
|
||||
@@ -185,7 +202,8 @@ class ChatOpenRouter(BaseChatModel):
|
||||
**(self.extra_body or {}),
|
||||
)
|
||||
|
||||
if response.choices[0].message.content is None:
|
||||
choice = self._get_first_choice(response)
|
||||
if choice.message.content is None:
|
||||
raise ModelProviderError(
|
||||
message='Failed to parse structured output from model response',
|
||||
status_code=500,
|
||||
@@ -193,13 +211,16 @@ class ChatOpenRouter(BaseChatModel):
|
||||
)
|
||||
usage = self._get_usage(response)
|
||||
|
||||
parsed = output_format.model_validate_json(response.choices[0].message.content)
|
||||
parsed = output_format.model_validate_json(choice.message.content)
|
||||
|
||||
return ChatInvokeCompletion(
|
||||
completion=parsed,
|
||||
usage=usage,
|
||||
)
|
||||
|
||||
except ModelProviderError:
|
||||
raise
|
||||
|
||||
except RateLimitError as e:
|
||||
raise ModelRateLimitError(message=e.message, model=self.name) from e
|
||||
|
||||
|
||||
@@ -348,6 +348,12 @@ class ChatVercel(BaseChatModel):
|
||||
def _get_client_params(self) -> dict[str, Any]:
|
||||
"""Prepare client parameters dictionary."""
|
||||
api_key = self.api_key or os.getenv('AI_GATEWAY_API_KEY') or os.getenv('VERCEL_OIDC_TOKEN')
|
||||
if not api_key:
|
||||
raise ModelProviderError(
|
||||
message='Missing Vercel AI Gateway API key. Set AI_GATEWAY_API_KEY or VERCEL_OIDC_TOKEN, or pass api_key.',
|
||||
status_code=401,
|
||||
model=self.name,
|
||||
)
|
||||
|
||||
base_params = {
|
||||
'api_key': api_key,
|
||||
@@ -661,6 +667,9 @@ class ChatVercel(BaseChatModel):
|
||||
stop_reason=response.choices[0].finish_reason if response.choices else None,
|
||||
)
|
||||
|
||||
except ModelProviderError:
|
||||
raise
|
||||
|
||||
except RateLimitError as e:
|
||||
raise ModelRateLimitError(message=e.message, model=self.name) from e
|
||||
|
||||
|
||||
+32
-25
@@ -12,8 +12,7 @@ from typing import Any
|
||||
|
||||
import mcp.server.stdio
|
||||
import mcp.types as types
|
||||
from mcp.server import NotificationOptions, Server
|
||||
from mcp.server.models import InitializationOptions
|
||||
from mcp.server import Server
|
||||
|
||||
from browser_use.utils import get_browser_use_version
|
||||
|
||||
@@ -34,7 +33,11 @@ class CLIMCPServer:
|
||||
"""Stateful stdio MCP server wrapping the browser-harness exec model."""
|
||||
|
||||
def __init__(self):
|
||||
self.server: Server = Server('browser-use')
|
||||
self.server: Server = Server(
|
||||
'browser-use',
|
||||
version=get_browser_use_version(),
|
||||
instructions=self._instructions(),
|
||||
)
|
||||
self._namespace: dict[str, Any] | None = None
|
||||
self._exec_lock = asyncio.Lock()
|
||||
self._register_handlers()
|
||||
@@ -50,7 +53,7 @@ class CLIMCPServer:
|
||||
'persists across calls. Returns whatever the code prints. First navigation '
|
||||
'should be new_tab(url).'
|
||||
),
|
||||
inputSchema={
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'code': {'type': 'string', 'description': 'Python code to execute'},
|
||||
@@ -61,7 +64,7 @@ class CLIMCPServer:
|
||||
types.Tool(
|
||||
name='browser_screenshot',
|
||||
description='Capture the current page and return it as an image. Prefer this over capture_screenshot() in browser_exec.',
|
||||
inputSchema={
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'full': {'type': 'boolean', 'description': 'Capture beyond the viewport (full page)', 'default': False},
|
||||
@@ -76,28 +79,40 @@ class CLIMCPServer:
|
||||
]
|
||||
|
||||
def _register_handlers(self):
|
||||
@self.server.list_tools()
|
||||
async def handle_list_tools() -> list[types.Tool]:
|
||||
return self._tool_definitions()
|
||||
async def handle_list_tools(_context: Any, _params: types.PaginatedRequestParams) -> types.ListToolsResult:
|
||||
return types.ListToolsResult(tools=self._tool_definitions())
|
||||
|
||||
@self.server.call_tool()
|
||||
async def handle_call_tool(name: str, arguments: dict[str, Any] | None) -> list[types.TextContent | types.ImageContent]:
|
||||
arguments = arguments or {}
|
||||
async def handle_call_tool(_context: Any, params: types.CallToolRequestParams) -> types.CallToolResult:
|
||||
name = params.name
|
||||
arguments = params.arguments or {}
|
||||
if name == 'browser_exec':
|
||||
code = arguments.get('code')
|
||||
if not isinstance(code, str) or not code.strip():
|
||||
return [types.TextContent(type='text', text="Error: 'code' must be a non-empty string")]
|
||||
return types.CallToolResult(
|
||||
content=[types.TextContent(type='text', text="Error: 'code' must be a non-empty string")],
|
||||
is_error=True,
|
||||
)
|
||||
async with self._exec_lock:
|
||||
output = await asyncio.to_thread(self._execute, code)
|
||||
return [types.TextContent(type='text', text=output or '(no output)')]
|
||||
return types.CallToolResult(content=[types.TextContent(type='text', text=output or '(no output)')])
|
||||
if name == 'browser_screenshot':
|
||||
max_dim = arguments.get('max_dim')
|
||||
if max_dim is not None and (isinstance(max_dim, bool) or not isinstance(max_dim, int) or max_dim < 1):
|
||||
return [types.TextContent(type='text', text="Error: 'max_dim' must be a positive integer")]
|
||||
return types.CallToolResult(
|
||||
content=[types.TextContent(type='text', text="Error: 'max_dim' must be a positive integer")],
|
||||
is_error=True,
|
||||
)
|
||||
async with self._exec_lock:
|
||||
png = await asyncio.to_thread(self._screenshot, bool(arguments.get('full', False)), max_dim)
|
||||
return [types.ImageContent(type='image', data=png, mimeType='image/png')]
|
||||
return [types.TextContent(type='text', text=f'Unknown tool: {name}')]
|
||||
content: list[types.ContentBlock] = [types.ImageContent(type='image', data=png, mime_type='image/png')]
|
||||
return types.CallToolResult(content=content)
|
||||
return types.CallToolResult(
|
||||
content=[types.TextContent(type='text', text=f'Unknown tool: {name}')],
|
||||
is_error=True,
|
||||
)
|
||||
|
||||
self.server.add_request_handler('tools/list', types.PaginatedRequestParams, handle_list_tools)
|
||||
self.server.add_request_handler('tools/call', types.CallToolRequestParams, handle_call_tool)
|
||||
|
||||
def _ensure_namespace(self) -> dict[str, Any]:
|
||||
if self._namespace is None:
|
||||
@@ -151,15 +166,7 @@ class CLIMCPServer:
|
||||
await self.server.run(
|
||||
read_stream,
|
||||
write_stream,
|
||||
InitializationOptions(
|
||||
server_name='browser-use',
|
||||
server_version=get_browser_use_version(),
|
||||
instructions=self._instructions(),
|
||||
capabilities=self.server.get_capabilities(
|
||||
notification_options=NotificationOptions(),
|
||||
experimental_capabilities={},
|
||||
),
|
||||
),
|
||||
self.server.create_initialization_options(),
|
||||
)
|
||||
except BrokenPipeError:
|
||||
pass
|
||||
|
||||
@@ -256,10 +256,10 @@ class MCPClient:
|
||||
# Parse tool parameters to create Pydantic model
|
||||
param_fields = {}
|
||||
|
||||
if tool.inputSchema:
|
||||
if tool.input_schema:
|
||||
# MCP tools use JSON Schema for parameters
|
||||
properties = tool.inputSchema.get('properties', {})
|
||||
required = set(tool.inputSchema.get('required', []))
|
||||
properties = tool.input_schema.get('properties', {})
|
||||
required = set(tool.input_schema.get('required', []))
|
||||
|
||||
for param_name, param_schema in properties.items():
|
||||
# Convert JSON Schema type to Python type
|
||||
@@ -326,7 +326,7 @@ class MCPClient:
|
||||
# Convert MCP result to ActionResult
|
||||
extracted_content = self._format_mcp_result(result)
|
||||
|
||||
if getattr(result, 'isError', False):
|
||||
if result.is_error:
|
||||
error_msg = f"MCP tool '{tool.name}' reported an error: {extracted_content}"
|
||||
return ActionResult(error=error_msg, success=False)
|
||||
|
||||
@@ -374,7 +374,7 @@ class MCPClient:
|
||||
# Convert MCP result to ActionResult
|
||||
extracted_content = self._format_mcp_result(result)
|
||||
|
||||
if getattr(result, 'isError', False):
|
||||
if result.is_error:
|
||||
error_msg = f"MCP tool '{tool.name}' reported an error: {extracted_content}"
|
||||
return ActionResult(error=error_msg, success=False)
|
||||
|
||||
|
||||
@@ -101,10 +101,10 @@ class MCPToolWrapper:
|
||||
# Parse tool parameters to create Pydantic model
|
||||
param_fields = {}
|
||||
|
||||
if tool.inputSchema:
|
||||
if tool.input_schema:
|
||||
# MCP tools use JSON Schema for parameters
|
||||
properties = tool.inputSchema.get('properties', {})
|
||||
required = set(tool.inputSchema.get('required', []))
|
||||
properties = tool.input_schema.get('properties', {})
|
||||
required = set(tool.input_schema.get('required', []))
|
||||
|
||||
for param_name, param_schema in properties.items():
|
||||
# Convert JSON Schema type to Python type
|
||||
|
||||
+253
-253
@@ -133,8 +133,7 @@ _ensure_all_loggers_use_stderr()
|
||||
try:
|
||||
import mcp.server.stdio
|
||||
import mcp.types as types
|
||||
from mcp.server import NotificationOptions, Server
|
||||
from mcp.server.models import InitializationOptions
|
||||
from mcp.server import Server
|
||||
|
||||
MCP_AVAILABLE = True
|
||||
|
||||
@@ -191,7 +190,7 @@ class BrowserUseServer:
|
||||
# Ensure all logging goes to stderr (in case new loggers were created)
|
||||
_ensure_all_loggers_use_stderr()
|
||||
|
||||
self.server = Server('browser-use')
|
||||
self.server = Server('browser-use', version=get_browser_use_version())
|
||||
self.config = load_browser_use_config()
|
||||
self.agent: Agent | None = None
|
||||
self.browser_session: BrowserSession | None = None
|
||||
@@ -212,271 +211,276 @@ class BrowserUseServer:
|
||||
def _setup_handlers(self):
|
||||
"""Setup MCP server handlers."""
|
||||
|
||||
@self.server.list_tools()
|
||||
async def handle_list_tools() -> list[types.Tool]:
|
||||
async def handle_list_tools(_context: Any, _params: types.PaginatedRequestParams) -> types.ListToolsResult:
|
||||
"""List all available browser-use tools."""
|
||||
return [
|
||||
# Agent tools
|
||||
# Direct browser control tools
|
||||
types.Tool(
|
||||
name='browser_navigate',
|
||||
description='Navigate to a URL in the browser',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'url': {'type': 'string', 'description': 'The URL to navigate to'},
|
||||
'new_tab': {'type': 'boolean', 'description': 'Whether to open in a new tab', 'default': False},
|
||||
return types.ListToolsResult(
|
||||
tools=[
|
||||
# Agent tools
|
||||
# Direct browser control tools
|
||||
types.Tool(
|
||||
name='browser_navigate',
|
||||
description='Navigate to a URL in the browser',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'url': {'type': 'string', 'description': 'The URL to navigate to'},
|
||||
'new_tab': {'type': 'boolean', 'description': 'Whether to open in a new tab', 'default': False},
|
||||
},
|
||||
'required': ['url'],
|
||||
},
|
||||
'required': ['url'],
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_click',
|
||||
description='Click an element by index or at specific viewport coordinates. Use index for elements from browser_get_state, or coordinate_x/coordinate_y for pixel-precise clicking.',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'index': {
|
||||
'type': 'integer',
|
||||
'description': 'The index of the element to click (from browser_get_state). Provide this OR coordinate_x+coordinate_y.',
|
||||
},
|
||||
'coordinate_x': {
|
||||
'type': 'integer',
|
||||
'description': 'X coordinate in pixels from the left edge of the viewport. Must be used together with coordinate_y. Provide this OR index.',
|
||||
},
|
||||
'coordinate_y': {
|
||||
'type': 'integer',
|
||||
'description': 'Y coordinate in pixels from the top edge of the viewport. Must be used together with coordinate_x. Provide this OR index.',
|
||||
},
|
||||
'new_tab': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to open any resulting navigation in a new tab',
|
||||
'default': False,
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_click',
|
||||
description='Click an element by index or at specific viewport coordinates. Use index for elements from browser_get_state, or coordinate_x/coordinate_y for pixel-precise clicking.',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'index': {
|
||||
'type': 'integer',
|
||||
'description': 'The index of the element to click (from browser_get_state). Provide this OR coordinate_x+coordinate_y.',
|
||||
},
|
||||
'coordinate_x': {
|
||||
'type': 'integer',
|
||||
'description': 'X coordinate in pixels from the left edge of the viewport. Must be used together with coordinate_y. Provide this OR index.',
|
||||
},
|
||||
'coordinate_y': {
|
||||
'type': 'integer',
|
||||
'description': 'Y coordinate in pixels from the top edge of the viewport. Must be used together with coordinate_x. Provide this OR index.',
|
||||
},
|
||||
'new_tab': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to open any resulting navigation in a new tab',
|
||||
'default': False,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_type',
|
||||
description='Type text into an input field. Clears existing text by default; pass text="" to clear only.',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'index': {
|
||||
'type': 'integer',
|
||||
'description': 'The index of the input element (from browser_get_state)',
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_type',
|
||||
description='Type text into an input field. Clears existing text by default; pass text="" to clear only.',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'index': {
|
||||
'type': 'integer',
|
||||
'description': 'The index of the input element (from browser_get_state)',
|
||||
},
|
||||
'text': {
|
||||
'type': 'string',
|
||||
'description': 'The text to type. Pass an empty string ("") to clear the field without typing.',
|
||||
},
|
||||
},
|
||||
'text': {
|
||||
'type': 'string',
|
||||
'description': 'The text to type. Pass an empty string ("") to clear the field without typing.',
|
||||
'required': ['index', 'text'],
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_get_state',
|
||||
description='Get the current state of the page including all interactive elements',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'include_screenshot': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to include a screenshot of the current page',
|
||||
'default': False,
|
||||
}
|
||||
},
|
||||
},
|
||||
'required': ['index', 'text'],
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_get_state',
|
||||
description='Get the current state of the page including all interactive elements',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'include_screenshot': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to include a screenshot of the current page',
|
||||
'default': False,
|
||||
}
|
||||
annotations=types.ToolAnnotations(read_only_hint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_extract_content',
|
||||
description='Extract structured content from the current page based on a query',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'query': {'type': 'string', 'description': 'What information to extract from the page'},
|
||||
'extract_links': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to include links in the extraction',
|
||||
'default': False,
|
||||
},
|
||||
},
|
||||
'required': ['query'],
|
||||
},
|
||||
},
|
||||
annotations=types.ToolAnnotations(readOnlyHint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_extract_content',
|
||||
description='Extract structured content from the current page based on a query',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'query': {'type': 'string', 'description': 'What information to extract from the page'},
|
||||
'extract_links': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to include links in the extraction',
|
||||
'default': False,
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_get_html',
|
||||
description='Get the raw HTML of the current page or a specific element by CSS selector',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'selector': {
|
||||
'type': 'string',
|
||||
'description': 'Optional CSS selector to get HTML of a specific element. If omitted, returns full page HTML.',
|
||||
},
|
||||
},
|
||||
},
|
||||
'required': ['query'],
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_get_html',
|
||||
description='Get the raw HTML of the current page or a specific element by CSS selector',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'selector': {
|
||||
'type': 'string',
|
||||
'description': 'Optional CSS selector to get HTML of a specific element. If omitted, returns full page HTML.',
|
||||
annotations=types.ToolAnnotations(read_only_hint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_screenshot',
|
||||
description='Take a screenshot of the current page. Returns viewport metadata as text and the screenshot as an image.',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'full_page': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to capture the full scrollable page or just the visible viewport',
|
||||
'default': False,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
annotations=types.ToolAnnotations(readOnlyHint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_screenshot',
|
||||
description='Take a screenshot of the current page. Returns viewport metadata as text and the screenshot as an image.',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'full_page': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to capture the full scrollable page or just the visible viewport',
|
||||
'default': False,
|
||||
annotations=types.ToolAnnotations(read_only_hint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_scroll',
|
||||
description='Scroll the page',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'direction': {
|
||||
'type': 'string',
|
||||
'enum': ['up', 'down'],
|
||||
'description': 'Direction to scroll',
|
||||
'default': 'down',
|
||||
}
|
||||
},
|
||||
},
|
||||
},
|
||||
annotations=types.ToolAnnotations(readOnlyHint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_scroll',
|
||||
description='Scroll the page',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'direction': {
|
||||
'type': 'string',
|
||||
'enum': ['up', 'down'],
|
||||
'description': 'Direction to scroll',
|
||||
'default': 'down',
|
||||
}
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_go_back',
|
||||
description='Go back to the previous page',
|
||||
input_schema={'type': 'object', 'properties': {}},
|
||||
),
|
||||
# Tab management
|
||||
types.Tool(
|
||||
name='browser_list_tabs',
|
||||
description='List all open tabs',
|
||||
input_schema={'type': 'object', 'properties': {}},
|
||||
annotations=types.ToolAnnotations(read_only_hint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_switch_tab',
|
||||
description='Switch to a different tab',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'tab_id': {'type': 'string', 'description': '4 Character Tab ID of the tab to switch to'}
|
||||
},
|
||||
'required': ['tab_id'],
|
||||
},
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_go_back',
|
||||
description='Go back to the previous page',
|
||||
inputSchema={'type': 'object', 'properties': {}},
|
||||
),
|
||||
# Tab management
|
||||
types.Tool(
|
||||
name='browser_list_tabs',
|
||||
description='List all open tabs',
|
||||
inputSchema={'type': 'object', 'properties': {}},
|
||||
annotations=types.ToolAnnotations(readOnlyHint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_switch_tab',
|
||||
description='Switch to a different tab',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {'tab_id': {'type': 'string', 'description': '4 Character Tab ID of the tab to switch to'}},
|
||||
'required': ['tab_id'],
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_close_tab',
|
||||
description='Close a tab',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {'tab_id': {'type': 'string', 'description': '4 Character Tab ID of the tab to close'}},
|
||||
'required': ['tab_id'],
|
||||
},
|
||||
),
|
||||
# types.Tool(
|
||||
# name="browser_close",
|
||||
# description="Close the browser session",
|
||||
# inputSchema={
|
||||
# "type": "object",
|
||||
# "properties": {}
|
||||
# }
|
||||
# ),
|
||||
types.Tool(
|
||||
name='retry_with_browser_use_agent',
|
||||
description='Retry a task using the browser-use agent. Only use this as a last resort if you fail to interact with a page multiple times.',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'task': {
|
||||
'type': 'string',
|
||||
'description': 'The high-level goal and detailed step-by-step description of the task the AI browser agent needs to attempt, along with any relevant data needed to complete the task and info about previous attempts.',
|
||||
},
|
||||
'max_steps': {
|
||||
'type': 'integer',
|
||||
'description': 'Maximum number of steps an agent can take.',
|
||||
'default': 100,
|
||||
},
|
||||
'model': {
|
||||
'type': 'string',
|
||||
'description': 'LLM model to use (e.g., gpt-4o, claude-3-opus-20240229). Defaults to the configured model.',
|
||||
},
|
||||
'allowed_domains': {
|
||||
'type': 'array',
|
||||
'items': {'type': 'string'},
|
||||
'description': (
|
||||
'List of domains the agent is allowed to visit (security feature). '
|
||||
'Omit to use the server-configured profile defaults. '
|
||||
'An empty list is treated the same as omitting the argument and '
|
||||
'will NOT disable server-configured restrictions.'
|
||||
),
|
||||
},
|
||||
'use_vision': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to use vision capabilities (screenshots) for the agent',
|
||||
'default': True,
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_close_tab',
|
||||
description='Close a tab',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {'tab_id': {'type': 'string', 'description': '4 Character Tab ID of the tab to close'}},
|
||||
'required': ['tab_id'],
|
||||
},
|
||||
'required': ['task'],
|
||||
},
|
||||
),
|
||||
# Browser session management tools
|
||||
types.Tool(
|
||||
name='browser_list_sessions',
|
||||
description='List all active browser sessions with their details and last activity time',
|
||||
inputSchema={'type': 'object', 'properties': {}},
|
||||
annotations=types.ToolAnnotations(readOnlyHint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_close_session',
|
||||
description='Close a specific browser session by its ID',
|
||||
inputSchema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'session_id': {
|
||||
'type': 'string',
|
||||
'description': 'The browser session ID to close (get from browser_list_sessions)',
|
||||
}
|
||||
),
|
||||
# types.Tool(
|
||||
# name="browser_close",
|
||||
# description="Close the browser session",
|
||||
# input_schema={
|
||||
# "type": "object",
|
||||
# "properties": {}
|
||||
# }
|
||||
# ),
|
||||
types.Tool(
|
||||
name='retry_with_browser_use_agent',
|
||||
description='Retry a task using the browser-use agent. Only use this as a last resort if you fail to interact with a page multiple times.',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'task': {
|
||||
'type': 'string',
|
||||
'description': 'The high-level goal and detailed step-by-step description of the task the AI browser agent needs to attempt, along with any relevant data needed to complete the task and info about previous attempts.',
|
||||
},
|
||||
'max_steps': {
|
||||
'type': 'integer',
|
||||
'description': 'Maximum number of steps an agent can take.',
|
||||
'default': 100,
|
||||
},
|
||||
'model': {
|
||||
'type': 'string',
|
||||
'description': 'LLM model to use (e.g., gpt-4o, claude-3-opus-20240229). Defaults to the configured model.',
|
||||
},
|
||||
'allowed_domains': {
|
||||
'type': 'array',
|
||||
'items': {'type': 'string'},
|
||||
'description': (
|
||||
'List of domains the agent is allowed to visit (security feature). '
|
||||
'Omit to use the server-configured profile defaults. '
|
||||
'An empty list is treated the same as omitting the argument and '
|
||||
'will NOT disable server-configured restrictions.'
|
||||
),
|
||||
},
|
||||
'use_vision': {
|
||||
'type': 'boolean',
|
||||
'description': 'Whether to use vision capabilities (screenshots) for the agent',
|
||||
'default': True,
|
||||
},
|
||||
},
|
||||
'required': ['task'],
|
||||
},
|
||||
'required': ['session_id'],
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_close_all',
|
||||
description='Close all active browser sessions and clean up resources',
|
||||
inputSchema={'type': 'object', 'properties': {}},
|
||||
),
|
||||
]
|
||||
),
|
||||
# Browser session management tools
|
||||
types.Tool(
|
||||
name='browser_list_sessions',
|
||||
description='List all active browser sessions with their details and last activity time',
|
||||
input_schema={'type': 'object', 'properties': {}},
|
||||
annotations=types.ToolAnnotations(read_only_hint=True),
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_close_session',
|
||||
description='Close a specific browser session by its ID',
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {
|
||||
'session_id': {
|
||||
'type': 'string',
|
||||
'description': 'The browser session ID to close (get from browser_list_sessions)',
|
||||
}
|
||||
},
|
||||
'required': ['session_id'],
|
||||
},
|
||||
),
|
||||
types.Tool(
|
||||
name='browser_close_all',
|
||||
description='Close all active browser sessions and clean up resources',
|
||||
input_schema={'type': 'object', 'properties': {}},
|
||||
),
|
||||
]
|
||||
)
|
||||
|
||||
@self.server.list_resources()
|
||||
async def handle_list_resources() -> list[types.Resource]:
|
||||
async def handle_list_resources(_context: Any, _params: types.PaginatedRequestParams) -> types.ListResourcesResult:
|
||||
"""List available resources (none for browser-use)."""
|
||||
return []
|
||||
return types.ListResourcesResult(resources=[])
|
||||
|
||||
@self.server.list_prompts()
|
||||
async def handle_list_prompts() -> list[types.Prompt]:
|
||||
async def handle_list_prompts(_context: Any, _params: types.PaginatedRequestParams) -> types.ListPromptsResult:
|
||||
"""List available prompts (none for browser-use)."""
|
||||
return []
|
||||
return types.ListPromptsResult(prompts=[])
|
||||
|
||||
@self.server.call_tool()
|
||||
async def handle_call_tool(name: str, arguments: dict[str, Any] | None) -> list[types.TextContent | types.ImageContent]:
|
||||
async def handle_call_tool(_context: Any, params: types.CallToolRequestParams) -> types.CallToolResult:
|
||||
"""Handle tool execution."""
|
||||
name = params.name
|
||||
arguments = params.arguments
|
||||
start_time = time.time()
|
||||
error_msg = None
|
||||
try:
|
||||
result = await self._execute_tool(name, arguments or {})
|
||||
if isinstance(result, list):
|
||||
return result
|
||||
return [types.TextContent(type='text', text=result)]
|
||||
return types.CallToolResult(content=result)
|
||||
return types.CallToolResult(content=[types.TextContent(type='text', text=result)])
|
||||
except Exception as e:
|
||||
error_msg = str(e)
|
||||
logger.error(f'Tool execution failed: {e}', exc_info=True)
|
||||
return [types.TextContent(type='text', text=f'Error: {str(e)}')]
|
||||
return types.CallToolResult(
|
||||
content=[types.TextContent(type='text', text=f'Error: {str(e)}')],
|
||||
is_error=True,
|
||||
)
|
||||
finally:
|
||||
# Capture telemetry for tool calls
|
||||
duration = time.time() - start_time
|
||||
@@ -490,9 +494,12 @@ class BrowserUseServer:
|
||||
)
|
||||
)
|
||||
|
||||
async def _execute_tool(
|
||||
self, tool_name: str, arguments: dict[str, Any]
|
||||
) -> str | list[types.TextContent | types.ImageContent]:
|
||||
self.server.add_request_handler('tools/list', types.PaginatedRequestParams, handle_list_tools)
|
||||
self.server.add_request_handler('resources/list', types.PaginatedRequestParams, handle_list_resources)
|
||||
self.server.add_request_handler('prompts/list', types.PaginatedRequestParams, handle_list_prompts)
|
||||
self.server.add_request_handler('tools/call', types.CallToolRequestParams, handle_call_tool)
|
||||
|
||||
async def _execute_tool(self, tool_name: str, arguments: dict[str, Any]) -> str | list[types.ContentBlock]:
|
||||
"""Execute a browser-use tool. Returns str for most tools, or a content list for tools with image output."""
|
||||
|
||||
# Agent-based tools
|
||||
@@ -537,9 +544,9 @@ class BrowserUseServer:
|
||||
|
||||
elif tool_name == 'browser_get_state':
|
||||
state_json, screenshot_b64 = await self._get_browser_state(arguments.get('include_screenshot', False))
|
||||
content: list[types.TextContent | types.ImageContent] = [types.TextContent(type='text', text=state_json)]
|
||||
content: list[types.ContentBlock] = [types.TextContent(type='text', text=state_json)]
|
||||
if screenshot_b64:
|
||||
content.append(types.ImageContent(type='image', data=screenshot_b64, mimeType='image/png'))
|
||||
content.append(types.ImageContent(type='image', data=screenshot_b64, mime_type='image/png'))
|
||||
return content
|
||||
|
||||
elif tool_name == 'browser_get_html':
|
||||
@@ -547,9 +554,9 @@ class BrowserUseServer:
|
||||
|
||||
elif tool_name == 'browser_screenshot':
|
||||
meta_json, screenshot_b64 = await self._screenshot(arguments.get('full_page', False))
|
||||
content: list[types.TextContent | types.ImageContent] = [types.TextContent(type='text', text=meta_json)]
|
||||
content: list[types.ContentBlock] = [types.TextContent(type='text', text=meta_json)]
|
||||
if screenshot_b64:
|
||||
content.append(types.ImageContent(type='image', data=screenshot_b64, mimeType='image/png'))
|
||||
content.append(types.ImageContent(type='image', data=screenshot_b64, mime_type='image/png'))
|
||||
return content
|
||||
|
||||
elif tool_name == 'browser_extract_content':
|
||||
@@ -573,7 +580,7 @@ class BrowserUseServer:
|
||||
elif tool_name == 'browser_close_tab':
|
||||
return await self._close_tab(arguments['tab_id'])
|
||||
|
||||
return f'Unknown tool: {tool_name}'
|
||||
raise ValueError(f'Unknown tool: {tool_name}')
|
||||
|
||||
async def _init_browser_session(self, allowed_domains: list[str] | None = None, **kwargs):
|
||||
"""Initialize browser session using config"""
|
||||
@@ -1248,14 +1255,7 @@ class BrowserUseServer:
|
||||
await self.server.run(
|
||||
read_stream,
|
||||
write_stream,
|
||||
InitializationOptions(
|
||||
server_name='browser-use',
|
||||
server_version='0.1.0',
|
||||
capabilities=self.server.get_capabilities(
|
||||
notification_options=NotificationOptions(),
|
||||
experimental_capabilities={},
|
||||
),
|
||||
),
|
||||
self.server.create_initialization_options(),
|
||||
)
|
||||
except BrokenPipeError:
|
||||
logger.warning('MCP client disconnected while writing to stdio; shutting down server cleanly.')
|
||||
|
||||
@@ -54,6 +54,8 @@ PY
|
||||
changing Chrome's visible tab. Screenshots and normal CDP input work in the
|
||||
background; call `activate_tab(target)` only when the user explicitly asks
|
||||
or a page demonstrably pauses rendering while hidden.
|
||||
- Set `BH_TAB_MARKER=0` before starting the daemon to leave page titles unchanged.
|
||||
The horse marker remains enabled by default.
|
||||
- A timed-out `scroll(...)` on an attached background tab is evidence that the
|
||||
page needs to be visible. Call `activate_tab(current_tab())`, retry the same
|
||||
scroll once, then re-read the scroll position. This visibly switches tabs,
|
||||
@@ -77,14 +79,22 @@ If Chrome is running but remote debugging is not enabled, the harness opens:
|
||||
chrome://inspect/#remote-debugging
|
||||
```
|
||||
|
||||
On macOS, when Chrome asks for remote-debugging permission, run:
|
||||
On macOS, when local Chrome asks for remote-debugging permission, keep the
|
||||
original browser command running and call `mac-approve` in another shell/tool
|
||||
call. Preserve the exact daemon name: if the waiting command used
|
||||
`BU_NAME=r7k2`, run:
|
||||
|
||||
```text
|
||||
browser-use mac-approve
|
||||
BU_NAME=r7k2 browser-use mac-approve
|
||||
```
|
||||
|
||||
Continue browser work when it returns `ready`; otherwise follow its printed
|
||||
instruction.
|
||||
For the default daemon, omit the `BU_NAME` prefix. The original command resumes
|
||||
when the helper returns `ready`; do not rerun it. If the helper reports
|
||||
`accessibility-required`, ask the user once to grant the app launching
|
||||
browser-use (for example Terminal, iTerm, or Codex) access in System
|
||||
Settings > Privacy & Security > Accessibility, then call `mac-approve` once
|
||||
again. This is only for local Chrome; do not call it for `BU_CDP_URL`,
|
||||
`BU_CDP_WS`, or Browser Use Cloud.
|
||||
|
||||
## Remote Browsers
|
||||
|
||||
@@ -136,6 +146,7 @@ Cloud profile cookie sync reference: https://github.com/browser-use/browser-harn
|
||||
- After navigation, call `wait_for_load()`.
|
||||
- If the current tab is stale or internal, call `ensure_real_tab()`.
|
||||
- Use `js(...)` for DOM inspection or extraction when coordinates are the wrong tool.
|
||||
- When entering unusually long text, avoid slow per-character typing: find a faster page-appropriate input method, then verify the page kept the exact value.
|
||||
- Login walls: stop and ask. Exception: use available SSO automatically when Chrome is already signed in; still stop for passwords, MFA, consent, or ambiguous account choice.
|
||||
- Raw CDP is available with `cdp("Domain.method", ...)`.
|
||||
|
||||
@@ -207,7 +218,7 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro
|
||||
## Gotchas
|
||||
|
||||
- `chrome://inspect/#remote-debugging` must be enabled for local Chrome control.
|
||||
- On macOS, if Chrome shows an "Allow remote debugging?" popup, run `browser-use mac-approve`. Do not poll in a loop — the daemon holds one connection.
|
||||
- On macOS, if local Chrome shows an "Allow remote debugging?" popup, call `mac-approve` once with the same `BU_NAME` while the original browser command waits. Do not poll or rerun the browser command; remote and cloud browsers do not use this helper.
|
||||
- Omnibox popups are not real work tabs.
|
||||
- CDP target order is not Chrome's visible tab-strip order.
|
||||
- `BU_CDP_URL` is an HTTP DevTools endpoint; the daemon resolves it to WebSocket.
|
||||
|
||||
@@ -1013,19 +1013,27 @@ class Tools(Generic[Context]):
|
||||
|
||||
event = browser_session.event_bus.dispatch(SwitchTabEvent(target_id=target_id))
|
||||
await event
|
||||
new_target_id = await event.event_result(raise_if_any=False, raise_if_none=False) # Don't raise on errors
|
||||
|
||||
if new_target_id:
|
||||
memory = f'Switched to tab #{new_target_id[-4:]}'
|
||||
else:
|
||||
memory = f'Switched to tab #{params.tab_id}'
|
||||
|
||||
logger.info(f'🔄 {memory}')
|
||||
return ActionResult(extracted_content=memory, long_term_memory=memory)
|
||||
# raise_if_any=True so a handler failure surfaces its real cause here instead
|
||||
# of silently becoming a "produced no result" below.
|
||||
new_target_id = await event.event_result(raise_if_any=True, raise_if_none=False)
|
||||
except Exception as e:
|
||||
logger.warning(f'Tab switch may have failed: {e}')
|
||||
memory = f'Attempted to switch to tab #{params.tab_id}'
|
||||
return ActionResult(extracted_content=memory, long_term_memory=memory)
|
||||
logger.warning(f'Tab switch failed: {e}')
|
||||
# Preserve the concrete cause (e.g. a stale tab_id) instead of a generic
|
||||
# message, so the agent gets actionable failure info in both memories.
|
||||
memory = f'Failed to switch to tab #{params.tab_id}: {e}'
|
||||
raise BrowserError(memory, short_term_memory=memory, long_term_memory=memory)
|
||||
|
||||
# on_SwitchTabEvent returns the newly focused TargetID on every success path, so a
|
||||
# missing result (with no exception raised) means the handler still failed rather
|
||||
# than having quietly succeeded.
|
||||
if not new_target_id:
|
||||
memory = f'Failed to switch to tab #{params.tab_id}: tab switch produced no result'
|
||||
logger.warning(memory)
|
||||
raise BrowserError(memory, short_term_memory=memory, long_term_memory=memory)
|
||||
|
||||
memory = f'Switched to tab #{new_target_id[-4:]}'
|
||||
logger.info(f'🔄 {memory}')
|
||||
return ActionResult(extracted_content=memory, long_term_memory=memory)
|
||||
|
||||
@self.registry.action(
|
||||
'Close a tab by tab_id. Tab IDs are shown in browser state tabs list (last 4 chars of target_id). Use to clean up tabs you no longer need.',
|
||||
@@ -1747,7 +1755,9 @@ You will be given a query and the markdown of a webpage that has been filtered t
|
||||
'Write content to a file. By default this OVERWRITES the entire file - use append=true to add to an existing file, or use replace_file for targeted edits within a file. '
|
||||
'FILENAME RULES: Use only letters, numbers, underscores, hyphens, dots, parentheses. Spaces are auto-converted to hyphens. '
|
||||
'SUPPORTED EXTENSIONS: .txt, .md, .json, .jsonl, .csv, .html, .xml, .pdf, .docx. '
|
||||
'CANNOT write binary/image files (.png, .jpg, .mp4, etc.) - do not attempt to save screenshots as files. '
|
||||
'For small images (.png, .gif, .jpg, .jpeg, .webp) — e.g. a tiny file to upload — set content to the '
|
||||
'base64 of a valid image (a 1x1 PNG is ~92 base64 chars); the bytes are decoded and written for you. '
|
||||
'CANNOT write other binary files (.mp4, .zip, etc.) and do not attempt to save screenshots as files. '
|
||||
'For PDF files, write content in markdown format and it will be auto-converted to PDF.'
|
||||
)
|
||||
async def write_file(
|
||||
|
||||
+9
-8
@@ -2,7 +2,7 @@
|
||||
name = "browser-use"
|
||||
description = "Make websites accessible for AI agents"
|
||||
authors = [{ name = "Gregor Zunic" }]
|
||||
version = "0.13.8"
|
||||
version = "0.13.10"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.11,<4.0"
|
||||
classifiers = [
|
||||
@@ -21,7 +21,8 @@ dependencies = [
|
||||
"httpx==0.28.1",
|
||||
"posthog==7.7.0",
|
||||
"psutil==7.2.2",
|
||||
"pydantic>=2.12.5,<2.14",
|
||||
"pydantic==2.13.5",
|
||||
"pydantic-settings==2.15.0",
|
||||
"pyobjc==12.1; platform_system == 'darwin'",
|
||||
"python-dotenv==1.2.2",
|
||||
"requests==2.33.0",
|
||||
@@ -36,8 +37,8 @@ dependencies = [
|
||||
"google-api-python-client==2.188.0",
|
||||
"google-auth==2.48.0",
|
||||
"google-auth-oauthlib==1.2.4",
|
||||
"mcp==1.28.1",
|
||||
"pypdf==6.15.0",
|
||||
"mcp==2.1.1",
|
||||
"pypdf==6.16.2",
|
||||
"reportlab==4.4.9",
|
||||
"cdp-use==1.4.5",
|
||||
"pyotp==2.9.0",
|
||||
@@ -46,7 +47,7 @@ dependencies = [
|
||||
"markdownify==1.2.2",
|
||||
"python-docx==1.2.0",
|
||||
"browser-use-sdk==3.4.2",
|
||||
"browser-harness==0.1.10",
|
||||
"browser-harness==0.1.13",
|
||||
]
|
||||
# google-api-core: only used for Google LLM APIs
|
||||
# pyperclip: only used for examples that use copy/paste
|
||||
@@ -108,12 +109,12 @@ browser = "browser_use.cli:main" # Alias for browser-use
|
||||
browser-use-tui = "browser_use.cli:browser_use_tui_main" # Deprecated alias for browser-use
|
||||
|
||||
[build-system]
|
||||
requires = ["hatchling"]
|
||||
requires = ["hatchling==1.32.0"]
|
||||
build-backend = "hatchling.build"
|
||||
|
||||
|
||||
[tool.codespell]
|
||||
ignore-words-list = "bu,wit,dont,cant,wont,re-use,re-used,re-using,re-usable,thats,doesnt,doubleclick,finaly,finalY"
|
||||
ignore-words-list = "bu,wit,dont,cant,wont,re-use,re-used,re-using,re-usable,thats,doesnt,doubleclick,finaly,finalY,iterm"
|
||||
skip = "*.json"
|
||||
|
||||
[tool.ruff]
|
||||
@@ -242,5 +243,5 @@ dev-dependencies = [
|
||||
"lmnr[all]==0.7.42",
|
||||
# "pytest-playwright-asyncio>=0.7.0", # not actually needed I think
|
||||
"pytest-timeout==2.4.0",
|
||||
"pydantic_settings==2.12.0",
|
||||
"pydantic_settings==2.15.0",
|
||||
]
|
||||
|
||||
@@ -54,6 +54,8 @@ PY
|
||||
changing Chrome's visible tab. Screenshots and normal CDP input work in the
|
||||
background; call `activate_tab(target)` only when the user explicitly asks
|
||||
or a page demonstrably pauses rendering while hidden.
|
||||
- Set `BH_TAB_MARKER=0` before starting the daemon to leave page titles unchanged.
|
||||
The horse marker remains enabled by default.
|
||||
- A timed-out `scroll(...)` on an attached background tab is evidence that the
|
||||
page needs to be visible. Call `activate_tab(current_tab())`, retry the same
|
||||
scroll once, then re-read the scroll position. This visibly switches tabs,
|
||||
@@ -77,14 +79,22 @@ If Chrome is running but remote debugging is not enabled, the harness opens:
|
||||
chrome://inspect/#remote-debugging
|
||||
```
|
||||
|
||||
On macOS, when Chrome asks for remote-debugging permission, run:
|
||||
On macOS, when local Chrome asks for remote-debugging permission, keep the
|
||||
original browser command running and call `mac-approve` in another shell/tool
|
||||
call. Preserve the exact daemon name: if the waiting command used
|
||||
`BU_NAME=r7k2`, run:
|
||||
|
||||
```text
|
||||
browser-use mac-approve
|
||||
BU_NAME=r7k2 browser-use mac-approve
|
||||
```
|
||||
|
||||
Continue browser work when it returns `ready`; otherwise follow its printed
|
||||
instruction.
|
||||
For the default daemon, omit the `BU_NAME` prefix. The original command resumes
|
||||
when the helper returns `ready`; do not rerun it. If the helper reports
|
||||
`accessibility-required`, ask the user once to grant the app launching
|
||||
browser-use (for example Terminal, iTerm, or Codex) access in System
|
||||
Settings > Privacy & Security > Accessibility, then call `mac-approve` once
|
||||
again. This is only for local Chrome; do not call it for `BU_CDP_URL`,
|
||||
`BU_CDP_WS`, or Browser Use Cloud.
|
||||
|
||||
## Remote Browsers
|
||||
|
||||
@@ -136,6 +146,7 @@ Cloud profile cookie sync reference: https://github.com/browser-use/browser-harn
|
||||
- After navigation, call `wait_for_load()`.
|
||||
- If the current tab is stale or internal, call `ensure_real_tab()`.
|
||||
- Use `js(...)` for DOM inspection or extraction when coordinates are the wrong tool.
|
||||
- When entering unusually long text, avoid slow per-character typing: find a faster page-appropriate input method, then verify the page kept the exact value.
|
||||
- Login walls: stop and ask. Exception: use available SSO automatically when Chrome is already signed in; still stop for passwords, MFA, consent, or ambiguous account choice.
|
||||
- Raw CDP is available with `cdp("Domain.method", ...)`.
|
||||
|
||||
@@ -207,7 +218,7 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro
|
||||
## Gotchas
|
||||
|
||||
- `chrome://inspect/#remote-debugging` must be enabled for local Chrome control.
|
||||
- On macOS, if Chrome shows an "Allow remote debugging?" popup, run `browser-use mac-approve`. Do not poll in a loop — the daemon holds one connection.
|
||||
- On macOS, if local Chrome shows an "Allow remote debugging?" popup, call `mac-approve` once with the same `BU_NAME` while the original browser command waits. Do not poll or rerun the browser command; remote and cloud browsers do not use this helper.
|
||||
- Omnibox popups are not real work tabs.
|
||||
- CDP target order is not Chrome's visible tab-strip order.
|
||||
- `BU_CDP_URL` is an HTTP DevTools endpoint; the daemon resolves it to WebSocket.
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
import asyncio
|
||||
from types import SimpleNamespace
|
||||
from typing import cast
|
||||
|
||||
from browser_use.browser.events import SendKeysEvent
|
||||
from browser_use.browser.watchdogs.default_action_watchdog import DefaultActionWatchdog
|
||||
|
||||
|
||||
def make_watchdog(recorded_params, dispatched_keys) -> DefaultActionWatchdog:
|
||||
class Input:
|
||||
async def dispatchKeyEvent(self, params=None, session_id=None):
|
||||
recorded_params.append(params or {})
|
||||
|
||||
cdp_session = SimpleNamespace(
|
||||
cdp_client=SimpleNamespace(send=SimpleNamespace(Input=Input())),
|
||||
session_id='session-1',
|
||||
)
|
||||
|
||||
class BrowserSession:
|
||||
async def get_or_create_cdp_session(self, focus=False):
|
||||
return cdp_session
|
||||
|
||||
async def dispatch_key_event(_session, event_type, key, modifiers=0):
|
||||
dispatched_keys.append((event_type, key, modifiers))
|
||||
|
||||
watchdog = SimpleNamespace(
|
||||
browser_session=BrowserSession(),
|
||||
logger=SimpleNamespace(info=lambda *args, **kwargs: None),
|
||||
_dispatch_key_event=dispatch_key_event,
|
||||
)
|
||||
watchdog._get_char_modifiers_and_vk = DefaultActionWatchdog._get_char_modifiers_and_vk.__get__(watchdog)
|
||||
watchdog._get_key_code_for_char = DefaultActionWatchdog._get_key_code_for_char.__get__(watchdog)
|
||||
return cast(DefaultActionWatchdog, watchdog)
|
||||
|
||||
|
||||
def test_send_keys_literal_plus_dispatches_char_event():
|
||||
recorded_params = []
|
||||
dispatched_keys = []
|
||||
watchdog = make_watchdog(recorded_params, dispatched_keys)
|
||||
|
||||
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='+')))
|
||||
|
||||
char_events = [params for params in recorded_params if params.get('type') == 'char']
|
||||
assert [(params.get('text'), params.get('key')) for params in char_events] == [('+', '+')]
|
||||
assert all(params.get('key') for params in recorded_params)
|
||||
|
||||
|
||||
def test_send_keys_text_with_plus_dispatches_all_characters():
|
||||
recorded_params = []
|
||||
dispatched_keys = []
|
||||
watchdog = make_watchdog(recorded_params, dispatched_keys)
|
||||
|
||||
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='C++')))
|
||||
|
||||
assert [params['text'] for params in recorded_params if params.get('type') == 'char'] == ['C', '+', '+']
|
||||
|
||||
|
||||
def test_send_keys_control_plus_keeps_plus_as_main_key():
|
||||
recorded_params = []
|
||||
dispatched_keys = []
|
||||
watchdog = make_watchdog(recorded_params, dispatched_keys)
|
||||
|
||||
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='Control++')))
|
||||
|
||||
assert dispatched_keys == [
|
||||
('keyDown', 'Control', 0),
|
||||
('keyDown', '+', 2),
|
||||
('keyUp', '+', 2),
|
||||
('keyUp', 'Control', 0),
|
||||
]
|
||||
|
||||
|
||||
def test_send_keys_existing_shortcut_and_special_key_still_work():
|
||||
recorded_params = []
|
||||
dispatched_keys = []
|
||||
watchdog = make_watchdog(recorded_params, dispatched_keys)
|
||||
|
||||
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='Control+a')))
|
||||
assert dispatched_keys == [
|
||||
('keyDown', 'Control', 0),
|
||||
('keyDown', 'a', 2),
|
||||
('keyUp', 'a', 2),
|
||||
('keyUp', 'Control', 0),
|
||||
]
|
||||
|
||||
recorded_params.clear()
|
||||
dispatched_keys.clear()
|
||||
asyncio.run(DefaultActionWatchdog.on_SendKeysEvent(watchdog, SendKeysEvent(keys='Enter')))
|
||||
assert dispatched_keys == [('keyDown', 'Enter', 0), ('keyUp', 'Enter', 0)]
|
||||
@@ -246,6 +246,26 @@ class TestBrowserSessionEventSystem:
|
||||
assert browser_session.event_bus.name.startswith('EventBus_')
|
||||
# Event bus name format may vary, just check it exists
|
||||
|
||||
@pytest.mark.parametrize('reset_method', ['stop', 'kill'])
|
||||
async def test_session_handlers_registered_after_event_bus_reset(self, browser_session: BrowserSession, reset_method: str):
|
||||
"""Session handlers must be restored when stop() or kill() replaces the event bus."""
|
||||
initial_bus = browser_session.event_bus
|
||||
initial_handlers = {
|
||||
event_name: [getattr(handler, '__name__', str(handler)) for handler in handlers]
|
||||
for event_name, handlers in initial_bus.handlers.items()
|
||||
}
|
||||
assert initial_handlers
|
||||
assert 'BrowserStartEvent' in initial_handlers
|
||||
assert 'BrowserStopEvent' in initial_handlers
|
||||
|
||||
await getattr(browser_session, reset_method)()
|
||||
|
||||
assert browser_session.event_bus is not initial_bus
|
||||
assert {
|
||||
event_name: [getattr(handler, '__name__', str(handler)) for handler in handlers]
|
||||
for event_name, handlers in browser_session.event_bus.handlers.items()
|
||||
} == initial_handlers
|
||||
|
||||
async def test_event_handlers_registration(self, browser_session: BrowserSession):
|
||||
"""Test that event handlers are properly registered."""
|
||||
# Attach all watchdogs to register their handlers
|
||||
|
||||
@@ -19,13 +19,17 @@ Usage:
|
||||
|
||||
import asyncio
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from pytest_httpserver import HTTPServer
|
||||
|
||||
from browser_use.agent.service import Agent
|
||||
from browser_use.agent.views import ActionModel
|
||||
from browser_use.browser import BrowserSession
|
||||
from browser_use.browser.events import SwitchTabEvent
|
||||
from browser_use.browser.profile import BrowserProfile
|
||||
from browser_use.tools.service import Tools
|
||||
from tests.ci.conftest import create_mock_llm
|
||||
|
||||
|
||||
@@ -669,3 +673,93 @@ class TestMultiTabOperations:
|
||||
assert 'Successfully' in final_result, 'Agent should report success'
|
||||
except TimeoutError:
|
||||
pytest.fail('Test timed out after 2 minutes - agent hung during multiple tab operations')
|
||||
|
||||
|
||||
class _TabActionModel(ActionModel):
|
||||
"""ActionModel with explicit slots for the tab actions driven directly via tools.act().
|
||||
|
||||
registry.create_action_model() builds its fields at runtime, so a statically declared
|
||||
subclass is what keeps pyright able to check these call sites. act() dispatches on the
|
||||
key returned by model_dump(exclude_unset=True), so the real registered actions still run.
|
||||
"""
|
||||
|
||||
switch: dict[str, Any] | None = None
|
||||
navigate: dict[str, Any] | None = None
|
||||
|
||||
|
||||
class TestSwitchTabFailureReporting:
|
||||
"""A failed `switch` must surface as ActionResult.error rather than a fake success.
|
||||
|
||||
Previously both failure paths (a stale/unknown tab_id, and a missing SwitchTabEvent
|
||||
result) returned a non-error ActionResult claiming the switch succeeded. The false
|
||||
claim was written into long_term_memory, so subsequent steps reasoned from a tab the
|
||||
agent never actually reached.
|
||||
"""
|
||||
|
||||
async def test_switch_to_nonexistent_tab_reports_error(self, browser_session):
|
||||
tools = Tools()
|
||||
|
||||
tabs_before = await browser_session.get_tabs()
|
||||
live_ids = {tab.target_id[-4:] for tab in tabs_before}
|
||||
bogus_tab_id = 'zzzz'
|
||||
assert bogus_tab_id not in live_ids
|
||||
|
||||
result = await tools.act(_TabActionModel(switch={'tab_id': bogus_tab_id}), browser_session=browser_session)
|
||||
|
||||
assert result.error is not None, 'a failed tab switch must set ActionResult.error'
|
||||
assert bogus_tab_id in result.error
|
||||
assert 'Switched to tab' not in (result.extracted_content or '')
|
||||
assert 'Switched to tab' not in (result.long_term_memory or '')
|
||||
|
||||
tabs_after = await browser_session.get_tabs()
|
||||
assert {tab.target_id[-4:] for tab in tabs_after} == live_ids, 'no tab switch should have happened'
|
||||
|
||||
async def test_switch_to_open_tab_still_succeeds(self, browser_session, base_url):
|
||||
tools = Tools()
|
||||
|
||||
original_tab_id = (await browser_session.get_tabs())[0].target_id[-4:]
|
||||
|
||||
open_result = await tools.act(
|
||||
_TabActionModel(navigate={'url': f'{base_url}/page1', 'new_tab': True}),
|
||||
browser_session=browser_session,
|
||||
)
|
||||
assert open_result.error is None, f'opening a new tab should not error: {open_result.error}'
|
||||
|
||||
result = await tools.act(_TabActionModel(switch={'tab_id': original_tab_id}), browser_session=browser_session)
|
||||
|
||||
assert result.error is None, f'switching to a live tab must not error: {result.error}'
|
||||
assert result.long_term_memory is not None
|
||||
assert 'Switched to tab' in result.long_term_memory
|
||||
|
||||
async def test_switch_reports_error_when_event_yields_no_result(self, browser_session, monkeypatch):
|
||||
"""A handler that completes without raising but yields no TargetID is still a failure.
|
||||
|
||||
on_SwitchTabEvent returns a TargetID on every success path, so a missing result is
|
||||
never a quiet success - this must be reported as an error even though nothing raised.
|
||||
"""
|
||||
tools = Tools()
|
||||
tab_id = (await browser_session.get_tabs())[0].target_id[-4:]
|
||||
|
||||
class NoResultEvent:
|
||||
async def _wait(self):
|
||||
return self
|
||||
|
||||
def __await__(self):
|
||||
return self._wait().__await__()
|
||||
|
||||
async def event_result(self, **_kwargs):
|
||||
return None
|
||||
|
||||
original_dispatch = browser_session.event_bus.dispatch
|
||||
monkeypatch.setattr(
|
||||
browser_session.event_bus,
|
||||
'dispatch',
|
||||
lambda event: NoResultEvent() if isinstance(event, SwitchTabEvent) else original_dispatch(event),
|
||||
)
|
||||
|
||||
result = await tools.act(_TabActionModel(switch={'tab_id': tab_id}), browser_session=browser_session)
|
||||
|
||||
assert result.error is not None, 'a switch that yields no result must set ActionResult.error'
|
||||
assert 'produced no result' in result.error
|
||||
assert 'Switched to tab' not in (result.extracted_content or '')
|
||||
assert 'Switched to tab' not in (result.long_term_memory or '')
|
||||
|
||||
@@ -465,8 +465,16 @@ class TestFileSystem:
|
||||
assert fs._is_valid_filename('.json') is False # no name
|
||||
assert fs._is_valid_filename('.jsonl') is False # no name
|
||||
assert fs._is_valid_filename('.csv') is False # no name
|
||||
assert fs._is_valid_filename('screenshot.png') is False # binary extension
|
||||
assert fs._is_valid_filename('image.jpg') is False # binary extension
|
||||
# Small image extensions are now supported (base64 content -> real bytes, for upload flows)
|
||||
assert fs._is_valid_filename('screenshot.png') is True
|
||||
assert fs._is_valid_filename('image.jpg') is True
|
||||
assert fs._is_valid_filename('pic.gif') is True
|
||||
assert fs._is_valid_filename('photo.webp') is True
|
||||
|
||||
# Other binary types remain unsupported
|
||||
assert fs._is_valid_filename('clip.mp4') is False # binary extension
|
||||
assert fs._is_valid_filename('archive.zip') is False # binary extension
|
||||
assert fs._is_valid_filename('icon.svg') is False # binary extension
|
||||
|
||||
def test_filename_parsing(self, temp_filesystem):
|
||||
"""Test filename parsing into name and extension."""
|
||||
@@ -1185,17 +1193,29 @@ class TestFilenameSanitization:
|
||||
fs.nuke()
|
||||
|
||||
async def test_write_file_binary_extension_error(self):
|
||||
"""Test that writing to binary extensions gives a clear error."""
|
||||
"""Unsupported binary extensions give a clear error; small images accept base64."""
|
||||
with tempfile.TemporaryDirectory() as tmp_dir:
|
||||
fs = FileSystem(base_dir=tmp_dir, create_default_files=False)
|
||||
|
||||
result = await fs.write_file('screenshot.png', 'content')
|
||||
# Non-image binaries are still rejected outright
|
||||
result = await fs.write_file('clip.mp4', 'content')
|
||||
assert 'binary/image' in result.lower() or 'Cannot write' in result
|
||||
assert 'screenshot.png' not in fs.list_files()
|
||||
assert 'clip.mp4' not in fs.list_files()
|
||||
|
||||
result = await fs.write_file('photo.jpg', 'content')
|
||||
result = await fs.write_file('archive.zip', 'content')
|
||||
assert 'binary/image' in result.lower() or 'Cannot write' in result
|
||||
|
||||
# Small images are supported: non-base64 content is rejected (no corrupt file),
|
||||
# valid base64 is written as real bytes.
|
||||
result = await fs.write_file('screenshot.png', 'not base64!!!')
|
||||
assert 'Error' in result
|
||||
assert not (fs.get_dir() / 'screenshot.png').exists()
|
||||
|
||||
png_1x1 = 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAAC0lEQVR4nGNgAAIAAAUAAXpeqz8AAAAASUVORK5CYII='
|
||||
result = await fs.write_file('logo.png', png_1x1)
|
||||
assert 'successfully' in result
|
||||
assert (fs.get_dir() / 'logo.png').read_bytes()[:8] == b'\x89PNG\r\n\x1a\n'
|
||||
|
||||
fs.nuke()
|
||||
|
||||
async def test_write_file_unsupported_extension_error(self):
|
||||
@@ -1315,8 +1335,8 @@ class TestFilenameSanitization:
|
||||
result = await fs.read_file('noextension')
|
||||
assert 'no extension' in result.lower()
|
||||
|
||||
# Binary extension - specific error
|
||||
result = await fs.write_file('image.png', 'data')
|
||||
# Unsupported binary extension - specific error
|
||||
result = await fs.write_file('clip.mp4', 'data')
|
||||
assert 'binary' in result.lower() or 'Cannot write' in result
|
||||
|
||||
fs.nuke()
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Regression tests for OpenRouter client setup and response handling."""
|
||||
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
from openai.types.chat import ChatCompletion, ChatCompletionMessage
|
||||
from openai.types.chat.chat_completion import Choice
|
||||
from pydantic import BaseModel
|
||||
|
||||
from browser_use.llm.exceptions import ModelProviderError
|
||||
from browser_use.llm.messages import UserMessage
|
||||
from browser_use.llm.openrouter.chat import ChatOpenRouter
|
||||
|
||||
|
||||
class Answer(BaseModel):
|
||||
answer: str
|
||||
|
||||
|
||||
def _completion(*, content: str | None = 'ok', choices: bool = True) -> ChatCompletion:
|
||||
return ChatCompletion(
|
||||
id='chatcmpl-test',
|
||||
choices=[Choice(finish_reason='stop', index=0, message=ChatCompletionMessage(role='assistant', content=content))]
|
||||
if choices
|
||||
else [],
|
||||
created=0,
|
||||
model='openai/gpt-4o',
|
||||
object='chat.completion',
|
||||
)
|
||||
|
||||
|
||||
async def test_request_params_reach_completion_not_client():
|
||||
llm = ChatOpenRouter(model='openai/gpt-4o', api_key='test-key', top_p=0.9, seed=42)
|
||||
|
||||
client = llm.get_client()
|
||||
|
||||
assert client.api_key == 'test-key'
|
||||
assert 'top_p' not in llm._get_client_params()
|
||||
assert 'seed' not in llm._get_client_params()
|
||||
|
||||
create = AsyncMock(return_value=_completion())
|
||||
with patch.object(type(client.chat.completions), 'create', create):
|
||||
await llm.ainvoke([UserMessage(content='question')])
|
||||
request_kwargs = create.await_args_list[0].kwargs
|
||||
assert request_kwargs['top_p'] == 0.9
|
||||
assert request_kwargs['seed'] == 42
|
||||
|
||||
|
||||
def test_provider_key_does_not_fall_back_to_openai_key(monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv('OPENAI_API_KEY', 'wrong-provider-key')
|
||||
monkeypatch.setenv('OPENROUTER_API_KEY', 'openrouter-key')
|
||||
assert ChatOpenRouter(model='openai/gpt-4o').get_client().api_key == 'openrouter-key'
|
||||
|
||||
monkeypatch.delenv('OPENROUTER_API_KEY')
|
||||
with pytest.raises(ModelProviderError, match='Missing OpenRouter API key') as exc_info:
|
||||
ChatOpenRouter(model='openai/gpt-4o').get_client()
|
||||
assert exc_info.value.status_code == 401
|
||||
|
||||
|
||||
async def test_empty_choices_raise_provider_error():
|
||||
llm = ChatOpenRouter(model='openai/gpt-4o', api_key='test-key')
|
||||
create = AsyncMock(return_value=_completion(choices=False))
|
||||
|
||||
with patch.object(type(llm.get_client().chat.completions), 'create', create):
|
||||
with pytest.raises(ModelProviderError, match='missing or empty `choices`') as exc_info:
|
||||
await llm.ainvoke([UserMessage(content='question')])
|
||||
|
||||
assert exc_info.value.status_code == 502
|
||||
|
||||
|
||||
async def test_structured_provider_error_keeps_status_code():
|
||||
llm = ChatOpenRouter(model='openai/gpt-4o', api_key='test-key')
|
||||
create = AsyncMock(return_value=_completion(content=None))
|
||||
|
||||
with patch.object(type(llm.get_client().chat.completions), 'create', create):
|
||||
with pytest.raises(ModelProviderError, match='Failed to parse structured output') as exc_info:
|
||||
await llm.ainvoke([UserMessage(content='question')], Answer)
|
||||
|
||||
assert exc_info.value.status_code == 500
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Regression tests for Vercel AI Gateway client setup."""
|
||||
|
||||
import pytest
|
||||
|
||||
from browser_use.llm.exceptions import ModelProviderError
|
||||
from browser_use.llm.vercel.chat import ChatVercel
|
||||
|
||||
|
||||
async def test_provider_key_does_not_fall_back_to_openai_key(monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setenv('OPENAI_API_KEY', 'wrong-provider-key')
|
||||
monkeypatch.setenv('AI_GATEWAY_API_KEY', 'gateway-key')
|
||||
assert ChatVercel(model='openai/gpt-4o').get_client().api_key == 'gateway-key'
|
||||
|
||||
monkeypatch.delenv('AI_GATEWAY_API_KEY')
|
||||
monkeypatch.delenv('VERCEL_OIDC_TOKEN', raising=False)
|
||||
with pytest.raises(ModelProviderError, match='Missing Vercel AI Gateway API key') as exc_info:
|
||||
await ChatVercel(model='openai/gpt-4o').ainvoke([])
|
||||
assert exc_info.value.status_code == 401
|
||||
@@ -623,3 +623,50 @@ def test_password_field_without_type_attribute():
|
||||
attrs_str = DOMTreeSerializer._build_attributes_string(node, list(DEFAULT_INCLUDE_ATTRIBUTES), '')
|
||||
|
||||
assert value in attrs_str, 'Input without type attribute should preserve its value'
|
||||
|
||||
|
||||
def test_history_filters_sensitive_data_inside_nested_lists(tmp_path):
|
||||
"""
|
||||
Saved history must not leak sensitive values that sit below a nested-list
|
||||
boundary in an action parameter (e.g. a list of rows, each row a list).
|
||||
"""
|
||||
from typing import Any
|
||||
|
||||
from pydantic import create_model
|
||||
|
||||
from browser_use.agent.views import AgentHistory, AgentHistoryList, AgentOutput
|
||||
from browser_use.browser.views import BrowserStateHistory
|
||||
from browser_use.tools.registry.views import ActionModel
|
||||
|
||||
class NestedInputAction(BaseModel):
|
||||
rows: list[list[str]]
|
||||
lookup: dict[str, list[dict[str, str]]]
|
||||
|
||||
InputActionModel = create_model('InputActionModel', __base__=ActionModel, input=(NestedInputAction | None, None))
|
||||
OutputModel = AgentOutput.type_with_custom_actions(InputActionModel)
|
||||
|
||||
# built via model_validate because create_model's field is invisible to static analysis
|
||||
action = InputActionModel.model_validate(
|
||||
{
|
||||
'input': {
|
||||
'rows': [['token-123']],
|
||||
'lookup': {'headers': [{'authorization': 'token-123'}]},
|
||||
}
|
||||
}
|
||||
)
|
||||
history = AgentHistoryList[Any](
|
||||
history=[
|
||||
AgentHistory(
|
||||
model_output=OutputModel(memory='', action=[action]),
|
||||
result=[],
|
||||
state=BrowserStateHistory(url='https://example.test', title='t', tabs=[], interacted_element=[None]),
|
||||
)
|
||||
]
|
||||
)
|
||||
|
||||
filepath = tmp_path / 'history.json'
|
||||
history.save_to_file(filepath, sensitive_data={'api_key': 'token-123'})
|
||||
saved = filepath.read_text(encoding='utf-8')
|
||||
|
||||
assert 'token-123' not in saved, 'Sensitive value leaked into the saved history file'
|
||||
assert saved.count('<secret>api_key</secret>') == 2
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
"""The DOM the agent sees must show the value a form field currently holds.
|
||||
|
||||
Regression test for #5647: values set by JavaScript, autofill, or a framework
|
||||
live in the element's `value` property, not in the `value` attribute, so the
|
||||
agent used to see pre-filled fields as empty.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from pytest_httpserver import HTTPServer
|
||||
|
||||
from browser_use.browser.events import NavigateToUrlEvent
|
||||
|
||||
PAGE = """<!DOCTYPE html>
|
||||
<html><head><title>Prefilled form</title></head>
|
||||
<body>
|
||||
<label for="name">Name</label>
|
||||
<input id="name" type="text" placeholder="Your name">
|
||||
<label for="notes">Notes</label>
|
||||
<textarea id="notes"></textarea>
|
||||
<input id="secret" type="password">
|
||||
<input id="otp" type="text" autocomplete="one-time-code">
|
||||
<input id="card" type="text" autocomplete="cc-number">
|
||||
<input id="agree" type="checkbox" checked>
|
||||
<input id="news" type="checkbox">
|
||||
<script>
|
||||
document.getElementById('name').value = 'Ada Lovelace';
|
||||
document.getElementById('notes').value = 'call back tuesday';
|
||||
document.getElementById('secret').value = 'hunter2';
|
||||
document.getElementById('otp').value = '493021';
|
||||
document.getElementById('card').value = '4242424242424242';
|
||||
document.getElementById('agree').checked = false;
|
||||
document.getElementById('news').checked = true;
|
||||
</script>
|
||||
</body></html>"""
|
||||
|
||||
|
||||
@pytest.fixture(scope='module')
|
||||
def http_server():
|
||||
server = HTTPServer()
|
||||
server.start()
|
||||
server.expect_request('/prefilled').respond_with_data(PAGE, content_type='text/html')
|
||||
yield server
|
||||
server.stop()
|
||||
|
||||
|
||||
async def test_live_input_values_reach_the_agent(browser_session, http_server):
|
||||
event = browser_session.event_bus.dispatch(NavigateToUrlEvent(url=http_server.url_for('/prefilled')))
|
||||
await event
|
||||
await event.event_result(raise_if_any=True, raise_if_none=False)
|
||||
|
||||
state = await browser_session.get_browser_state_summary()
|
||||
by_id = {node.attributes.get('id'): node for node in state.dom_state.selector_map.values() if node.attributes}
|
||||
|
||||
assert by_id['name'].attributes.get('value') == 'Ada Lovelace'
|
||||
assert by_id['notes'].attributes.get('value') == 'call back tuesday'
|
||||
assert by_id['secret'].attributes.get('value') is None, 'password values must not be exposed'
|
||||
assert by_id['otp'].attributes.get('value') is None, 'one-time codes must not be exposed'
|
||||
assert by_id['card'].attributes.get('value') is None, 'card numbers must not be exposed'
|
||||
assert by_id['secret'].snapshot_node is not None and by_id['secret'].snapshot_node.input_value is None
|
||||
assert by_id['agree'].attributes.get('checked') is None, 'live unchecked state wins over the checked attribute'
|
||||
assert by_id['news'].attributes.get('checked') == 'true'
|
||||
|
||||
llm_view = state.dom_state.llm_representation()
|
||||
assert 'Ada Lovelace' in llm_view
|
||||
assert 'call back tuesday' in llm_view
|
||||
assert 'hunter2' not in llm_view
|
||||
assert '493021' not in llm_view
|
||||
assert '4242424242424242' not in llm_view
|
||||
@@ -7,7 +7,7 @@ from pathlib import Path
|
||||
import pytest
|
||||
from PIL import Image
|
||||
|
||||
from browser_use.filesystem.file_system import FileSystem
|
||||
from browser_use.filesystem.file_system import Base64BinaryFile, FileSystem
|
||||
|
||||
|
||||
class TestImageFiles:
|
||||
@@ -247,3 +247,67 @@ class TestActionResultImages:
|
||||
|
||||
if __name__ == '__main__':
|
||||
pytest.main([__file__, '-v'])
|
||||
|
||||
|
||||
class TestBinaryFileCreation:
|
||||
"""The agent can author a tiny binary file (base64) that becomes a real, uploadable image."""
|
||||
|
||||
PNG_1X1 = 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAAC0lEQVR4nGNgAAIAAAUAAXpeqz8AAAAASUVORK5CYII='
|
||||
|
||||
async def test_write_png_produces_real_bytes(self, tmp_path: Path):
|
||||
fs = FileSystem(str(tmp_path))
|
||||
# write_file action appends a trailing newline; that must be tolerated
|
||||
result = await fs.write_file('logo.png', self.PNG_1X1 + '\n')
|
||||
assert 'successfully' in result
|
||||
raw = (fs.get_dir() / 'logo.png').read_bytes()
|
||||
assert raw[:8] == b'\x89PNG\r\n\x1a\n' # real PNG magic, not base64 text
|
||||
assert len(raw) > 0
|
||||
|
||||
async def test_binary_file_is_uploadable_by_basename(self, tmp_path: Path):
|
||||
fs = FileSystem(str(tmp_path))
|
||||
await fs.write_file('pic.gif', 'R0lGODlhAQABAIAAAAAAAP///yH5BAEAAAAALAAAAAABAAEAAAIBRAA7')
|
||||
fobj = fs.get_file('pic.gif') # the exact resolution upload_file uses
|
||||
assert fobj is not None
|
||||
assert fobj.full_name == 'pic.gif'
|
||||
|
||||
async def test_describe_does_not_leak_base64(self, tmp_path: Path):
|
||||
fs = FileSystem(str(tmp_path))
|
||||
await fs.write_file('logo.png', self.PNG_1X1)
|
||||
desc = fs.describe()
|
||||
assert self.PNG_1X1[:24] not in desc # base64 never enters the prompt
|
||||
assert '[binary png file' in desc
|
||||
|
||||
async def test_invalid_base64_is_rejected_no_corrupt_file(self, tmp_path: Path):
|
||||
fs = FileSystem(str(tmp_path))
|
||||
result = await fs.write_file('bad.png', 'this is definitely not base64 @@@')
|
||||
assert 'Error' in result
|
||||
assert not (fs.get_dir() / 'bad.png').exists() # no 0-byte / corrupt file left for upload
|
||||
# No ghost entry in the in-memory filesystem / state either
|
||||
assert 'bad.png' not in fs.list_files()
|
||||
assert 'bad.png' not in [f for f in fs.get_state().model_dump().get('files', {})]
|
||||
|
||||
async def test_valid_base64_but_not_an_image_is_rejected(self, tmp_path: Path):
|
||||
"""'aGVsbG8=' is valid base64 (-> b'hello') but not a PNG: must be rejected, not written."""
|
||||
fs = FileSystem(str(tmp_path))
|
||||
result = await fs.write_file('fake.png', 'aGVsbG8=')
|
||||
assert 'Error' in result
|
||||
assert not (fs.get_dir() / 'fake.png').exists()
|
||||
assert 'fake.png' not in fs.list_files()
|
||||
|
||||
async def test_wrong_magic_for_extension_is_rejected(self, tmp_path: Path):
|
||||
"""PNG bytes written under a .gif name are rejected (magic mismatch)."""
|
||||
fs = FileSystem(str(tmp_path))
|
||||
result = await fs.write_file('mislabeled.gif', self.PNG_1X1)
|
||||
assert 'Error' in result
|
||||
assert 'mislabeled.gif' not in fs.list_files()
|
||||
|
||||
async def test_state_round_trip_preserves_binary(self, tmp_path: Path):
|
||||
fs = FileSystem(str(tmp_path))
|
||||
await fs.write_file('logo.png', self.PNG_1X1)
|
||||
original_bytes = (fs.get_dir() / 'logo.png').read_bytes()
|
||||
fs2 = FileSystem.from_state(fs.get_state())
|
||||
restored = fs2.get_file('logo.png')
|
||||
assert isinstance(restored, Base64BinaryFile)
|
||||
# Content integrity, not just existence: decoded bytes must round-trip and be a real PNG
|
||||
assert restored._decoded() == original_bytes
|
||||
assert restored._decoded()[:8] == b'\x89PNG\r\n\x1a\n'
|
||||
|
||||
@@ -81,6 +81,79 @@ def test_image_only_interactive_parent_includes_child_image_context_in_llm_dom()
|
||||
assert 'acme-bank-primary-card.png' in llm_dom
|
||||
|
||||
|
||||
def test_direct_interactive_image_includes_own_context_in_llm_dom():
|
||||
"""A directly clickable image should expose its own sanitized source context."""
|
||||
image = _make_element_node(
|
||||
221,
|
||||
'img',
|
||||
{'src': 'https://cdn.example.test/logos/acme-card.png?token=must-not-leak#preview'},
|
||||
x=18,
|
||||
y=18,
|
||||
)
|
||||
|
||||
llm_dom = DOMTreeSerializer.serialize_tree(
|
||||
SimplifiedNode(
|
||||
original_node=image,
|
||||
children=[],
|
||||
is_interactive=True,
|
||||
selector_index=221,
|
||||
),
|
||||
DEFAULT_INCLUDE_ATTRIBUTES,
|
||||
)
|
||||
|
||||
assert '[221]<img' in llm_dom
|
||||
assert 'image_src=acme-card.png' in llm_dom
|
||||
assert 'must-not-leak' not in llm_dom
|
||||
|
||||
|
||||
def test_direct_interactive_image_rejects_browser_normalized_data_src():
|
||||
"""Data URLs remain hidden when URL parsing ignores control characters in the scheme."""
|
||||
image = _make_element_node(
|
||||
222,
|
||||
'img',
|
||||
{'src': '\x00Da\nTa:image/png;base64,must-not-leak'},
|
||||
x=18,
|
||||
y=18,
|
||||
)
|
||||
|
||||
llm_dom = DOMTreeSerializer.serialize_tree(
|
||||
SimplifiedNode(
|
||||
original_node=image,
|
||||
children=[],
|
||||
is_interactive=True,
|
||||
selector_index=222,
|
||||
),
|
||||
DEFAULT_INCLUDE_ATTRIBUTES,
|
||||
)
|
||||
|
||||
assert llm_dom == '[222]<img />'
|
||||
assert 'must-not-leak' not in llm_dom
|
||||
|
||||
|
||||
def test_direct_interactive_image_omits_oversized_src_context():
|
||||
"""Image context does not scan or serialize oversized raw source attributes."""
|
||||
image = _make_element_node(
|
||||
223,
|
||||
'img',
|
||||
{'src': f'/{"a" * DOMTreeSerializer.MAX_IMAGE_CONTEXT_ATTRIBUTE_LENGTH}/must-not-leak.png'},
|
||||
x=18,
|
||||
y=18,
|
||||
)
|
||||
|
||||
llm_dom = DOMTreeSerializer.serialize_tree(
|
||||
SimplifiedNode(
|
||||
original_node=image,
|
||||
children=[],
|
||||
is_interactive=True,
|
||||
selector_index=223,
|
||||
),
|
||||
DEFAULT_INCLUDE_ATTRIBUTES,
|
||||
)
|
||||
|
||||
assert llm_dom == '[223]<img />'
|
||||
assert 'must-not-leak' not in llm_dom
|
||||
|
||||
|
||||
def _simplified_image(backend_node_id: int, attributes: dict[str, str]) -> SimplifiedNode:
|
||||
image = _make_element_node(backend_node_id, 'img', attributes, x=18, y=18)
|
||||
return SimplifiedNode(original_node=image, children=[])
|
||||
|
||||
@@ -30,12 +30,12 @@ async def test_mcp_tool_isError_true_is_surfaced_as_action_result_error():
|
||||
client.session.call_tool = AsyncMock( # type: ignore[union-attr]
|
||||
return_value=types.CallToolResult(
|
||||
content=[types.TextContent(type='text', text='File not found: /tmp/does-not-exist.txt')],
|
||||
isError=True,
|
||||
is_error=True,
|
||||
)
|
||||
)
|
||||
|
||||
tools = Tools()
|
||||
tool = types.Tool(name='read_file', description='Read a file', inputSchema={'type': 'object', 'properties': {}})
|
||||
tool = types.Tool(name='read_file', description='Read a file', input_schema={'type': 'object', 'properties': {}})
|
||||
client._register_tool_as_action(tools.registry, 'read_file', tool)
|
||||
|
||||
result = await tools.registry.execute_action('read_file', {})
|
||||
@@ -52,7 +52,7 @@ async def test_parameterized_mcp_tool_isError_true_is_surfaced_as_action_result_
|
||||
client.session.call_tool = AsyncMock( # type: ignore[union-attr]
|
||||
return_value=types.CallToolResult(
|
||||
content=[types.TextContent(type='text', text='File not found: /tmp/does-not-exist.txt')],
|
||||
isError=True,
|
||||
is_error=True,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -60,7 +60,7 @@ async def test_parameterized_mcp_tool_isError_true_is_surfaced_as_action_result_
|
||||
tool = types.Tool(
|
||||
name='read_file',
|
||||
description='Read a file',
|
||||
inputSchema={
|
||||
input_schema={
|
||||
'type': 'object',
|
||||
'properties': {'path': {'type': 'string'}},
|
||||
'required': ['path'],
|
||||
@@ -83,12 +83,12 @@ async def test_mcp_tool_isError_false_still_succeeds():
|
||||
client.session.call_tool = AsyncMock( # type: ignore[union-attr]
|
||||
return_value=types.CallToolResult(
|
||||
content=[types.TextContent(type='text', text='ok')],
|
||||
isError=False,
|
||||
is_error=False,
|
||||
)
|
||||
)
|
||||
|
||||
tools = Tools()
|
||||
tool = types.Tool(name='read_file', description='Read a file', inputSchema={'type': 'object', 'properties': {}})
|
||||
tool = types.Tool(name='read_file', description='Read a file', input_schema={'type': 'object', 'properties': {}})
|
||||
client._register_tool_as_action(tools.registry, 'read_file', tool)
|
||||
|
||||
result = await tools.registry.execute_action('read_file', {})
|
||||
|
||||
@@ -35,14 +35,13 @@ def server() -> BrowserUseServer:
|
||||
|
||||
|
||||
def _is_read_only(tool: types.Tool) -> bool:
|
||||
return tool.annotations is not None and tool.annotations.readOnlyHint is True
|
||||
return tool.annotations is not None and tool.annotations.read_only_hint is True
|
||||
|
||||
|
||||
async def _list_tools(server: BrowserUseServer) -> list[types.Tool]:
|
||||
handler = server.server.request_handlers[types.ListToolsRequest]
|
||||
result = await handler(types.ListToolsRequest(method='tools/list'))
|
||||
assert isinstance(result, types.ServerResult), f'expected ServerResult, got {type(result).__name__}'
|
||||
list_result = result.root
|
||||
handler = server.server.get_request_handler('tools/list')
|
||||
assert handler is not None, 'tools/list handler is not registered'
|
||||
list_result = await handler.handler(None, types.PaginatedRequestParams()) # type: ignore[arg-type]
|
||||
assert isinstance(list_result, types.ListToolsResult), f'expected ListToolsResult, got {type(list_result).__name__}'
|
||||
assert len(list_result.tools) > 0, 'tools/list returned an empty catalogue'
|
||||
return list_result.tools
|
||||
@@ -73,3 +72,20 @@ async def test_mutating_tools_never_advertise_read_only_hint(server: BrowserUseS
|
||||
|
||||
mislabeled = sorted(tool.name for tool in tools if tool.name not in EXPECTED_READ_ONLY_TOOLS and _is_read_only(tool))
|
||||
assert not mislabeled, f'state-changing tools wrongly advertise readOnlyHint=True: {mislabeled}'
|
||||
|
||||
|
||||
async def test_unknown_tool_is_reported_as_mcp_error(server: BrowserUseServer) -> None:
|
||||
"""Unknown tool calls must not be returned as successful MCP results."""
|
||||
handler = server.server.get_request_handler('tools/call')
|
||||
assert handler is not None, 'tools/call handler is not registered'
|
||||
|
||||
result = await handler.handler(
|
||||
None, # type: ignore[arg-type]
|
||||
types.CallToolRequestParams(name='does_not_exist', arguments={}),
|
||||
)
|
||||
|
||||
assert isinstance(result, types.CallToolResult)
|
||||
assert result.is_error is True
|
||||
assert len(result.content) == 1
|
||||
assert isinstance(result.content[0], types.TextContent)
|
||||
assert 'Unknown tool: does_not_exist' in result.content[0].text
|
||||
|
||||
Reference in New Issue
Block a user