Merge branch 'main' into codex/fix-agent-download-path-order

This commit is contained in:
Magnus Müller
2026-08-30 22:47:23 -07:00
committed by GitHub
75 changed files with 1784 additions and 2114 deletions
-35
View File
@@ -1,35 +0,0 @@
name: cloud_evals
# Cancel in-progress runs when a new commit is pushed to the same branch/PR
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
on:
push:
branches:
- main
- 'releases/*'
workflow_dispatch:
inputs:
commit_hash:
description: Commit hash of the library to build the Cloud eval image for
required: false
permissions: {}
jobs:
trigger_cloud_eval_image_build:
runs-on: ubuntu-latest
steps:
- uses: actions/github-script@v7
with:
github-token: ${{ secrets.TRIGGER_CLOUD_BUILD_GH_KEY }}
script: |
const result = await github.rest.repos.createDispatchEvent({
owner: 'browser-use',
repo: 'cloud',
event_type: 'trigger-workflow',
client_payload: {"commit_hash": "${{ github.event.inputs.commit_hash || github.sha }}"}
})
console.log(result)
+1 -1
View File
@@ -19,7 +19,7 @@ The key product of Browser Use Cloud is the completion of user tasks.
- Profile Sync is the best way to handle authentication for tasks. This feature allows users to upload their local browser cookies (where the user is already logged into the services they need authentication for) to a Browser Profile that can be used for tasks on the cloud. To initiate a Profile Sync, a user must run `export BROWSER_USE_API_KEY=<your_key> && curl -fsSL https://browser-use.com/profile.sh | sh` and follow the steps in the interactive terminal.
## Quickstart
To get started, direct the user to first must create an account, purchase credits (or simply claim the five free tasks given on account creation), and generate an API key on the Browser Use online platform: https://cloud.browser-use.com/. These are the only steps that can only be done on the platform.
To get started, direct the user to first create an account, claim the $15 one-time signup credit if eligible (or purchase credits), and generate an API key on the Browser Use online platform: https://cloud.browser-use.com/. These are the only steps that can only be done on the platform.
Avoid giving the user all of the following steps at once as it may seem overwheling. Instead present one step at a time and only continue when asked. Do as much for the user as you are able to.
+5 -12
View File
@@ -41,13 +41,13 @@
# What can Browser Use do?
Browser Use lets an AI agent use a web browser the same way you do — it opens pages, clicks buttons, types, and fills in forms. You describe the task, and it completes it. For example, you can have it:
Browser Use lets an AI agent use a web browser the same way humans do — it opens pages, clicks buttons, types, and fills in forms. You describe the task, and it completes it. For example, you can have it:
### 📋 Fill Forms
#### Task: "Fill in this job application with my resume and information."
![Job Application Demo](https://github.com/user-attachments/assets/57865ee6-6004-49d5-b2c2-6dff39ec2ba9)
![Job Application Demo](https://github.com/user-attachments/assets/57611d8e-0474-4de6-84b7-37a0c0cd27e7)
[Example code ↗](https://github.com/browser-use/browser-use/blob/main/examples/use-cases/apply_to_job.py)
@@ -55,19 +55,11 @@ Browser Use lets an AI agent use a web browser the same way you do — it opens
### 🍎 Extract data
#### Task: "Extract structured data about my followers and export it as a CSV."
https://github.com/user-attachments/assets/93714c75-98f4-4cfc-add1-69c38b5138b5
https://github.com/user-attachments/assets/485fd3ec-61b9-4afc-9e86-ee9b85acb592
[Browser Use Cloud Docs ↗](https://docs.browser-use.com/cloud/quickstart)
### 💻 QA Automation
#### Task: "QA test my local website and report any bugs, usability issues, and visual inconsistencies."
<img width="1920" height="1080" alt="qa-demo-small" src="https://github.com/user-attachments/assets/bf590697-df9c-4e79-b646-d6f52bfea976" />
[Browser Use CLI ↗](https://docs.browser-use.com/open-source/browser-use-cli)
<br/>
# Quickstart
@@ -113,7 +105,7 @@ async def main():
agent = Agent(
task="Find the number of stars of the browser-use repo",
llm=ChatBrowserUse(model='openai/gpt-5.5'),
# llm=ChatBrowserUse(model='bu-2-0'), # Browser Use's optimized model
# llm=ChatBrowserUse(model='bu-2-0-mini-preview'), # Browser Use's optimized model
# llm=ChatOpenAI(model='gpt-5.5'),
# llm=ChatAnthropic(model='claude-opus-4-8'), # Sonnet also works well
)
@@ -150,6 +142,7 @@ Browser Use is also **#1 on the [Odysseys leaderboard](https://odysseysbench.com
- Best stealth with proxy rotation and captcha solving
- 1000+ integrations (Gmail, Slack, Notion, and more)
- Persistent filesystem and memory
- Rerunnable scripts fetch live data, even when sites change ([guide](https://docs.browser-use.com/cloud/agent/scripts))
```sh
curl -X POST https://api.browser-use.com/api/v4/runs \
+5 -13
View File
@@ -77,6 +77,7 @@ from browser_use.utils import (
_log_pretty_path,
check_latest_browser_use_version,
get_browser_use_version,
has_url_negation,
is_placeholder_url,
sanitize_url_candidate,
time_execution_async,
@@ -2372,13 +2373,6 @@ class Agent(Generic[Context, AgentStructuredOutput]):
'polynomial',
}
excluded_words = {
'never',
'dont',
'not',
"don't",
}
found_urls = []
matched_spans: list[tuple[int, int]] = []
for pattern in patterns:
@@ -2416,16 +2410,14 @@ class Agent(Generic[Context, AgentStructuredOutput]):
self.logger.debug(f'Excluding URL with file extension from auto-navigation: {url}')
continue
# If in the 20 characters before the url position is a word in excluded_words skip to avoid "Never go to this url"
# Skip URLs explicitly negated by nearby prose, such as "Never go to this URL".
context_start = max(0, original_position - 20)
context_text = task_without_emails[context_start:original_position]
if any(word.lower() in context_text.lower() for word in excluded_words):
self.logger.debug(
f'Excluding URL with word in excluded words from auto-navigation: {url} (context: "{context_text.strip()}")'
)
if has_url_negation(context_text):
self.logger.debug(f'Excluding negated URL from auto-navigation: {url} (context: "{context_text.strip()}")')
continue
# Add https:// if missing (after excluded words check to avoid position calculation issues)
# Add https:// after the negation check to preserve source positions.
if not has_scheme:
url = 'https://' + url
+2 -3
View File
@@ -76,6 +76,7 @@ from browser_use.utils import (
check_latest_browser_use_version,
get_browser_use_version,
get_git_info,
has_url_negation,
is_placeholder_url,
sanitize_url_candidate,
)
@@ -1460,8 +1461,6 @@ def _extract_start_url(task: str) -> str | None:
'rpm',
'iso',
}
excluded_words = {'never', 'dont', 'not', "don't"}
found_urls = []
matched_spans: list[tuple[int, int]] = []
for pattern in patterns:
@@ -1483,7 +1482,7 @@ def _extract_start_url(task: str) -> str | None:
continue
context_start = max(0, match.start() - 20)
context_text = task_without_emails[context_start : match.start()]
if any(word in context_text.lower() for word in excluded_words):
if has_url_negation(context_text):
continue
if not has_scheme:
url = 'https://' + url
+12 -1
View File
@@ -25,6 +25,14 @@ def _get_enable_default_extensions_default() -> bool:
return True
def _get_headless_default() -> bool | None:
"""Get the default value for headless from BROWSER_USE_HEADLESS env var, or None to fall back to display detection."""
env_val = os.getenv('BROWSER_USE_HEADLESS')
if env_val is not None:
return env_val.lower() not in ('0', 'false', 'no', 'off', '')
return None
CHROME_DEBUG_PORT = 9242 # use a non-default port to avoid conflicts with other tools / devs using 9222
DOMAIN_OPTIMIZATION_THRESHOLD = 100 # Convert domain lists to sets for O(1) lookup when >= this size
CHROME_PROFILE_TRANSIENT_FILE_PATTERNS = (
@@ -419,7 +427,10 @@ class BrowserLaunchArgs(BaseModel):
validation_alias=AliasChoices('browser_binary_path', 'chrome_binary_path'),
description='Path to the chromium-based browser executable to use.',
)
headless: bool | None = Field(default=None, description='Whether to run the browser in headless or windowed mode.')
headless: bool | None = Field(
default_factory=_get_headless_default,
description='Whether to run the browser in headless or windowed mode. Can be set via BROWSER_USE_HEADLESS environment variable.',
)
args: list[CliArgStr] = Field(
default_factory=list, description='List of *extra* CLI args to pass to the browser when launching.'
)
+17 -3
View File
@@ -590,6 +590,7 @@ class BrowserSession(BaseModel):
_reconnect_event: asyncio.Event = PrivateAttr(default_factory=asyncio.Event)
_reconnect_lock: asyncio.Lock = PrivateAttr(default_factory=asyncio.Lock)
_reconnect_task: asyncio.Task | None = PrivateAttr(default=None)
_reconnect_pending: bool = PrivateAttr(default=False)
_intentional_stop: bool = PrivateAttr(default=False)
_logger: Any = PrivateAttr(default=None)
@@ -634,6 +635,7 @@ class BrowserSession(BaseModel):
if self._reconnect_task and not self._reconnect_task.done():
self._reconnect_task.cancel()
self._reconnect_task = None
self._reconnect_pending = False
self._reconnecting = False
self._reconnect_event.set() # unblock any waiters
@@ -1422,7 +1424,7 @@ class BrowserSession(BaseModel):
async def clear_cookies(self) -> None:
"""Clear all cookies."""
await self.cdp_client.send.Network.clearBrowserCookies()
await self.cdp_client.send.Storage.clearCookies()
async def export_storage_state(self, output_path: str | Path | None = None) -> dict[str, Any]:
"""Export all browser cookies and storage to storage_state format.
@@ -2291,9 +2293,18 @@ class BrowserSession(BaseModel):
)
)
finally:
reconnect_pending = self._reconnect_pending
self._reconnect_pending = False
self._reconnecting = False
self._reconnect_event.set() # wake up all waiters regardless of outcome
if reconnect_pending and not self._intentional_stop and self.cdp_url:
try:
loop = asyncio.get_running_loop()
self._reconnect_task = loop.create_task(self._auto_reconnect())
except RuntimeError:
self.logger.error('🔌 No event loop available for pending auto-reconnect')
def _attach_ws_drop_callback(self) -> None:
"""Attach a done callback to the CDPClient's message handler task to detect WS drops."""
if not self._cdp_client_root or not hasattr(self._cdp_client_root, '_message_handler_task'):
@@ -2304,8 +2315,11 @@ class BrowserSession(BaseModel):
return
def _on_message_handler_done(fut: asyncio.Future) -> None:
# Guard: skip if intentionally stopped, already reconnecting, or no cdp_url
if self._intentional_stop or self._reconnecting or not self.cdp_url:
# Guard: skip if intentionally stopped or no cdp_url
if self._intentional_stop or not self.cdp_url:
return
if self._reconnecting:
self._reconnect_pending = True
return
# The message handler task exiting means the WS connection dropped
+61
View File
@@ -55,6 +55,8 @@ class DOMTreeSerializer:
# {'tag': 'span', 'role': 'link'}, # <span role="link">
]
DEFAULT_CONTAINMENT_THRESHOLD = 0.99 # 99% containment by default
MAX_CHILD_IMAGE_CONTEXTS = 3
MAX_CHILD_IMAGE_DESCENDANTS = 100
def __init__(
self,
@@ -918,6 +920,58 @@ class DOMTreeSerializer:
return False
@staticmethod
def _get_child_image_context(node: SimplifiedNode) -> str:
"""Extract compact context from image descendants of an interactive element."""
image_context: list[str] = []
def normalize_src(src: str) -> str:
clean_src = src.strip()
if clean_src.lower().startswith('data:'):
return ''
path_without_query = clean_src.split('?', 1)[0].split('#', 1)[0].rstrip('/')
return path_without_query.rsplit('/', 1)[-1]
child_iterators = [iter(node.children)]
visited_descendants = 0
while (
child_iterators
and visited_descendants < DOMTreeSerializer.MAX_CHILD_IMAGE_DESCENDANTS
and len(image_context) < DOMTreeSerializer.MAX_CHILD_IMAGE_CONTEXTS
):
try:
current = next(child_iterators[-1])
except StopIteration:
child_iterators.pop()
continue
visited_descendants += 1
original_node = current.original_node
if original_node.node_type == NodeType.ELEMENT_NODE and original_node.tag_name == 'img':
attributes = original_node.attributes or {}
parts = []
for attr_name, output_name in (
('alt', 'image_alt'),
('title', 'image_title'),
('aria-label', 'image_label'),
):
attr_value = str(attributes.get(attr_name) or '').strip()
if attr_value:
parts.append(f'{output_name}={cap_text_length(attr_value, 100)}')
src = normalize_src(str(attributes.get('src') or ''))
if src:
parts.append(f'image_src={cap_text_length(src, 100)}')
if parts:
image_context.append(' '.join(parts))
if current.children:
child_iterators.append(iter(current.children))
return ' '.join(image_context)
@staticmethod
def serialize_tree(node: SimplifiedNode | None, include_attributes: list[str], depth: int = 0) -> str:
"""Serialize the optimized tree to string format."""
@@ -989,6 +1043,13 @@ class DOMTreeSerializer:
attributes_html_str = DOMTreeSerializer._build_attributes_string(
node.original_node, include_attributes, text_content
)
if node.is_interactive:
image_context = DOMTreeSerializer._get_child_image_context(node)
if image_context:
if attributes_html_str:
attributes_html_str += f' {image_context}'
else:
attributes_html_str = image_context
# Add compound component information to attributes if present
if node.original_node._compound_children:
+78 -24
View File
@@ -1,6 +1,7 @@
import asyncio
import base64
import csv
import html
import io
import os
import re
@@ -73,6 +74,51 @@ def _build_filename_error_message(file_name: str, supported_extensions: list[str
)
def _split_heading(line: str) -> tuple[str, int | None]:
"""Split a markdown ATX heading into (text, level).
Only ``# `` / ``## `` / ``### `` (note the required space) are headings.
Anything else, including ``#hashtag``, is returned unchanged with level None.
"""
if line.startswith('### '):
return line[4:], 3
if line.startswith('## '):
return line[3:], 2
if line.startswith('# '):
return line[2:], 1
return line, None
_BULLET_RE = re.compile(r'^(\s*)[-*]\s+(.*)$')
def _markdown_inline_to_rml(text: str) -> str:
"""Escape plain text, then convert a markdown subset to ReportLab markup.
Order matters: after html.escape there are no user-supplied ``<`` left, so
injected ``<b>`` / ``<i>`` / ``<font>`` tags are unambiguous.
Underscore emphasis is intentionally unsupported so ``snake_case`` identifiers
survive unchanged. Inline code is stashed before emphasis so markers inside
backticks stay Courier-only. Bold content cannot start with ``/``, so globs
like ``**/foo/**`` stay literal.
"""
text = html.escape(text)
rendered: list[str] = []
for part in re.split(r'(`[^`]+`)', text):
if part.startswith('`') and part.endswith('`'):
rendered.append(f'<font face="Courier">{part[1:-1]}</font>')
continue
# Bold before italic so ``**`` is not treated as two italic markers.
# Content cannot contain ``*`` — otherwise globs like ``*.txt and **/*.py`` pair across tokens.
# Content cannot start with ``/`` — otherwise ``**/foo/**`` is treated as bold.
part = re.sub(r'\*\*([^\s*/](?:[^*]*[^\s*])?)\*\*', r'<b>\1</b>', part)
# Non-space boundaries keep ``2 * 3 * 4`` literal; leading ``* `` is a bullet, not italic
part = re.sub(r'(?<!\*)\*([^\s*](?:[^*]*[^\s*])?)\*(?!\*)', r'<i>\1</i>', part)
rendered.append(part)
return ''.join(rendered)
DEFAULT_FILE_SYSTEM_PATH = 'browseruse_agent_data'
@@ -256,26 +302,36 @@ class PdfFile(BaseFile):
doc = SimpleDocTemplate(str(file_path), pagesize=letter)
styles = getSampleStyleSheet()
story = []
heading_styles = {1: styles['Title'], 2: styles['Heading1'], 3: styles['Heading2']}
# Convert markdown content to simple text and add to PDF
# For basic implementation, we'll treat content as plain text
# This avoids the AGPL license issue while maintaining functionality
content_lines = self.content.split('\n')
# Escape first, then markdown → RML. Avoids an AGPL markdown-to-PDF dependency.
in_fence = False
for line in self.content.split('\n'):
stripped = line.strip()
if stripped.startswith('```'):
in_fence = not in_fence
continue
for line in content_lines:
if line.strip():
# Handle basic markdown headers
if line.startswith('# '):
para = Paragraph(line[2:], styles['Title'])
elif line.startswith('## '):
para = Paragraph(line[3:], styles['Heading1'])
elif line.startswith('### '):
para = Paragraph(line[4:], styles['Heading2'])
else:
para = Paragraph(line, styles['Normal'])
story.append(para)
else:
if not stripped:
story.append(Spacer(1, 6))
continue
if in_fence:
# Fenced blocks are literal: no emphasis / inline-code conversion
story.append(Paragraph(html.escape(line), styles['Code']))
continue
text, heading_level = _split_heading(line)
if heading_level is not None:
story.append(Paragraph(_markdown_inline_to_rml(text), heading_styles[heading_level]))
continue
bullet = _BULLET_RE.match(line)
if bullet:
story.append(Paragraph(f'&bull; {_markdown_inline_to_rml(bullet.group(2))}', styles['Normal']))
continue
story.append(Paragraph(_markdown_inline_to_rml(line), styles['Normal']))
doc.build(story)
except Exception as e:
@@ -305,13 +361,9 @@ class DocxFile(BaseFile):
for line in content_lines:
if line.strip():
# Handle basic markdown headers
if line.startswith('# '):
doc.add_heading(line[2:], level=1)
elif line.startswith('## '):
doc.add_heading(line[3:], level=2)
elif line.startswith('### '):
doc.add_heading(line[4:], level=3)
text, heading_level = _split_heading(line)
if heading_level is not None:
doc.add_heading(text, level=heading_level)
else:
doc.add_paragraph(line)
else:
@@ -792,6 +844,8 @@ class FileSystem:
try:
content = file_obj.read()
if old_str not in content:
return f'Error: Could not find the specified text in file {full_filename}.'
content = content.replace(old_str, new_str)
await file_obj.write(content, self.data_dir)
sanitize_note = f" (auto-corrected from '{original_filename}')" if was_sanitized else ''
+1 -2
View File
@@ -89,8 +89,7 @@ class ChatAnthropicBedrock(ChatAWSBedrock):
client_params['aws_session_token'] = self.aws_session_token
# Add optional parameters
if self.max_retries:
client_params['max_retries'] = self.max_retries
client_params['max_retries'] = self.max_retries
if self.default_headers:
client_params['default_headers'] = self.default_headers
if self.default_query:
+2 -1
View File
@@ -9,6 +9,7 @@ from openai.types.responses import Response
from openai.types.shared import ChatModel
from pydantic import BaseModel
from browser_use.llm.base import is_reasoning_model
from browser_use.llm.exceptions import ModelProviderError, ModelRateLimitError
from browser_use.llm.messages import BaseMessage
from browser_use.llm.openai.like import ChatOpenAILike
@@ -179,7 +180,7 @@ class ChatAzureOpenAI(ChatOpenAILike):
model_params['service_tier'] = self.service_tier
# Handle reasoning models
if self.reasoning_models and any(str(m).lower() in str(self.model).lower() for m in self.reasoning_models):
if is_reasoning_model(self.model, self.reasoning_models):
# For reasoning models, use reasoning parameter instead of reasoning_effort
model_params['reasoning'] = {'effort': self.reasoning_effort}
model_params.pop('temperature', None)
+15
View File
@@ -4,6 +4,7 @@ We have switched all of our code from langchain to openai.types.chat.chat_comple
For easier transition we have
"""
from collections.abc import Iterable
from typing import Any, Protocol, TypeVar, overload, runtime_checkable
from pydantic import BaseModel
@@ -14,6 +15,20 @@ from browser_use.llm.views import ChatInvokeCompletion
T = TypeVar('T', bound=BaseModel)
def is_reasoning_model(model: object, reasoning_models: Iterable[object] | None) -> bool:
"""Return whether a model matches a non-empty reasoning-model pattern."""
if not reasoning_models:
return False
model_name = str(model).lower()
for pattern in reasoning_models:
pattern_name = str(pattern).lower()
if pattern_name.strip() and pattern_name in model_name:
return True
return False
@runtime_checkable
class BaseChatModel(Protocol):
_verified_api_keys: bool = False
+6 -4
View File
@@ -58,8 +58,9 @@ class ChatBrowserUse(BaseChatModel):
Args:
model: Model name to use. Options:
- 'bu-2-0' or 'bu-latest': Default model (latest premium)
- 'bu-1-0': Previous generation model
- 'bu-2-0' or 'bu-latest': Default model (premium)
- 'bu-2-0-mini-preview': Cheaper and faster per token, opt-in while in preview
- 'bu-1-0': Previous generation model, redirected to bu-2-0 at the gateway
- 'bu-qa-1': Website QA model (tests a site and scores functionality/aesthetics)
- 'browser-use/bu-30b-a3b-preview': Browser Use Open Source Model
- Provider-prefixed ids resolved by the gateway, e.g. 'anthropic/claude-sonnet-4-6',
@@ -73,7 +74,7 @@ class ChatBrowserUse(BaseChatModel):
"""
# Accept 'bu-*' aliases and any provider-prefixed id; the gateway resolves the
# latter (anthropic/*, openai/*, google/*, browser-use/*), so we don't enumerate them.
bu_aliases = ['bu-latest', 'bu-1-0', 'bu-2-0', 'bu-qa-1']
bu_aliases = ['bu-latest', 'bu-1-0', 'bu-2-0', 'bu-2-0-mini-preview', 'bu-qa-1']
is_valid = model in bu_aliases or '/' in model
if not is_valid:
raise ValueError(
@@ -82,7 +83,8 @@ class ChatBrowserUse(BaseChatModel):
"'openai/gpt-5.5', or 'google/gemini-3-pro'."
)
# Normalize bu-latest to the current latest model
# Normalize bu-latest to the current latest model, which is also the default: a
# preview model is opt-in, never something a caller lands on by omission.
if model == 'bu-latest':
self.model = 'bu-2-0'
else:
+25 -11
View File
@@ -29,7 +29,7 @@ T = TypeVar('T', bound=BaseModel)
class ChatDeepSeek(BaseChatModel):
"""DeepSeek /chat/completions wrapper (OpenAI-compatible)."""
model: str = 'deepseek-chat'
model: str = 'deepseek-v4-flash'
# Generation parameters
max_tokens: int | None = None
@@ -43,6 +43,8 @@ class ChatDeepSeek(BaseChatModel):
timeout: float | httpx.Timeout | None = None
client_params: dict[str, Any] | None = None
thinking: bool = False
@property
def provider(self) -> str:
return 'deepseek'
@@ -59,6 +61,27 @@ class ChatDeepSeek(BaseChatModel):
def name(self) -> str:
return self.model
def _supports_thinking(self) -> bool:
return 'deepseek-v4' in self.model.lower()
def _request_kwargs(self) -> dict[str, Any]:
common: dict[str, Any] = {}
if self.temperature is not None:
common['temperature'] = self.temperature
if self.max_tokens is not None:
common['max_tokens'] = self.max_tokens
if self.top_p is not None:
common['top_p'] = self.top_p
if self.seed is not None:
common['seed'] = self.seed
if self._supports_thinking():
common['extra_body'] = {
'thinking': {'type': 'enabled' if self.thinking else 'disabled'},
}
return common
@overload
async def ainvoke(
self,
@@ -96,16 +119,7 @@ class ChatDeepSeek(BaseChatModel):
"""
client = self._client()
ds_messages = DeepSeekMessageSerializer.serialize_messages(messages)
common: dict[str, Any] = {}
if self.temperature is not None:
common['temperature'] = self.temperature
if self.max_tokens is not None:
common['max_tokens'] = self.max_tokens
if self.top_p is not None:
common['top_p'] = self.top_p
if self.seed is not None:
common['seed'] = self.seed
common = self._request_kwargs()
# Beta conversation prefix continuation (see official documentation)
if self.base_url and str(self.base_url).endswith('/beta'):
+11 -10
View File
@@ -77,20 +77,21 @@ class GoogleMessageSerializer:
message_parts: list[Part] = []
# If this is the first user message and we have system parts, prepend them
system_text = None
if include_system_in_user and system_parts and role == 'user' and not formatted_messages:
system_text = '\n\n'.join(system_parts)
if isinstance(message.content, str):
message_parts.append(Part.from_text(text=f'{system_text}\n\n{message.content}'))
else:
# Add system text as the first part
message_parts.append(Part.from_text(text=system_text))
system_parts = [] # Clear after using
# Extract content and create parts
if isinstance(message.content, str):
# Regular text content
text = f'{system_text}\n\n{message.content}' if system_text is not None else message.content
message_parts.append(Part.from_text(text=text))
else:
# Extract content and create parts normally
if isinstance(message.content, str):
# Regular text content
message_parts = [Part.from_text(text=message.content)]
elif message.content is not None:
if system_text is not None:
# Add system text as the first part, the message's own parts still follow
message_parts.append(Part.from_text(text=system_text))
if message.content is not None:
# Handle Iterable of content parts
for part in message.content:
if part.type == 'text':
+3 -1
View File
@@ -8,7 +8,7 @@ Usage:
model = llm.azure_gpt_4_1_mini
model = llm.openai_gpt_4o
model = llm.google_gemini_2_5_pro
model = llm.bu_latest # or bu_1_0, bu_2_0
model = llm.bu_latest # or bu_2_0_mini_preview, bu_2_0, bu_1_0
"""
import os
@@ -83,6 +83,7 @@ cerebras_gemma_4_31b: 'BaseChatModel'
bu_latest: 'BaseChatModel'
bu_1_0: 'BaseChatModel'
bu_2_0: 'BaseChatModel'
bu_2_0_mini_preview: 'BaseChatModel'
def get_llm_by_name(model_name: str):
@@ -319,6 +320,7 @@ __all__ += [
'bu_latest',
'bu_1_0',
'bu_2_0',
'bu_2_0_mini_preview',
]
# NOTE: OCI backend is optional. The try/except ImportError and conditional __all__ are required
+63 -14
View File
@@ -1,3 +1,5 @@
import logging
import re
from collections.abc import Mapping
from dataclasses import dataclass
from typing import Any, TypeVar, overload
@@ -5,7 +7,7 @@ from typing import Any, TypeVar, overload
import httpx
from ollama import AsyncClient as OllamaAsyncClient
from ollama import Options
from pydantic import BaseModel
from pydantic import BaseModel, ValidationError
from browser_use.llm.base import BaseChatModel
from browser_use.llm.exceptions import ModelProviderError
@@ -14,6 +16,21 @@ from browser_use.llm.ollama.serializer import OllamaMessageSerializer
from browser_use.llm.views import ChatInvokeCompletion
T = TypeVar('T', bound=BaseModel)
logger = logging.getLogger(__name__)
# These belong on AsyncClient.chat(), not in the model `options` dict.
_PASSTHROUGH_CHAT_KEYS = frozenset({'think', 'logprobs', 'top_logprobs', 'keep_alive'})
_IGNORED_CHAT_KEYS = frozenset({'format', 'stream'})
_JSON_FENCE_RE = re.compile(r'\A```[ \t]*(?:json)?[ \t]*\r?\n(?P<body>.*?)\r?\n?```[ \t]*\Z', re.IGNORECASE | re.DOTALL)
def _unwrap_json_content(content: str) -> str:
"""Strip markdown code fences that Ollama vision models often wrap around JSON."""
text = content.strip()
match = _JSON_FENCE_RE.fullmatch(text)
if match:
return match.group('body').strip()
return text
@dataclass
@@ -57,6 +74,29 @@ class ChatOllama(BaseChatModel):
def name(self) -> str:
return self.model
def _split_chat_options(self) -> tuple[Mapping[str, Any] | Options | None, dict[str, Any]]:
"""Split model options from supported top-level ``chat()`` parameters.
``format`` and ``stream`` cannot be honored here because this wrapper owns
the structured-output schema and requires a non-streaming response.
"""
options = self.ollama_options
if not options or not isinstance(options, Mapping):
return options, {}
top_level = {key: options[key] for key in _PASSTHROUGH_CHAT_KEYS if key in options}
ignored = sorted(key for key in options if key in _IGNORED_CHAT_KEYS)
if ignored:
logger.warning(
'Ignoring %s in ollama_options; ChatOllama controls structured output and streaming',
', '.join(ignored),
)
extracted = _PASSTHROUGH_CHAT_KEYS | _IGNORED_CHAT_KEYS
model_options = {key: value for key, value in options.items() if key not in extracted}
return model_options, top_level
@overload
async def ainvoke(
self, messages: list[BaseMessage], output_format: None = None, **kwargs: Any
@@ -71,29 +111,38 @@ class ChatOllama(BaseChatModel):
ollama_messages = OllamaMessageSerializer.serialize_messages(messages)
try:
options, top_level = self._split_chat_options()
if output_format is None:
response = await self.get_client().chat(
model=self.model,
messages=ollama_messages,
options=self.ollama_options,
options=options,
**top_level,
)
return ChatInvokeCompletion(completion=response.message.content or '', usage=None)
else:
schema = output_format.model_json_schema()
response = await self.get_client().chat(
model=self.model,
messages=ollama_messages,
format=schema,
options=self.ollama_options,
)
schema = output_format.model_json_schema()
response = await self.get_client().chat(
model=self.model,
messages=ollama_messages,
format=schema,
options=options,
**top_level,
)
completion = response.message.content or ''
if output_format is not None:
completion = output_format.model_validate_json(completion)
completion = _unwrap_json_content(response.message.content or '')
try:
parsed = output_format.model_validate_json(completion)
except ValidationError as e:
raise ModelProviderError(
message=f'Ollama returned invalid JSON for structured output: {e}',
model=self.name,
) from e
return ChatInvokeCompletion(completion=completion, usage=None)
return ChatInvokeCompletion(completion=parsed, usage=None)
except ModelProviderError:
raise
except Exception as e:
raise ModelProviderError(message=str(e), model=self.name) from e
+2 -2
View File
@@ -11,7 +11,7 @@ from openai.types.shared_params.reasoning_effort import ReasoningEffort
from openai.types.shared_params.response_format_json_schema import JSONSchema, ResponseFormatJSONSchema
from pydantic import BaseModel
from browser_use.llm.base import BaseChatModel
from browser_use.llm.base import BaseChatModel, is_reasoning_model
from browser_use.llm.exceptions import ModelOutputTruncatedError, ModelProviderError, ModelRateLimitError
from browser_use.llm.messages import BaseMessage
from browser_use.llm.openai.serializer import OpenAIMessageSerializer
@@ -186,7 +186,7 @@ class ChatOpenAI(BaseChatModel):
if self.service_tier is not None:
model_params['service_tier'] = self.service_tier
if self.reasoning_models and any(str(m).lower() in str(self.model).lower() for m in self.reasoning_models):
if is_reasoning_model(self.model, self.reasoning_models):
model_params['reasoning_effort'] = self.reasoning_effort
model_params.pop('temperature', None)
model_params.pop('frequency_penalty', None)
+7 -2
View File
@@ -45,11 +45,16 @@ class SchemaOptimizer:
skip_fields = ['additionalProperties', '$defs']
for key, value in obj.items():
# Keys inside `properties` are user field names, not schema keywords.
if in_properties:
optimized[key] = optimize_schema(value, defs_lookup)
continue
if key in skip_fields:
continue
# Skip metadata "title" unless we're iterating inside an actual `properties` map
if key == 'title' and not in_properties:
# Skip metadata "title"
if key == 'title':
continue
# Preserve FULL descriptions without truncation, skip empty ones
+3 -5
View File
@@ -13,7 +13,7 @@ from openai.types.shared_params.response_format_json_schema import (
)
from pydantic import BaseModel
from browser_use.llm.base import BaseChatModel
from browser_use.llm.base import BaseChatModel, is_reasoning_model
from browser_use.llm.exceptions import ModelProviderError, ModelRateLimitError
from browser_use.llm.messages import BaseMessage, ContentPartTextParam, SystemMessage
from browser_use.llm.schema import SchemaOptimizer
@@ -556,11 +556,9 @@ class ChatVercel(BaseChatModel):
else:
is_google_model = self.model.startswith('google/')
is_anthropic_model = self.model.startswith('anthropic/')
is_reasoning_model = self.reasoning_models and any(
str(pattern).lower() in str(self.model).lower() for pattern in self.reasoning_models
)
is_reasoning = is_reasoning_model(self.model, self.reasoning_models)
if is_google_model or is_anthropic_model or is_reasoning_model:
if is_google_model or is_anthropic_model or is_reasoning:
modified_messages = [m.model_copy(deep=True) for m in messages]
schema = SchemaOptimizer.create_gemini_optimized_schema(output_format)
+51 -4
View File
@@ -1,6 +1,24 @@
---
name: browser-use
description: "Direct browser control via CDP for web interaction: automation, scraping, testing, screenshots, and site/app work."
homepage: https://browser-use.com
metadata:
{
"openclaw":
{
"requires": { "bins": ["browser-use"] },
"install":
[
{
"id": "uv",
"kind": "uv",
"package": "browser-use",
"bins": ["browser-use"],
"label": "Install Browser Use CLI (uv)",
},
],
},
}
---
# Browser Use
@@ -25,7 +43,22 @@ PY
- Invoke as `browser-use`. Use heredocs for multi-line commands.
- Helpers are pre-imported. `run.py` calls `ensure_daemon()` before `exec`.
- First navigation is `new_tab(url)`, not `goto_url(url)`.
- First navigation for a task is `new_tab(url)`, not `goto_url(url)`. The daemon
preserves the attached tab across separate CLI invocations, so do not call
`new_tab()` again in every script.
- Keep one working tab per task/site. Before opening another, inspect
`current_tab()` and `list_tabs()` and use `switch_tab()` to reuse a matching
tab. Do not leave duplicate tabs on the same URL or close tabs you did not
create.
- `new_tab()` and `switch_tab()` attach and move the horse marker without
changing Chrome's visible tab. Screenshots and normal CDP input work in the
background; call `activate_tab(target)` only when the user explicitly asks
or a page demonstrably pauses rendering while hidden.
- A timed-out `scroll(...)` on an attached background tab is evidence that the
page needs to be visible. Call `activate_tab(current_tab())`, retry the same
scroll once, then re-read the scroll position. This visibly switches tabs,
so do not use it when the user has forbidden foreground changes. Do not
invent a `Runtime.evaluate` scroll replacement or a cross-frame JS walker.
- The normal local flow attaches to the running Chrome/Chromium CDP endpoint. No browser ids or local profile selection.
## Local Chrome
@@ -36,7 +69,7 @@ If the daemon cannot connect, run diagnostics:
browser-use --doctor
```
If Chrome is not running at all, the harness launches it automatically and retries — no user action needed beyond clicking Allow if a permission popup appears.
If Chrome is not running at all, the harness launches it automatically and retries.
If Chrome is running but remote debugging is not enabled, the harness opens:
@@ -44,7 +77,14 @@ If Chrome is running but remote debugging is not enabled, the harness opens:
chrome://inspect/#remote-debugging
```
Ask the user to tick "Allow remote debugging for this browser instance" and click Allow if Chrome shows a permission popup. Then retry the same `browser-use` command.
On macOS, when Chrome asks for remote-debugging permission, run:
```text
browser-use mac-approve
```
Continue browser work when it returns `ready`; otherwise follow its printed
instruction.
## Remote Browsers
@@ -155,12 +195,19 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro
- Coordinate clicks default. CDP mouse events pass through iframes/shadow/cross-origin at the compositor level.
- Keep the connection model simple: use the default daemon, `BU_NAME`, `BU_CDP_URL`, `BU_CDP_WS`, or `start_remote_daemon(...)`.
- Trusted orchestrators can set `BH_OPEN_LIVE_URL=0` while provisioning a Cloud
daemon to keep its interactive live-view URL from being printed or opened.
The URL is still created and returned by `start_remote_daemon()`; callers must
avoid logging or serializing that returned field.
- Trusted orchestrators that already provisioned an exact named daemon can set
`BH_REQUIRE_EXISTING_DAEMON=1`. Each CLI call then health-checks and reuses
that daemon or fails closed; it never auto-starts or discovers another Chrome.
- Core helpers stay short. Put task-specific helper additions in `$BH_AGENT_WORKSPACE/agent_helpers.py`.
## Gotchas
- `chrome://inspect/#remote-debugging` must be enabled for local Chrome control.
- Chrome may show an "Allow remote debugging?" popup; wait for the user to click Allow. Do not retry in a loop — Chrome pops a fresh dialog for every new connection, and the daemon's single held connection is what makes this a one-time click.
- On macOS, if Chrome shows an "Allow remote debugging?" popup, run `browser-use mac-approve`. Do not poll in a loop — the daemon holds one connection.
- Omnibox popups are not real work tabs.
- CDP target order is not Chrome's visible tab-strip order.
- `BU_CDP_URL` is an HTTP DevTools endpoint; the daemon resolves it to WebSocket.
+26
View File
@@ -6,6 +6,28 @@ import re
from importlib import resources
from pathlib import Path
# Browser Use-only frontmatter added while generating both checked-in SKILL.md copies.
# Keep this as the source of truth; scripts/sync_browser_harness_skill.py verifies the outputs.
OPENCLAW_METADATA_LINES = (
'metadata:',
' {',
' "openclaw":',
' {',
' "requires": { "bins": ["browser-use"] },',
' "install":',
' [',
' {',
' "id": "uv",',
' "kind": "uv",',
' "package": "browser-use",',
' "bins": ["browser-use"],',
' "label": "Install Browser Use CLI (uv)",',
' },',
' ],',
' },',
' }',
)
def as_browser_use_skill(text: str) -> str:
"""Expose the Browser Harness skill under the Browser Use skill identity."""
@@ -39,6 +61,10 @@ def as_browser_use_skill(text: str) -> str:
1,
'description: "Direct browser control via CDP for web interaction: automation, scraping, testing, screenshots, and site/app work."',
)
if not any(line.startswith('homepage:') for line in lines):
lines.append('homepage: https://browser-use.com')
if not any(line.startswith('metadata:') for line in lines):
lines.extend(OPENCLAW_METADATA_LINES)
body = body.replace('# browser-harness', '# Browser Use', 1).replace('# Browser Harness', '# Browser Use', 1)
# Rebrand every mention except repo URLs (github.com/browser-use/browser-harness/...)
+1
View File
@@ -29,6 +29,7 @@ TARGET_DIR_BUILDERS = {
'copilot': lambda: _home_skill_dir('copilot'),
'cursor': lambda: _home_skill_dir('cursor'),
'gemini': lambda: _home_skill_dir('gemini'),
'openclaw': lambda: _home_skill_dir('openclaw'),
'opencode': lambda: _xdg_config_home() / 'opencode' / 'skills' / SKILL_NAME,
}
+12 -8
View File
@@ -9,19 +9,20 @@ from typing import Any
# Custom model pricing data
# Format matches LiteLLM's model_prices_and_context_window.json structure
CUSTOM_MODEL_PRICING: dict[str, dict[str, Any]] = {
'bu-1-0': {
'input_cost_per_token': 0.2 / 1_000_000, # $0.20 per 1M tokens
'output_cost_per_token': 2.00 / 1_000_000, # $2.00 per 1M tokens
'cache_read_input_token_cost': 0.02 / 1_000_000, # $0.02 per 1M tokens
'bu-2-0': {
'input_cost_per_token': 0.60 / 1_000_000, # $0.60 per 1M tokens
'output_cost_per_token': 3.50 / 1_000_000, # $3.50 per 1M tokens
'cache_read_input_token_cost': 0.06 / 1_000_000, # $0.06 per 1M tokens
'cache_creation_input_token_cost': None, # Not specified
'max_tokens': None, # Not specified
'max_input_tokens': None, # Not specified
'max_output_tokens': None, # Not specified
},
'bu-2-0': {
'input_cost_per_token': 0.60 / 1_000_000, # $0.60 per 1M tokens
'output_cost_per_token': 3.50 / 1_000_000, # $3.50 per 1M tokens
'cache_read_input_token_cost': 0.06 / 1_000_000, # $0.06 per 1M tokens
'bu-2-0-mini-preview': {
'input_cost_per_token': 0.15 / 1_000_000, # $0.15 per 1M tokens
'output_cost_per_token': 1.50 / 1_000_000, # $1.50 per 1M tokens
# No cache discount on this model: cached reads bill at the input rate.
'cache_read_input_token_cost': 0.15 / 1_000_000, # $0.15 per 1M tokens
'cache_creation_input_token_cost': None, # Not specified
'max_tokens': None, # Not specified
'max_input_tokens': None, # Not specified
@@ -90,4 +91,7 @@ CUSTOM_MODEL_PRICING: dict[str, dict[str, Any]] = {
}
CUSTOM_MODEL_PRICING['bu-latest'] = CUSTOM_MODEL_PRICING['bu-2-0']
# bu-1-0 is redirected to bu-2-0 at the gateway, so it bills at bu-2-0 rates.
CUSTOM_MODEL_PRICING['bu-1-0'] = CUSTOM_MODEL_PRICING['bu-2-0']
CUSTOM_MODEL_PRICING['smart'] = CUSTOM_MODEL_PRICING['bu-2-0']
+3 -1
View File
@@ -464,7 +464,7 @@ class Registry(Generic[Context]):
# Filter out empty values
applicable_secrets = {k: v for k, v in applicable_secrets.items() if v}
def recursively_replace_secrets(value: str | dict | list) -> str | dict | list:
def recursively_replace_secrets(value: str | dict | list | tuple) -> str | dict | list | tuple:
if isinstance(value, str):
# 1. Handle tagged secrets: <secret>label</secret>
matches = secret_pattern.findall(value)
@@ -499,6 +499,8 @@ class Registry(Generic[Context]):
return {k: recursively_replace_secrets(v) for k, v in value.items()}
elif isinstance(value, list):
return [recursively_replace_secrets(v) for v in value]
elif isinstance(value, tuple):
return tuple(recursively_replace_secrets(v) for v in value)
return value
params_dump = params.model_dump()
+33 -1
View File
@@ -20,6 +20,7 @@ load_dotenv()
# Pre-compiled regex for URL detection - used in URL shortening
URL_PATTERN = re.compile(r'https?://[^\s<>"\']+|www\.[^\s<>"\']+|[^\s<>"\']+\.[a-z]{2,}(?:/[^\s<>"\']*)?', re.IGNORECASE)
URL_NEGATION_PATTERN = re.compile(r"\b(?:never|not|don['\u2019]?t)\b", re.IGNORECASE)
logger = logging.getLogger(__name__)
@@ -39,6 +40,10 @@ def is_placeholder_url(url: str) -> bool:
return len(labels) >= 2 and all(re.fullmatch(r'x+', label) for label in labels)
_TRAILING_PROSE_PUNCTUATION = frozenset('.,;:!?([')
_CLOSING_TO_OPENING_BRACKET = {')': '(', ']': '['}
def sanitize_url_candidate(url: str) -> str:
"""Normalize a URL candidate captured from prose before auto-navigation."""
candidate = url.strip()
@@ -46,7 +51,34 @@ def sanitize_url_candidate(url: str) -> str:
# "https://example.com/search.\\n2. Next step". Those are task text,
# not part of the URL.
candidate = re.split(r'\\[nrt]', candidate, maxsplit=1)[0]
return re.sub(r'[.,;:!?()\[\]]+$', '', candidate)
# Strip trailing prose punctuation, but keep a closing bracket the URL opened
# itself, e.g. /wiki/Python_(programming_language). A closing bracket is only
# prose when it has no opener inside the candidate, as in "(see https://x.com/a)".
# Bracket totals are counted once and decremented as characters are trimmed, so
# a candidate ending in many brackets stays linear.
bracket_counts = {bracket: candidate.count(bracket) for bracket in '()[]'}
end = len(candidate)
while end:
last_char = candidate[end - 1]
if last_char in _TRAILING_PROSE_PUNCTUATION:
if last_char in bracket_counts:
bracket_counts[last_char] -= 1
end -= 1
continue
opening_bracket = _CLOSING_TO_OPENING_BRACKET.get(last_char)
if opening_bracket is not None and bracket_counts[last_char] > bracket_counts[opening_bracket]:
bracket_counts[last_char] -= 1
end -= 1
continue
break
return candidate[:end]
def has_url_negation(context: str) -> bool:
"""Return whether nearby prose explicitly negates navigation to a URL."""
return URL_NEGATION_PATTERN.search(context) is not None
# Lazy import for error types
+2 -2
View File
@@ -21,7 +21,7 @@ async def basic():
agent = Agent(
task='Go to github.com/browser-use/browser-use and tell me the star count',
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
browser=browser,
)
@@ -39,7 +39,7 @@ async def full_config():
agent = Agent(
task='go and check my ip address and the location',
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
browser=browser,
)
+1 -1
View File
@@ -90,7 +90,7 @@ async def main():
'Open https://httpbin.org/headers in two different tabs and extract the full JSON response. '
'Look for the custom headers X-Custom-Auth, X-Request-Source, and X-Trace-Id in the output and compare the results.'
),
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
browser=browser,
)
+43 -160
View File
@@ -1,187 +1,70 @@
"""
Cloud Example 1: Your First Browser Use Cloud Task
==================================================
This example demonstrates the most basic Browser Use Cloud functionality:
- Create a simple automation task
- Get the task ID
- Monitor completion
- Retrieve results
Perfect for first-time cloud users to understand the API basics.
Cost: ~$0.04 (1 task + 3 steps with GPT-4.1 mini)
"""
"""Run one Browser Use Cloud API V4 task."""
import os
import time
from math import isfinite
from typing import Any
import requests
from requests.exceptions import RequestException
from dotenv import load_dotenv
# Configuration
load_dotenv()
API_URL = os.getenv('BROWSER_USE_API_URL', 'https://api.browser-use.com/api/v4').rstrip('/')
API_KEY = os.getenv('BROWSER_USE_API_KEY')
if not API_KEY:
raise ValueError(
'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key'
)
raise RuntimeError('Set BROWSER_USE_API_KEY or add it to .env')
BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1')
TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30'))
HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'}
try:
RUN_TIMEOUT_SECONDS = float(os.getenv('BROWSER_USE_RUN_TIMEOUT', '900'))
except ValueError as error:
raise RuntimeError('BROWSER_USE_RUN_TIMEOUT must be a positive number of seconds') from error
if not isfinite(RUN_TIMEOUT_SECONDS) or RUN_TIMEOUT_SECONDS <= 0:
raise RuntimeError('BROWSER_USE_RUN_TIMEOUT must be a positive number of seconds')
HEADERS = {'X-Browser-Use-API-Key': API_KEY}
TERMINAL_STATUSES = {'completed', 'failed', 'cancelled'}
def _request_with_retry(method: str, url: str, **kwargs) -> requests.Response:
"""Make HTTP request with timeout and retry logic."""
kwargs.setdefault('timeout', TIMEOUT)
for attempt in range(3):
try:
response = requests.request(method, url, **kwargs)
response.raise_for_status()
return response
except RequestException as e:
if attempt == 2: # Last attempt
raise
sleep_time = 2**attempt
print(f'⚠️ Request failed (attempt {attempt + 1}/3), retrying in {sleep_time}s: {e}')
time.sleep(sleep_time)
# This line should never be reached, but satisfies type checker
raise RuntimeError('Unexpected error in retry logic')
def create_run(task: str) -> str:
response = requests.post(f'{API_URL}/runs', headers=HEADERS, json={'task': task}, timeout=30)
response.raise_for_status()
return response.json()['id']
def create_task(instructions: str) -> str:
"""
Create a new browser automation task.
Args:
instructions: Natural language description of what the agent should do
Returns:
task_id: Unique identifier for the created task
"""
print(f'📝 Creating task: {instructions}')
payload = {
'task': instructions,
'llm_model': 'gpt-4.1-mini', # Cost-effective model
'max_agent_steps': 10, # Prevent runaway costs
'enable_public_share': True, # Enable shareable execution URLs
}
response = _request_with_retry('post', f'{BASE_URL}/run-task', headers=HEADERS, json=payload)
task_id = response.json()['id']
print(f'✅ Task created with ID: {task_id}')
return task_id
def get_task_status(task_id: str) -> dict[str, Any]:
"""Get the current status of a task."""
response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}/status', headers=HEADERS)
return response.json()
def get_task_details(task_id: str) -> dict[str, Any]:
"""Get full task details including steps and output."""
response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS)
return response.json()
def wait_for_completion(task_id: str, poll_interval: int = 3) -> dict[str, Any]:
"""
Wait for task completion and show progress.
Args:
task_id: The task to monitor
poll_interval: How often to check status (seconds)
Returns:
Complete task details with output
"""
print(f'⏳ Monitoring task {task_id}...')
step_count = 0
start_time = time.time()
def wait_for_run(run_id: str, poll_seconds: float = 2, timeout_seconds: float = RUN_TIMEOUT_SECONDS) -> dict[str, Any]:
deadline = time.monotonic() + timeout_seconds
while True:
details = get_task_details(task_id)
status = details['status']
current_steps = len(details.get('steps', []))
elapsed = time.time() - start_time
response = requests.get(f'{API_URL}/runs/{run_id}/status', headers=HEADERS, timeout=30)
response.raise_for_status()
status = response.json()['status']
print(f'Status: {status}')
# Clear line and show current progress
if current_steps > step_count:
step_count = current_steps
if status in TERMINAL_STATUSES:
break
# Build status message
if status == 'running':
if current_steps > 0:
status_msg = f'🔄 Step {current_steps} | ⏱️ {elapsed:.0f}s | 🤖 Agent working...'
else:
status_msg = f'🤖 Agent starting... | ⏱️ {elapsed:.0f}s'
else:
status_msg = f'🔄 Step {current_steps} | ⏱️ {elapsed:.0f}s | Status: {status}'
if time.monotonic() >= deadline:
response = requests.post(f'{API_URL}/runs/{run_id}/cancel', headers=HEADERS, timeout=30)
response.raise_for_status()
raise TimeoutError(f'Cancelled run {run_id} after {timeout_seconds:g} seconds')
# Clear line and print status
print(f'\r{status_msg:<80}', end='', flush=True)
time.sleep(poll_seconds)
# Check if finished
if status == 'finished':
print(f'\r✅ Task completed successfully! ({current_steps} steps in {elapsed:.1f}s)' + ' ' * 20)
return details
elif status in ['failed', 'stopped']:
print(f'\r❌ Task {status} after {current_steps} steps' + ' ' * 30)
return details
time.sleep(poll_interval)
response = requests.get(f'{API_URL}/runs/{run_id}', headers=HEADERS, timeout=30)
response.raise_for_status()
return response.json()
def main():
"""Run a basic cloud automation task."""
print('🚀 Browser Use Cloud - Basic Task Example')
print('=' * 50)
def main() -> None:
run_id = create_run('Find the top story on Hacker News and summarize it in one sentence.')
print(f'Run: {run_id}')
# Define a simple search task (using DuckDuckGo to avoid captchas)
task_description = (
"Go to DuckDuckGo and search for 'browser automation tools'. Tell me the top 3 results with their titles and URLs."
)
run = wait_for_run(run_id)
if run['status'] != 'completed':
raise RuntimeError(run.get('error') or f'Run {run["status"]}')
try:
# Step 1: Create the task
task_id = create_task(task_description)
# Step 2: Wait for completion
result = wait_for_completion(task_id)
# Step 3: Display results
print('\n📊 Results:')
print('-' * 30)
print(f'Status: {result["status"]}')
print(f'Steps taken: {len(result.get("steps", []))}')
if result.get('output'):
print(f'Output: {result["output"]}')
else:
print('No output available')
# Show share URLs for viewing execution
if result.get('live_url'):
print(f'\n🔗 Live Preview: {result["live_url"]}')
if result.get('public_share_url'):
print(f'🌐 Share URL: {result["public_share_url"]}')
elif result.get('share_url'):
print(f'🌐 Share URL: {result["share_url"]}')
if not result.get('live_url') and not result.get('public_share_url') and not result.get('share_url'):
print("\n💡 Tip: Add 'enable_public_share': True to task payload to get shareable URLs")
except requests.exceptions.RequestException as e:
print(f'❌ API Error: {e}')
except Exception as e:
print(f'❌ Error: {e}')
print(f'Result: {run["result"]}')
print(f'Cost: ${run["totalCostUsd"]}')
if __name__ == '__main__':
-265
View File
@@ -1,265 +0,0 @@
"""
Cloud Example 2: Ultra-Fast Mode with Gemini Flash ⚡
====================================================
This example demonstrates the fastest and most cost-effective configuration:
- Gemini 2.5 Flash model ($0.01 per step)
- No proxy (faster execution, but no captcha solving)
- No element highlighting (better performance)
- Optimized viewport size
- Maximum speed configuration
Perfect for: Quick content generation, humor tasks, fast web scraping
Cost: ~$0.03 (1 task + 2-3 steps with Gemini Flash)
Speed: 2-3x faster than default configuration
Fun Factor: 💯 (Creates hilarious tech commentary)
"""
import argparse
import os
import time
from typing import Any
import requests
from requests.exceptions import RequestException
# Configuration
API_KEY = os.getenv('BROWSER_USE_API_KEY')
if not API_KEY:
raise ValueError(
'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key'
)
BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1')
TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30'))
HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'}
def _request_with_retry(method: str, url: str, **kwargs) -> requests.Response:
"""Make HTTP request with timeout and retry logic."""
kwargs.setdefault('timeout', TIMEOUT)
for attempt in range(3):
try:
response = requests.request(method, url, **kwargs)
response.raise_for_status()
return response
except RequestException as e:
if attempt == 2: # Last attempt
raise
sleep_time = 2**attempt
print(f'⚠️ Request failed (attempt {attempt + 1}/3), retrying in {sleep_time}s: {e}')
time.sleep(sleep_time)
raise RuntimeError('Unexpected error in retry logic')
def create_fast_task(instructions: str) -> str:
"""
Create a browser automation task optimized for speed and cost.
Args:
instructions: Natural language description of what the agent should do
Returns:
task_id: Unique identifier for the created task
"""
print(f'⚡ Creating FAST task: {instructions}')
# Ultra-fast configuration
payload = {
'task': instructions,
# Model: Fastest and cheapest
'llm_model': 'gemini-2.5-flash',
# Performance optimizations
'use_proxy': False, # No proxy = faster execution
'highlight_elements': False, # No highlighting = better performance
'use_adblock': True, # Block ads for faster loading
# Viewport optimization (smaller = faster)
'browser_viewport_width': 1024,
'browser_viewport_height': 768,
# Cost control
'max_agent_steps': 25, # Reasonable limit for fast tasks
# Enable sharing for viewing execution
'enable_public_share': True, # Get shareable URLs
# Optional: Speed up with domain restrictions
# "allowed_domains": ["google.com", "*.google.com"]
}
response = _request_with_retry('post', f'{BASE_URL}/run-task', headers=HEADERS, json=payload)
task_id = response.json()['id']
print(f'✅ Fast task created with ID: {task_id}')
print('⚡ Configuration: Gemini Flash + No Proxy + No Highlighting')
return task_id
def monitor_fast_task(task_id: str) -> dict[str, Any]:
"""
Monitor task with optimized polling for fast execution.
Args:
task_id: The task to monitor
Returns:
Complete task details with output
"""
print(f'🚀 Fast monitoring task {task_id}...')
start_time = time.time()
step_count = 0
last_step_time = start_time
# Faster polling for quick tasks
poll_interval = 1 # Check every second for fast tasks
while True:
response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS)
details = response.json()
status = details['status']
# Show progress with timing
current_steps = len(details.get('steps', []))
elapsed = time.time() - start_time
# Build status message
if current_steps > step_count:
step_time = time.time() - last_step_time
last_step_time = time.time()
step_count = current_steps
step_msg = f'🔥 Step {current_steps} | ⚡ {step_time:.1f}s | Total: {elapsed:.1f}s'
else:
if status == 'running':
step_msg = f'🚀 Step {current_steps} | ⏱️ {elapsed:.1f}s | Fast processing...'
else:
step_msg = f'🚀 Step {current_steps} | ⏱️ {elapsed:.1f}s | Status: {status}'
# Clear line and show progress
print(f'\r{step_msg:<80}', end='', flush=True)
# Check completion
if status == 'finished':
total_time = time.time() - start_time
if current_steps > 0:
avg_msg = f'⚡ Average: {total_time / current_steps:.1f}s per step'
else:
avg_msg = '⚡ No steps recorded'
print(f'\r🏁 Task completed in {total_time:.1f}s! {avg_msg}' + ' ' * 20)
return details
elif status in ['failed', 'stopped']:
print(f'\r❌ Task {status} after {elapsed:.1f}s' + ' ' * 30)
return details
time.sleep(poll_interval)
def run_speed_comparison():
"""Run multiple tasks to compare speed vs accuracy."""
print('\n🏃‍♂️ Speed Comparison Demo')
print('=' * 40)
tasks = [
'Go to ProductHunt and roast the top product like a sarcastic tech reviewer',
'Visit Reddit r/ProgrammerHumor and summarize the top post as a dramatic news story',
"Check GitHub trending and write a conspiracy theory about why everyone's switching to Rust",
]
results = []
for i, task in enumerate(tasks, 1):
print(f'\n📝 Fast Task {i}/{len(tasks)}')
print(f'Task: {task}')
start = time.time()
task_id = create_fast_task(task)
result = monitor_fast_task(task_id)
end = time.time()
results.append(
{
'task': task,
'duration': end - start,
'steps': len(result.get('steps', [])),
'status': result['status'],
'output': result.get('output', '')[:100] + '...' if result.get('output') else 'No output',
}
)
# Summary
print('\n📊 Speed Summary')
print('=' * 50)
total_time = sum(r['duration'] for r in results)
total_steps = sum(r['steps'] for r in results)
for i, result in enumerate(results, 1):
print(f'Task {i}: {result["duration"]:.1f}s ({result["steps"]} steps) - {result["status"]}')
print(f'\n⚡ Total time: {total_time:.1f}s')
print(f'🔥 Average per task: {total_time / len(results):.1f}s')
if total_steps > 0:
print(f'💨 Average per step: {total_time / total_steps:.1f}s')
else:
print('💨 Average per step: N/A (no steps recorded)')
def main():
"""Demonstrate ultra-fast cloud automation."""
print('⚡ Browser Use Cloud - Ultra-Fast Mode with Gemini Flash')
print('=' * 60)
print('🎯 Configuration Benefits:')
print('• Gemini Flash: $0.01 per step (cheapest)')
print('• No proxy: 30% faster execution')
print('• No highlighting: Better performance')
print('• Optimized viewport: Faster rendering')
try:
# Single fast task
print('\n🚀 Single Fast Task Demo')
print('-' * 30)
task = """
Go to Hacker News (news.ycombinator.com) and get the top 3 articles from the front page.
Then, write a funny tech news segment in the style of Fireship YouTube channel:
- Be sarcastic and witty about tech trends
- Use developer humor and memes
- Make fun of common programming struggles
- Include phrases like "And yes, it runs on JavaScript" or "Plot twist: it's written in Rust"
- Keep it under 250 words but make it entertaining
- Structure it like a news anchor delivering breaking tech news
Make each story sound dramatic but also hilarious, like you're reporting on the most important events in human history.
"""
task_id = create_fast_task(task)
result = monitor_fast_task(task_id)
print(f'\n📊 Result: {result.get("output", "No output")}')
# Show execution URLs
if result.get('live_url'):
print(f'\n🔗 Live Preview: {result["live_url"]}')
if result.get('public_share_url'):
print(f'🌐 Share URL: {result["public_share_url"]}')
elif result.get('share_url'):
print(f'🌐 Share URL: {result["share_url"]}')
# Optional: Run speed comparison with --compare flag
parser = argparse.ArgumentParser(description='Fast mode demo with Gemini Flash')
parser.add_argument('--compare', action='store_true', help='Run speed comparison with 3 tasks')
args = parser.parse_args()
if args.compare:
print('\n🏃‍♂️ Running speed comparison...')
run_speed_comparison()
except requests.exceptions.RequestException as e:
print(f'❌ API Error: {e}')
except Exception as e:
print(f'❌ Error: {e}')
if __name__ == '__main__':
main()
-362
View File
@@ -1,362 +0,0 @@
"""
Cloud Example 3: Structured JSON Output 📋
==========================================
This example demonstrates how to get structured, validated JSON output:
- Define Pydantic schemas for type safety
- Extract structured data from websites
- Validate and parse JSON responses
- Handle different data types and nested structures
Perfect for: Data extraction, API integration, structured analysis
Cost: ~$0.06 (1 task + 5-6 steps with GPT-4.1 mini)
"""
import argparse
import json
import os
import time
from typing import Any
import requests
from pydantic import BaseModel, Field, ValidationError
from requests.exceptions import RequestException
# Configuration
API_KEY = os.getenv('BROWSER_USE_API_KEY')
if not API_KEY:
raise ValueError(
'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key'
)
BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1')
TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30'))
HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'}
def _request_with_retry(method: str, url: str, **kwargs) -> requests.Response:
"""Make HTTP request with timeout and retry logic."""
kwargs.setdefault('timeout', TIMEOUT)
for attempt in range(3):
try:
response = requests.request(method, url, **kwargs)
response.raise_for_status()
return response
except RequestException as e:
if attempt == 2: # Last attempt
raise
sleep_time = 2**attempt
print(f'⚠️ Request failed (attempt {attempt + 1}/3), retrying in {sleep_time}s: {e}')
time.sleep(sleep_time)
raise RuntimeError('Unexpected error in retry logic')
# Define structured output schemas using Pydantic
class NewsArticle(BaseModel):
"""Schema for a news article."""
title: str = Field(description='The headline of the article')
summary: str = Field(description='Brief summary of the article')
url: str = Field(description='Direct link to the article')
published_date: str | None = Field(description='Publication date if available')
category: str | None = Field(description='Article category/section')
class NewsResponse(BaseModel):
"""Schema for multiple news articles."""
articles: list[NewsArticle] = Field(description='List of news articles')
source_website: str = Field(description='The website where articles were found')
extracted_at: str = Field(description='When the data was extracted')
class ProductInfo(BaseModel):
"""Schema for product information."""
name: str = Field(description='Product name')
price: float = Field(description='Product price in USD')
rating: float | None = Field(description='Average rating (0-5 scale)')
availability: str = Field(description='Stock status (in stock, out of stock, etc.)')
description: str = Field(description='Product description')
class CompanyInfo(BaseModel):
"""Schema for company information."""
name: str = Field(description='Company name')
stock_symbol: str | None = Field(description='Stock ticker symbol')
market_cap: str | None = Field(description='Market capitalization')
industry: str = Field(description='Primary industry')
headquarters: str = Field(description='Headquarters location')
founded_year: int | None = Field(description='Year founded')
def create_structured_task(instructions: str, schema_model: type[BaseModel], **kwargs) -> str:
"""
Create a task that returns structured JSON output.
Args:
instructions: Task description
schema_model: Pydantic model defining the expected output structure
**kwargs: Additional task parameters
Returns:
task_id: Unique identifier for the created task
"""
print(f'📝 Creating structured task: {instructions}')
print(f'🏗️ Expected schema: {schema_model.__name__}')
# Generate JSON schema from Pydantic model
json_schema = schema_model.model_json_schema()
payload = {
'task': instructions,
'structured_output_json': json.dumps(json_schema),
'llm_model': 'gpt-4.1-mini',
'max_agent_steps': 15,
'enable_public_share': True, # Enable shareable execution URLs
**kwargs,
}
response = _request_with_retry('post', f'{BASE_URL}/run-task', headers=HEADERS, json=payload)
task_id = response.json()['id']
print(f'✅ Structured task created: {task_id}')
return task_id
def wait_for_structured_completion(task_id: str, max_wait_time: int = 300) -> dict[str, Any]:
"""Wait for task completion and return the result."""
print(f'⏳ Waiting for structured output (max {max_wait_time}s)...')
start_time = time.time()
while True:
response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}/status', headers=HEADERS)
status = response.json()
elapsed = time.time() - start_time
# Check for timeout
if elapsed > max_wait_time:
print(f'\r⏰ Task timeout after {max_wait_time}s - stopping wait' + ' ' * 30)
# Get final details before timeout
details_response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS)
details = details_response.json()
return details
# Get step count from full details for better progress tracking
details_response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS)
details = details_response.json()
steps = len(details.get('steps', []))
# Build status message
if status == 'running':
status_msg = f'📋 Structured task | Step {steps} | ⏱️ {elapsed:.0f}s | 🔄 Extracting...'
else:
status_msg = f'📋 Structured task | Step {steps} | ⏱️ {elapsed:.0f}s | Status: {status}'
# Clear line and show status
print(f'\r{status_msg:<80}', end='', flush=True)
if status == 'finished':
print(f'\r✅ Structured data extracted! ({steps} steps in {elapsed:.1f}s)' + ' ' * 20)
return details
elif status in ['failed', 'stopped']:
print(f'\r❌ Task {status} after {steps} steps' + ' ' * 30)
return details
time.sleep(3)
def validate_and_display_output(output: str, schema_model: type[BaseModel]):
"""
Validate the JSON output against the schema and display results.
Args:
output: Raw JSON string from the task
schema_model: Pydantic model for validation
"""
print('\n📊 Structured Output Analysis')
print('=' * 40)
try:
# Parse and validate the JSON
parsed_data = schema_model.model_validate_json(output)
print('✅ JSON validation successful!')
# Pretty print the structured data
print('\n📋 Parsed Data:')
print('-' * 20)
print(parsed_data.model_dump_json(indent=2))
# Display specific fields based on model type
if isinstance(parsed_data, NewsResponse):
print(f'\n📰 Found {len(parsed_data.articles)} articles from {parsed_data.source_website}')
for i, article in enumerate(parsed_data.articles[:3], 1):
print(f'\n{i}. {article.title}')
print(f' Summary: {article.summary[:100]}...')
print(f' URL: {article.url}')
elif isinstance(parsed_data, ProductInfo):
print(f'\n🛍️ Product: {parsed_data.name}')
print(f' Price: ${parsed_data.price}')
print(f' Rating: {parsed_data.rating}/5' if parsed_data.rating else ' Rating: N/A')
print(f' Status: {parsed_data.availability}')
elif isinstance(parsed_data, CompanyInfo):
print(f'\n🏢 Company: {parsed_data.name}')
print(f' Industry: {parsed_data.industry}')
print(f' Headquarters: {parsed_data.headquarters}')
if parsed_data.founded_year:
print(f' Founded: {parsed_data.founded_year}')
return parsed_data
except ValidationError as e:
print('❌ JSON validation failed!')
print(f'Errors: {e}')
print(f'\nRaw output: {output[:500]}...')
return None
except json.JSONDecodeError as e:
print('❌ Invalid JSON format!')
print(f'Error: {e}')
print(f'\nRaw output: {output[:500]}...')
return None
def demo_news_extraction():
"""Demo: Extract structured news data."""
print('\n📰 Demo 1: News Article Extraction')
print('-' * 40)
task = """
Go to a major news website (like BBC, CNN, or Reuters) and extract information
about the top 3 news articles. For each article, get the title, summary, URL,
and any other available metadata.
"""
task_id = create_structured_task(task, NewsResponse)
result = wait_for_structured_completion(task_id)
if result.get('output'):
parsed_result = validate_and_display_output(result['output'], NewsResponse)
# Show execution URLs
if result.get('live_url'):
print(f'\n🔗 Live Preview: {result["live_url"]}')
if result.get('public_share_url'):
print(f'🌐 Share URL: {result["public_share_url"]}')
elif result.get('share_url'):
print(f'🌐 Share URL: {result["share_url"]}')
return parsed_result
else:
print('❌ No structured output received')
return None
def demo_product_extraction():
"""Demo: Extract structured product data."""
print('\n🛍️ Demo 2: Product Information Extraction')
print('-' * 40)
task = """
Go to Amazon and search for 'wireless headphones'. Find the first product result
and extract detailed information including name, price, rating, availability,
and description.
"""
task_id = create_structured_task(task, ProductInfo)
result = wait_for_structured_completion(task_id)
if result.get('output'):
parsed_result = validate_and_display_output(result['output'], ProductInfo)
# Show execution URLs
if result.get('live_url'):
print(f'\n🔗 Live Preview: {result["live_url"]}')
if result.get('public_share_url'):
print(f'🌐 Share URL: {result["public_share_url"]}')
elif result.get('share_url'):
print(f'🌐 Share URL: {result["share_url"]}')
return parsed_result
else:
print('❌ No structured output received')
return None
def demo_company_extraction():
"""Demo: Extract structured company data."""
print('\n🏢 Demo 3: Company Information Extraction')
print('-' * 40)
task = """
Go to a financial website and look up information about Apple Inc.
Extract company details including name, stock symbol, market cap,
industry, headquarters, and founding year.
"""
task_id = create_structured_task(task, CompanyInfo)
result = wait_for_structured_completion(task_id)
if result.get('output'):
parsed_result = validate_and_display_output(result['output'], CompanyInfo)
# Show execution URLs
if result.get('live_url'):
print(f'\n🔗 Live Preview: {result["live_url"]}')
if result.get('public_share_url'):
print(f'🌐 Share URL: {result["public_share_url"]}')
elif result.get('share_url'):
print(f'🌐 Share URL: {result["share_url"]}')
return parsed_result
else:
print('❌ No structured output received')
return None
def main():
"""Demonstrate structured output extraction."""
print('📋 Browser Use Cloud - Structured JSON Output')
print('=' * 50)
print('🎯 Features:')
print('• Type-safe Pydantic schemas')
print('• Automatic JSON validation')
print('• Structured data extraction')
print('• Multiple output formats')
try:
# Parse command line arguments
parser = argparse.ArgumentParser(description='Structured output extraction demo')
parser.add_argument('--demo', choices=['news', 'product', 'company', 'all'], default='news', help='Which demo to run')
args = parser.parse_args()
print(f'\n🔍 Running {args.demo} demo(s)...')
if args.demo == 'news':
demo_news_extraction()
elif args.demo == 'product':
demo_product_extraction()
elif args.demo == 'company':
demo_company_extraction()
elif args.demo == 'all':
demo_news_extraction()
demo_product_extraction()
demo_company_extraction()
except requests.exceptions.RequestException as e:
print(f'❌ API Error: {e}')
except Exception as e:
print(f'❌ Error: {e}')
if __name__ == '__main__':
main()
-331
View File
@@ -1,331 +0,0 @@
"""
Cloud Example 4: Proxy Usage 🌍
===============================
This example demonstrates reliable proxy usage scenarios:
- Different country proxies for geo-restrictions
- IP address and location verification
- Region-specific content access (streaming, news)
- Search result localization by country
- Mobile/residential proxy benefits
Perfect for: Geo-restricted content, location testing, regional analysis
Cost: ~$0.08 (1 task + 6-8 steps with proxy enabled)
"""
import argparse
import os
import time
from typing import Any
import requests
from requests.exceptions import RequestException
# Configuration
API_KEY = os.getenv('BROWSER_USE_API_KEY')
if not API_KEY:
raise ValueError(
'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key'
)
BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1')
TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30'))
HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'}
def _request_with_retry(method: str, url: str, **kwargs) -> requests.Response:
"""Make HTTP request with timeout and retry logic."""
kwargs.setdefault('timeout', TIMEOUT)
for attempt in range(3):
try:
response = requests.request(method, url, **kwargs)
response.raise_for_status()
return response
except RequestException as e:
if attempt == 2: # Last attempt
raise
sleep_time = 2**attempt
print(f'⚠️ Request failed (attempt {attempt + 1}/3), retrying in {sleep_time}s: {e}')
time.sleep(sleep_time)
raise RuntimeError('Unexpected error in retry logic')
def create_task_with_proxy(instructions: str, country_code: str = 'us') -> str:
"""
Create a task with proxy enabled from a specific country.
Args:
instructions: Task description
country_code: Proxy country ('us', 'fr', 'it', 'jp', 'au', 'de', 'fi', 'ca')
Returns:
task_id: Unique identifier for the created task
"""
print(f'🌍 Creating task with {country_code.upper()} proxy')
print(f'📝 Task: {instructions}')
payload = {
'task': instructions,
'llm_model': 'gpt-4.1-mini',
# Proxy configuration
'use_proxy': True, # Required for captcha solving
'proxy_country_code': country_code, # Choose proxy location
# Standard settings
'use_adblock': True, # Block ads for faster loading
'highlight_elements': True, # Keep highlighting for visibility
'max_agent_steps': 15,
# Enable sharing for viewing execution
'enable_public_share': True, # Get shareable URLs
}
response = _request_with_retry('post', f'{BASE_URL}/run-task', headers=HEADERS, json=payload)
task_id = response.json()['id']
print(f'✅ Task created with {country_code.upper()} proxy: {task_id}')
return task_id
def test_ip_location(country_code: str) -> dict[str, Any]:
"""Test IP address and location detection with proxy."""
task = """
Go to whatismyipaddress.com and tell me:
1. The detected IP address
2. The detected country/location
3. The ISP/organization
4. Any other location details shown
Please be specific about what you see on the page.
"""
task_id = create_task_with_proxy(task, country_code)
return wait_for_completion(task_id)
def test_geo_restricted_content(country_code: str) -> dict[str, Any]:
"""Test access to geo-restricted content."""
task = """
Go to a major news website (like BBC, CNN, or local news) and check:
1. What content is available
2. Any geo-restriction messages
3. Local/regional content differences
4. Language or currency preferences shown
Note any differences from what you might expect.
"""
task_id = create_task_with_proxy(task, country_code)
return wait_for_completion(task_id)
def test_streaming_service_access(country_code: str) -> dict[str, Any]:
"""Test access to region-specific streaming content."""
task = """
Go to a major streaming service website (like Netflix, YouTube, or BBC iPlayer)
and check what content or messaging appears.
Report:
1. What homepage content is shown
2. Any geo-restriction messages or content differences
3. Available content regions or language options
4. Any pricing or availability differences
Note: Don't try to log in, just observe the publicly available content.
"""
task_id = create_task_with_proxy(task, country_code)
return wait_for_completion(task_id)
def test_search_results_by_location(country_code: str) -> dict[str, Any]:
"""Test how search results vary by location."""
task = """
Go to Google and search for "best restaurants near me" or "local news".
Report:
1. What local results appear
2. The detected location in search results
3. Any location-specific content or ads
4. Language preferences
This will show how search results change based on proxy location.
"""
task_id = create_task_with_proxy(task, country_code)
return wait_for_completion(task_id)
def wait_for_completion(task_id: str) -> dict[str, Any]:
"""Wait for task completion and return results."""
print(f'⏳ Waiting for task {task_id} to complete...')
start_time = time.time()
while True:
response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS)
details = response.json()
status = details['status']
steps = len(details.get('steps', []))
elapsed = time.time() - start_time
# Build status message
if status == 'running':
status_msg = f'🌍 Proxy task | Step {steps} | ⏱️ {elapsed:.0f}s | 🤖 Processing...'
else:
status_msg = f'🌍 Proxy task | Step {steps} | ⏱️ {elapsed:.0f}s | Status: {status}'
# Clear line and show status
print(f'\r{status_msg:<80}', end='', flush=True)
if status == 'finished':
print(f'\r✅ Task completed in {steps} steps! ({elapsed:.1f}s total)' + ' ' * 20)
return details
elif status in ['failed', 'stopped']:
print(f'\r❌ Task {status} after {steps} steps' + ' ' * 30)
return details
time.sleep(3)
def demo_proxy_countries():
"""Demonstrate proxy usage across different countries."""
print('\n🌍 Demo 1: Proxy Countries Comparison')
print('-' * 45)
countries = [('us', 'United States'), ('de', 'Germany'), ('jp', 'Japan'), ('au', 'Australia')]
results = {}
for code, name in countries:
print(f'\n🌍 Testing {name} ({code.upper()}) proxy:')
print('=' * 40)
result = test_ip_location(code)
results[code] = result
if result.get('output'):
print(f'📍 Location Result: {result["output"][:200]}...')
# Show execution URLs
if result.get('live_url'):
print(f'🔗 Live Preview: {result["live_url"]}')
if result.get('public_share_url'):
print(f'🌐 Share URL: {result["public_share_url"]}')
elif result.get('share_url'):
print(f'🌐 Share URL: {result["share_url"]}')
print('-' * 40)
time.sleep(2) # Brief pause between tests
# Summary comparison
print('\n📊 Proxy Location Summary:')
print('=' * 30)
for code, result in results.items():
status = result.get('status', 'unknown')
print(f'{code.upper()}: {status}')
def demo_geo_restrictions():
"""Demonstrate geo-restriction bypass."""
print('\n🚫 Demo 2: Geo-Restriction Testing')
print('-' * 40)
# Test from different locations
locations = [('us', 'US content'), ('de', 'European content')]
for code, description in locations:
print(f'\n🌍 Testing {description} with {code.upper()} proxy:')
result = test_geo_restricted_content(code)
if result.get('output'):
print(f'📰 Content Access: {result["output"][:200]}...')
time.sleep(2)
def demo_streaming_access():
"""Demonstrate streaming service access with different proxies."""
print('\n📺 Demo 3: Streaming Service Access')
print('-' * 40)
locations = [('us', 'US'), ('de', 'Germany')]
for code, name in locations:
print(f'\n🌍 Testing streaming access from {name}:')
result = test_streaming_service_access(code)
if result.get('output'):
print(f'📺 Access Result: {result["output"][:200]}...')
time.sleep(2)
def demo_search_localization():
"""Demonstrate search result localization."""
print('\n🔍 Demo 4: Search Localization')
print('-' * 35)
locations = [('us', 'US'), ('de', 'Germany')]
for code, name in locations:
print(f'\n🌍 Testing search results from {name}:')
result = test_search_results_by_location(code)
if result.get('output'):
print(f'🔍 Search Results: {result["output"][:200]}...')
time.sleep(2)
def main():
"""Demonstrate comprehensive proxy usage."""
print('🌍 Browser Use Cloud - Proxy Usage Examples')
print('=' * 50)
print('🎯 Proxy Benefits:')
print('• Bypass geo-restrictions')
print('• Test location-specific content')
print('• Access region-locked websites')
print('• Mobile/residential IP addresses')
print('• Verify IP geolocation')
print('\n🌐 Available Countries:')
countries = ['🇺🇸 US', '🇫🇷 France', '🇮🇹 Italy', '🇯🇵 Japan', '🇦🇺 Australia', '🇩🇪 Germany', '🇫🇮 Finland', '🇨🇦 Canada']
print(' • '.join(countries))
try:
# Parse command line arguments
parser = argparse.ArgumentParser(description='Proxy usage examples')
parser.add_argument(
'--demo', choices=['countries', 'geo', 'streaming', 'search', 'all'], default='countries', help='Which demo to run'
)
args = parser.parse_args()
print(f'\n🔍 Running {args.demo} demo(s)...')
if args.demo == 'countries':
demo_proxy_countries()
elif args.demo == 'geo':
demo_geo_restrictions()
elif args.demo == 'streaming':
demo_streaming_access()
elif args.demo == 'search':
demo_search_localization()
elif args.demo == 'all':
demo_proxy_countries()
demo_geo_restrictions()
demo_streaming_access()
demo_search_localization()
except requests.exceptions.RequestException as e:
print(f'❌ API Error: {e}')
except Exception as e:
print(f'❌ Error: {e}')
if __name__ == '__main__':
main()
-355
View File
@@ -1,355 +0,0 @@
"""
Cloud Example 5: Search API (Beta) 🔍
=====================================
This example demonstrates the Browser Use Search API (BETA):
- Simple search: Search Google and extract from multiple results
- URL search: Extract specific content from a target URL
- Deep navigation through websites (depth parameter)
- Real-time content extraction vs cached results
Perfect for: Content extraction, research, competitive analysis
"""
import argparse
import asyncio
import json
import os
import time
from typing import Any
import aiohttp
# Configuration
API_KEY = os.getenv('BROWSER_USE_API_KEY')
if not API_KEY:
raise ValueError(
'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key'
)
BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1')
TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30'))
HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'}
async def simple_search(query: str, max_websites: int = 5, depth: int = 2) -> dict[str, Any]:
"""
Search Google and extract content from multiple top results.
Args:
query: Search query to process
max_websites: Number of websites to process (1-10)
depth: How deep to navigate (2-5)
Returns:
Dictionary with results from multiple websites
"""
# Validate input parameters
max_websites = max(1, min(max_websites, 10)) # Clamp to 1-10
depth = max(2, min(depth, 5)) # Clamp to 2-5
start_time = time.time()
print(f"🔍 Simple Search: '{query}'")
print(f'📊 Processing {max_websites} websites at depth {depth}')
print(f'💰 Estimated cost: {depth * max_websites}¢')
payload = {'query': query, 'max_websites': max_websites, 'depth': depth}
timeout = aiohttp.ClientTimeout(total=TIMEOUT)
connector = aiohttp.TCPConnector(limit=10) # Limit concurrent connections
async with aiohttp.ClientSession(timeout=timeout, connector=connector) as session:
async with session.post(f'{BASE_URL}/simple-search', json=payload, headers=HEADERS) as response:
elapsed = time.time() - start_time
if response.status == 200:
try:
result = await response.json()
print(f'✅ Found results from {len(result.get("results", []))} websites in {elapsed:.1f}s')
return result
except (aiohttp.ContentTypeError, json.JSONDecodeError) as e:
error_text = await response.text()
print(f'❌ Invalid JSON response: {e} (after {elapsed:.1f}s)')
return {'error': 'Invalid JSON', 'details': error_text}
else:
error_text = await response.text()
print(f'❌ Search failed: {response.status} - {error_text} (after {elapsed:.1f}s)')
return {'error': f'HTTP {response.status}', 'details': error_text}
async def search_url(url: str, query: str, depth: int = 2) -> dict[str, Any]:
"""
Extract specific content from a target URL.
Args:
url: Target URL to extract from
query: What specific content to look for
depth: How deep to navigate (2-5)
Returns:
Dictionary with extracted content
"""
# Validate input parameters
depth = max(2, min(depth, 5)) # Clamp to 2-5
start_time = time.time()
print(f'🎯 URL Search: {url}')
print(f"🔍 Looking for: '{query}'")
print(f'📊 Navigation depth: {depth}')
print(f'💰 Estimated cost: {depth}¢')
payload = {'url': url, 'query': query, 'depth': depth}
timeout = aiohttp.ClientTimeout(total=TIMEOUT)
connector = aiohttp.TCPConnector(limit=10) # Limit concurrent connections
async with aiohttp.ClientSession(timeout=timeout, connector=connector) as session:
async with session.post(f'{BASE_URL}/search-url', json=payload, headers=HEADERS) as response:
elapsed = time.time() - start_time
if response.status == 200:
try:
result = await response.json()
print(f'✅ Extracted content from {result.get("url", "website")} in {elapsed:.1f}s')
return result
except (aiohttp.ContentTypeError, json.JSONDecodeError) as e:
error_text = await response.text()
print(f'❌ Invalid JSON response: {e} (after {elapsed:.1f}s)')
return {'error': 'Invalid JSON', 'details': error_text}
else:
error_text = await response.text()
print(f'❌ URL search failed: {response.status} - {error_text} (after {elapsed:.1f}s)')
return {'error': f'HTTP {response.status}', 'details': error_text}
def display_simple_search_results(results: dict[str, Any]):
"""Display simple search results in a readable format."""
if 'error' in results:
print(f'❌ Error: {results["error"]}')
return
websites = results.get('results', [])
print(f'\n📋 Search Results ({len(websites)} websites)')
print('=' * 50)
for i, site in enumerate(websites, 1):
url = site.get('url', 'Unknown URL')
content = site.get('content', 'No content')
print(f'\n{i}. 🌐 {url}')
print('-' * 40)
# Show first 300 chars of content
if len(content) > 300:
print(f'{content[:300]}...')
print(f'[Content truncated - {len(content)} total characters]')
else:
print(content)
# Show execution URLs if available
if results.get('live_url'):
print(f'\n🔗 Live Preview: {results["live_url"]}')
if results.get('public_share_url'):
print(f'🌐 Share URL: {results["public_share_url"]}')
elif results.get('share_url'):
print(f'🌐 Share URL: {results["share_url"]}')
def display_url_search_results(results: dict[str, Any]):
"""Display URL search results in a readable format."""
if 'error' in results:
print(f'❌ Error: {results["error"]}')
return
url = results.get('url', 'Unknown URL')
content = results.get('content', 'No content')
print(f'\n📄 Extracted Content from: {url}')
print('=' * 60)
print(content)
# Show execution URLs if available
if results.get('live_url'):
print(f'\n🔗 Live Preview: {results["live_url"]}')
if results.get('public_share_url'):
print(f'🌐 Share URL: {results["public_share_url"]}')
elif results.get('share_url'):
print(f'🌐 Share URL: {results["share_url"]}')
async def demo_news_search():
"""Demo: Search for latest news across multiple sources."""
print('\n📰 Demo 1: Latest News Search')
print('-' * 35)
demo_start = time.time()
query = 'latest developments in artificial intelligence 2024'
results = await simple_search(query, max_websites=4, depth=2)
demo_elapsed = time.time() - demo_start
display_simple_search_results(results)
print(f'\n⏱️ Total demo time: {demo_elapsed:.1f}s')
return results
async def demo_competitive_analysis():
"""Demo: Analyze competitor websites."""
print('\n🏢 Demo 2: Competitive Analysis')
print('-' * 35)
query = 'browser automation tools comparison features pricing'
results = await simple_search(query, max_websites=3, depth=3)
display_simple_search_results(results)
return results
async def demo_deep_website_analysis():
"""Demo: Deep analysis of a specific website."""
print('\n🎯 Demo 3: Deep Website Analysis')
print('-' * 35)
demo_start = time.time()
url = 'https://docs.browser-use.com'
query = 'Browser Use features, pricing, and API capabilities'
results = await search_url(url, query, depth=3)
demo_elapsed = time.time() - demo_start
display_url_search_results(results)
print(f'\n⏱️ Total demo time: {demo_elapsed:.1f}s')
return results
async def demo_product_research():
"""Demo: Product research and comparison."""
print('\n🛍️ Demo 4: Product Research')
print('-' * 30)
query = 'best wireless headphones 2024 reviews comparison'
results = await simple_search(query, max_websites=5, depth=2)
display_simple_search_results(results)
return results
async def demo_real_time_vs_cached():
"""Demo: Show difference between real-time and cached results."""
print('\n⚡ Demo 5: Real-time vs Cached Data')
print('-' * 40)
print('🔄 Browser Use Search API benefits:')
print('• Actually browses websites like a human')
print('• Gets live, current data (not cached)')
print('• Navigates deep into sites via clicks')
print('• Handles JavaScript and dynamic content')
print('• Accesses pages requiring navigation')
# Example with live data
query = 'current Bitcoin price USD live'
results = await simple_search(query, max_websites=3, depth=2)
print('\n💰 Live Bitcoin Price Search Results:')
display_simple_search_results(results)
return results
async def demo_search_depth_comparison():
"""Demo: Compare different search depths."""
print('\n📊 Demo 6: Search Depth Comparison')
print('-' * 40)
url = 'https://news.ycombinator.com'
query = 'trending technology discussions'
depths = [2, 3, 4]
results = {}
for depth in depths:
print(f'\n🔍 Testing depth {depth}:')
result = await search_url(url, query, depth)
results[depth] = result
if 'content' in result:
content_length = len(result['content'])
print(f'📏 Content length: {content_length} characters')
# Brief pause between requests
await asyncio.sleep(1)
# Summary
print('\n📊 Depth Comparison Summary:')
print('-' * 30)
for depth, result in results.items():
if 'content' in result:
length = len(result['content'])
print(f'Depth {depth}: {length} characters')
else:
print(f'Depth {depth}: Error or no content')
return results
async def main():
"""Demonstrate comprehensive Search API usage."""
print('🔍 Browser Use Cloud - Search API (BETA)')
print('=' * 45)
print('⚠️ Note: This API is in BETA and may change')
print()
print('🎯 Search API Features:')
print('• Real-time website browsing (not cached)')
print('• Deep navigation through multiple pages')
print('• Dynamic content and JavaScript handling')
print('• Multiple result aggregation')
print('• Cost-effective content extraction')
print('\n💰 Pricing:')
print('• Simple Search: 1¢ × depth × websites')
print('• URL Search: 1¢ × depth')
print('• Example: depth=2, 5 websites = 10¢')
try:
# Parse command line arguments
parser = argparse.ArgumentParser(description='Search API (BETA) examples')
parser.add_argument(
'--demo',
choices=['news', 'competitive', 'deep', 'product', 'realtime', 'depth', 'all'],
default='news',
help='Which demo to run',
)
args = parser.parse_args()
print(f'\n🔍 Running {args.demo} demo(s)...')
if args.demo == 'news':
await demo_news_search()
elif args.demo == 'competitive':
await demo_competitive_analysis()
elif args.demo == 'deep':
await demo_deep_website_analysis()
elif args.demo == 'product':
await demo_product_research()
elif args.demo == 'realtime':
await demo_real_time_vs_cached()
elif args.demo == 'depth':
await demo_search_depth_comparison()
elif args.demo == 'all':
await demo_news_search()
await demo_competitive_analysis()
await demo_deep_website_analysis()
await demo_product_research()
await demo_real_time_vs_cached()
await demo_search_depth_comparison()
except aiohttp.ClientError as e:
print(f'❌ Network Error: {e}')
except Exception as e:
print(f'❌ Error: {e}')
if __name__ == '__main__':
asyncio.run(main())
+14 -123
View File
@@ -1,137 +1,28 @@
# Browser Use Cloud Examples 🚀
# Browser Use Cloud API V4
Welcome to the Browser Use Cloud examples! This folder contains progressively complex examples to help you get started with the Browser Use Cloud API quickly and efficiently.
Run one browser task through the current Cloud API. The example creates a run, polls the lightweight status endpoint, and fetches the result once the run is terminal. It cancels a run that exceeds the configurable 15-minute wait limit.
## 📋 Prerequisites
## Setup
1. **API Key**: Get your API key from [cloud.browser-use.com](https://cloud.browser-use.com/new-api-key)
2. **Python Environment**: Python 3.11+ with dependencies
3. **Environment Variables**: Configure your API settings
### Quick Setup
From the repository root:
```bash
# Create virtual environment and install dependencies (from project root)
uv venv --python 3.11
source .venv/bin/activate # On Windows: .venv\Scripts\activate
uv sync
# Set environment variables
export BROWSER_USE_API_KEY="your_api_key_here"
export BROWSER_USE_BASE_URL="https://api.browser-use.com/api/v1" # Optional
export BROWSER_USE_TIMEOUT="30" # Optional: request timeout in seconds
# Or use .env file (recommended)
cp examples/cloud/env.example .env
# Edit .env with your values
# Run examples from project root
python examples/cloud/01_basic_task.py
# Add your API key to .env
uv run python examples/cloud/01_basic_task.py
```
## 🎯 Examples Overview
Create an API key at [cloud.browser-use.com/new-api-key](https://cloud.browser-use.com/new-api-key).
### 🚀 Easy Cloud Setup Examples
## V4 request flow
- **[01_basic_task.py](./01_basic_task.py)** - Your first cloud task (start here!)
- **[02_fast_mode_gemini.py](./02_fast_mode_gemini.py)** - ⚡ Ultra-fast mode with Gemini Flash & Fireship humor
- **[03_structured_output.py](./03_structured_output.py)** - Get structured JSON responses
- **[04_proxy_usage.py](./04_proxy_usage.py)** - 🌍 Proxy for geo-restrictions & captcha solving
- **[05_search_api.py](./05_search_api.py)** - 🔍 Search API for content extraction (BETA)
The example uses the three endpoints needed for a basic run:
## 💰 Cost Optimization Tips
1. `POST /api/v4/runs`
2. `GET /api/v4/runs/{run_id}/status` until the run is `completed`, `failed`, or `cancelled`
3. `GET /api/v4/runs/{run_id}` for the result, error, and cost
1. **Use Gemini Flash** for fastest/cheapest execution ($0.01/step)
2. **Disable proxy** when not needed for captcha solving
3. **Disable element highlighting** for better performance
4. **Set max_agent_steps** to prevent runaway costs
5. **Use structured output** to reduce parsing overhead
6. **Add timeouts and retries** for reliability in production
7. **Use domain restrictions** when working with secrets
Authentication uses the `X-Browser-Use-API-Key` header. See the live [V4 OpenAPI specification](https://api.browser-use.com/api/v4/openapi.json) for optional models, browser settings, sessions, files, secrets, and judge settings.
## 🎨 Fast Mode Configuration
For maximum speed and cost efficiency:
```python
{
"llm_model": "gemini-2.5-flash",
"use_proxy": False,
"highlight_elements": False,
"use_adblock": True,
"max_agent_steps": 50
}
```
## 🔐 Security & Advanced Features
### Using Proxy
```python
{
"use_proxy": True,
"proxy_country_code": "us", # 'us', 'fr', 'it', 'jp', 'au', 'de', 'fi', 'ca'
}
```
### Passing Secrets Securely
```python
{
"secrets": {
"username": "your_username",
"password": "your_password",
"api_key": "your_api_key"
},
"allowed_domains": ["*.yoursite.com"] # Recommended with secrets
}
```
## 🔍 Search API (BETA)
The Search API extracts content by actually browsing websites (not cached results):
### Simple Search (Multi-site)
```python
# Cost: 1¢ × depth × websites
{
"query": "latest AI news",
"max_websites": 5,
"depth": 2
}
```
### URL Search (Single site)
```python
# Cost: 1¢ × depth
{
"url": "https://example.com",
"query": "pricing information",
"depth": 3
}
```
## 🔗 Quick Links
- [Cloud API Documentation](https://docs.browser-use.com/cloud)
- [API Reference](https://docs.browser-use.com/api-reference)
- [Pricing](https://cloud.browser-use.com/billing)
- [Discord Community](https://link.browser-use.com/discord)
## 🔧 Production Best Practices
- **Timeouts**: All examples include 30-second timeouts with retry logic
- **Error Handling**: Comprehensive error catching and status code validation
- **Security**: Use environment variables, domain restrictions with secrets
- **Reliability**: Built-in retries for network issues and rate limits
- **Automation**: CLI arguments instead of interactive prompts for CI/CD
## 🆘 Support
Need help?
- 📧 Email: support@browser-use.com
- 💬 Discord: [Join our community](https://link.browser-use.com/discord)
- 📖 Docs: <https://docs.browser-use.com>
---
**💡 Pro Tip**: Start with `01_basic_task.py` and work your way up. Each example builds on the previous ones!
Review usage and credits in [Cloud billing](https://cloud.browser-use.com/billing).
+5 -18
View File
@@ -1,21 +1,8 @@
# Browser Use Cloud API Configuration
# Copy this file to .env and fill in your values
# Required: Your Browser Use Cloud API key
# Get it from: https://cloud.browser-use.com/new-api-key
# Create a key at https://cloud.browser-use.com/new-api-key
BROWSER_USE_API_KEY=your_api_key_here
# Optional: Custom API base URL (for enterprise installations)
# BROWSER_USE_BASE_URL=https://api.browser-use.com/api/v1
# Optional: override the API root for an enterprise installation.
# BROWSER_USE_API_URL=https://api.browser-use.com/api/v4
# Optional: Default model preference
# BROWSER_USE_DEFAULT_MODEL=gemini-2.5-flash
# Optional: Cost limits
# BROWSER_USE_MAX_COST_PER_TASK=5.0
# Optional: Request timeout (seconds)
# BROWSER_USE_TIMEOUT=30
# Optional: Logging configuration
# LOG_LEVEL=INFO
# Optional: cancel a run if it has not finished after this many seconds.
# BROWSER_USE_RUN_TIMEOUT=900
+2
View File
@@ -261,6 +261,8 @@ async def openai_cua_fallback(params: OpenAICUAAction, browser_session: BrowserS
raise Exception('No computer calls found in CUA response')
action = computer_call.action
if action is None:
raise Exception('No action found in CUA computer call')
print(f'🎬 Executing CUA action: {action.type} - {action}')
action_result = await handle_model_action(browser_session, action)
+1 -1
View File
@@ -6,7 +6,7 @@ from browser_use import Agent, ChatBrowserUse
async def main() -> None:
agent = Agent(
task='Please find the latest commit on browser-use/browser-use repo and tell me the commit message. Please summarize what it is about.',
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
demo_mode=True,
)
await agent.run(max_steps=5)
+1 -1
View File
@@ -33,7 +33,7 @@ async def main():
'Create a CSV file called "top_cities.csv" with columns: rank, city name, country, population. '
'Make sure to include all cities even if some data is missing — leave those cells empty.'
),
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
)
history = await agent.run()
+1 -1
View File
@@ -27,7 +27,7 @@ Round your result to the nearest 1000 hours and do not use any comma separators
async def main():
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
agent = Agent(
task=task,
llm=llm,
+1 -1
View File
@@ -59,7 +59,7 @@ async def main():
# Example task to demonstrate history saving and rerunning
history_file = Path('agent_history.json')
task = 'Go to https://browser-use.github.io/stress-tests/challenges/reference-number-form.html and fill the form with example data and submit and extract the refernence number.'
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
# Optional: Use custom LLMs for AI features during rerun
# Uncomment to use a custom LLM:
+1 -1
View File
@@ -34,7 +34,7 @@ async def main():
'Go to https://news.ycombinator.com and save the front page as a PDF named "hackernews". '
'Then go to https://en.wikipedia.org/wiki/Web_browser and save just that article as a PDF in A4 format.'
),
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
)
history = await agent.run()
+1 -1
View File
@@ -19,7 +19,7 @@ from browser_use import Agent, ChatBrowserUse
async def main():
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
task = "Search Google for 'what is browser automation' and tell me the top 3 results"
agent = Agent(task=task, llm=llm)
await agent.run()
+1 -1
View File
@@ -30,7 +30,7 @@ from browser_use import Agent, ChatBrowserUse
async def main():
# Initialize the model
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
# Define a form filling task
task = """
@@ -30,7 +30,7 @@ from browser_use import Agent, ChatBrowserUse
async def main():
# Initialize the model
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
# Define a data extraction task
task = """
@@ -30,7 +30,7 @@ from browser_use import Agent, ChatBrowserUse
async def main():
# Initialize the model
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
# Define a multi-step task
task = """
+1 -1
View File
@@ -28,7 +28,7 @@ async def main():
tools = EmailTools(email_client=email_client, inbox=inbox)
# Initialize the LLM for browser-use agent
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
# Set your local browser path
browser = Browser(executable_path='/Applications/Google Chrome.app/Contents/MacOS/Google Chrome')
+7 -4
View File
@@ -20,12 +20,15 @@ if not os.getenv('BROWSER_USE_API_KEY'):
async def main():
# `bu-2-0` is the optimized default. ChatBrowserUse can also route to
# provider-prefixed models (e.g. 'anthropic/claude-sonnet-4-6', 'openai/gpt-5.5',
# 'google/gemini-3-pro') through the same gateway - see browser_use_provider_models.py.
# A bare `ChatBrowserUse()` gives you `bu-2-0`, the premium default (as does 'bu-latest').
# `bu-2-0-mini-preview`, used below, is cheaper and faster per token but is in preview, so
# you opt into it by name rather than getting it by default.
# ChatBrowserUse can also route to provider-prefixed models (e.g. 'anthropic/claude-sonnet-4-6',
# 'openai/gpt-5.5', 'google/gemini-3-pro') through the same gateway - see
# browser_use_provider_models.py.
agent = Agent(
task='Find the number of stars of the browser-use repo',
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
)
# Run the agent
+1 -1
View File
@@ -20,7 +20,7 @@ if deepseek_api_key is None:
async def main():
llm = ChatDeepSeek(
base_url='https://api.deepseek.com/v1',
model='deepseek-chat',
model='deepseek-v4-flash',
api_key=deepseek_api_key,
)
+1 -1
View File
@@ -37,7 +37,7 @@ async def pydantic_example(browser: Browser):
agent = Agent(
"""go and check my ip address and the location. return the result in json format""",
browser=browser,
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
)
res = await agent.run()
+1 -1
View File
@@ -29,7 +29,7 @@ async def get_ip_location(browser: Browser) -> AgentHistoryList:
agent = Agent(
task='Go to ipinfo.io and extract my IP address and location details (country, city, region)',
browser=browser,
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
output_model_schema=IPLocation,
)
return await agent.run(max_steps=10)
+1 -1
View File
@@ -12,6 +12,6 @@ load_dotenv()
agent = Agent(
task='Find the number of stars of the following repos: browser-use, playwright, stagehand, react, nextjs',
llm=ChatBrowserUse(model='bu-2-0'),
llm=ChatBrowserUse(model='bu-2-0-mini-preview'),
)
agent.run_sync()
+1 -1
View File
@@ -24,7 +24,7 @@ class GroceryCart(BaseModel):
async def add_to_cart(items: list[str] = ['milk', 'eggs', 'bread']):
browser = Browser(cdp_url='http://localhost:9222')
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
# Task prompt
task = f"""
+1 -1
View File
@@ -6,7 +6,7 @@ from browser_use import Agent, Browser, ChatBrowserUse, Tools
async def main():
browser = Browser(cdp_url='http://localhost:9222')
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
tools = Tools()
+1 -1
View File
@@ -34,7 +34,7 @@ async def find(item: str = 'Used iPhone 12'):
"""
browser = Browser(cdp_url='http://localhost:9222')
llm = ChatBrowserUse(model='bu-2-0')
llm = ChatBrowserUse(model='bu-2-0-mini-preview')
# Task prompt
task = f"""
+16 -16
View File
@@ -2,7 +2,7 @@
name = "browser-use"
description = "Make websites accessible for AI agents"
authors = [{ name = "Gregor Zunic" }]
version = "0.13.7"
version = "0.13.8"
readme = "README.md"
requires-python = ">=3.11,<4.0"
classifiers = [
@@ -14,14 +14,14 @@ dependencies = [
"aiohttp==3.14.3",
"anyio==4.12.1",
"bubus==1.5.6",
"click==8.3.1",
"click==8.3.3",
"InquirerPy==0.3.4",
"rich==14.3.1",
"rich==14.3.3",
"google-api-core==2.29.0",
"httpx==0.28.1",
"posthog==7.7.0",
"psutil==7.2.2",
"pydantic==2.12.5",
"pydantic>=2.12.5,<2.14",
"pyobjc==12.1; platform_system == 'darwin'",
"python-dotenv==1.2.2",
"requests==2.33.0",
@@ -29,24 +29,24 @@ dependencies = [
"typing-extensions==4.15.0",
"uuid7==0.1.0",
"google-genai==1.65.0",
"openai==2.16.0",
"openai==2.26.0",
"anthropic==0.76.0",
"groq==1.0.0",
"ollama==0.6.1",
"google-api-python-client==2.188.0",
"google-auth==2.48.0",
"google-auth-oauthlib==1.2.4",
"mcp==1.26.0",
"pypdf==6.14.2",
"mcp==1.28.1",
"pypdf==6.15.0",
"reportlab==4.4.9",
"cdp-use==1.4.5",
"pyotp==2.9.0",
"pillow==12.2.0",
"pillow==12.3.0",
"cloudpickle==3.1.2",
"markdownify==1.2.2",
"python-docx==1.2.0",
"browser-use-sdk==3.4.2",
"browser-harness==0.1.8",
"browser-harness==0.1.10",
]
# google-api-core: only used for Google LLM APIs
# pyperclip: only used for examples that use copy/paste
@@ -60,11 +60,11 @@ dependencies = [
[project.optional-dependencies]
cli = []
core = [
"browser-use-core==0.13.2; sys_platform == 'darwin' and platform_machine == 'arm64'",
"browser-use-core==0.13.2; sys_platform == 'darwin' and platform_machine == 'x86_64'",
"browser-use-core==0.13.2; sys_platform == 'linux' and platform_machine == 'x86_64'",
"browser-use-core==0.13.2; sys_platform == 'linux' and platform_machine == 'aarch64'",
"browser-use-core==0.13.2; sys_platform == 'win32' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
"browser-use-core==0.13.3; sys_platform == 'darwin' and platform_machine == 'arm64'",
"browser-use-core==0.13.3; sys_platform == 'darwin' and platform_machine == 'x86_64'",
"browser-use-core==0.13.3; sys_platform == 'linux' and platform_machine == 'x86_64'",
"browser-use-core==0.13.3; sys_platform == 'linux' and platform_machine == 'aarch64'",
"browser-use-core==0.13.3; sys_platform == 'win32' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')",
]
aws = ["boto3==1.42.37"]
oci = ["oci==2.166.0"]
@@ -76,13 +76,13 @@ examples = [
"imgcat==0.6.0",
# "stagehand-py>=0.3.6",
# "browserbase>=0.4.0",
"langchain-openai==1.1.7",
"langchain-openai==1.1.14",
]
eval = [
"lmnr[all]==0.7.42",
"anyio==4.12.1",
"psutil==7.2.2",
"datamodel-code-generator==0.53.0",
"datamodel-code-generator==0.75.1",
]
cli-oci = ["browser-use[cli,oci]"]
all = ["browser-use[cli,examples,aws,oci]"]
+51 -4
View File
@@ -1,6 +1,24 @@
---
name: browser-use
description: "Direct browser control via CDP for web interaction: automation, scraping, testing, screenshots, and site/app work."
homepage: https://browser-use.com
metadata:
{
"openclaw":
{
"requires": { "bins": ["browser-use"] },
"install":
[
{
"id": "uv",
"kind": "uv",
"package": "browser-use",
"bins": ["browser-use"],
"label": "Install Browser Use CLI (uv)",
},
],
},
}
---
# Browser Use
@@ -25,7 +43,22 @@ PY
- Invoke as `browser-use`. Use heredocs for multi-line commands.
- Helpers are pre-imported. `run.py` calls `ensure_daemon()` before `exec`.
- First navigation is `new_tab(url)`, not `goto_url(url)`.
- First navigation for a task is `new_tab(url)`, not `goto_url(url)`. The daemon
preserves the attached tab across separate CLI invocations, so do not call
`new_tab()` again in every script.
- Keep one working tab per task/site. Before opening another, inspect
`current_tab()` and `list_tabs()` and use `switch_tab()` to reuse a matching
tab. Do not leave duplicate tabs on the same URL or close tabs you did not
create.
- `new_tab()` and `switch_tab()` attach and move the horse marker without
changing Chrome's visible tab. Screenshots and normal CDP input work in the
background; call `activate_tab(target)` only when the user explicitly asks
or a page demonstrably pauses rendering while hidden.
- A timed-out `scroll(...)` on an attached background tab is evidence that the
page needs to be visible. Call `activate_tab(current_tab())`, retry the same
scroll once, then re-read the scroll position. This visibly switches tabs,
so do not use it when the user has forbidden foreground changes. Do not
invent a `Runtime.evaluate` scroll replacement or a cross-frame JS walker.
- The normal local flow attaches to the running Chrome/Chromium CDP endpoint. No browser ids or local profile selection.
## Local Chrome
@@ -36,7 +69,7 @@ If the daemon cannot connect, run diagnostics:
browser-use --doctor
```
If Chrome is not running at all, the harness launches it automatically and retries — no user action needed beyond clicking Allow if a permission popup appears.
If Chrome is not running at all, the harness launches it automatically and retries.
If Chrome is running but remote debugging is not enabled, the harness opens:
@@ -44,7 +77,14 @@ If Chrome is running but remote debugging is not enabled, the harness opens:
chrome://inspect/#remote-debugging
```
Ask the user to tick "Allow remote debugging for this browser instance" and click Allow if Chrome shows a permission popup. Then retry the same `browser-use` command.
On macOS, when Chrome asks for remote-debugging permission, run:
```text
browser-use mac-approve
```
Continue browser work when it returns `ready`; otherwise follow its printed
instruction.
## Remote Browsers
@@ -155,12 +195,19 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro
- Coordinate clicks default. CDP mouse events pass through iframes/shadow/cross-origin at the compositor level.
- Keep the connection model simple: use the default daemon, `BU_NAME`, `BU_CDP_URL`, `BU_CDP_WS`, or `start_remote_daemon(...)`.
- Trusted orchestrators can set `BH_OPEN_LIVE_URL=0` while provisioning a Cloud
daemon to keep its interactive live-view URL from being printed or opened.
The URL is still created and returned by `start_remote_daemon()`; callers must
avoid logging or serializing that returned field.
- Trusted orchestrators that already provisioned an exact named daemon can set
`BH_REQUIRE_EXISTING_DAEMON=1`. Each CLI call then health-checks and reuses
that daemon or fails closed; it never auto-starts or discovers another Chrome.
- Core helpers stay short. Put task-specific helper additions in `$BH_AGENT_WORKSPACE/agent_helpers.py`.
## Gotchas
- `chrome://inspect/#remote-debugging` must be enabled for local Chrome control.
- Chrome may show an "Allow remote debugging?" popup; wait for the user to click Allow. Do not retry in a loop — Chrome pops a fresh dialog for every new connection, and the daemon's single held connection is what makes this a one-time click.
- On macOS, if Chrome shows an "Allow remote debugging?" popup, run `browser-use mac-approve`. Do not poll in a loop — the daemon holds one connection.
- Omnibox popups are not real work tabs.
- CDP target order is not Chrome's visible tab-strip order.
- `BU_CDP_URL` is an HTTP DevTools endpoint; the daemon resolves it to WebSocket.
+8 -3
View File
@@ -3,7 +3,7 @@ name: cloud
description: >
Documentation reference for using Browser Use Cloud — the hosted API
and SDK for browser automation. Use this skill whenever the user needs
help with the Cloud REST API (v2 or v3), browser-use-sdk (Python or
help with the Cloud REST API (v2, v3, or v4), browser-use-sdk (Python or
TypeScript), X-Browser-Use-API-Key authentication, cloud sessions,
browser profiles, profile sync, CDP WebSocket connections, stealth
browsers, residential proxies, CAPTCHA handling, webhooks, workspaces,
@@ -25,7 +25,8 @@ Read the relevant file based on what the user needs.
| Topic | Read |
|-------|------|
| Setup, first task, pricing, FAQ | `references/quickstart.md` |
| Current v4 setup, first run, sessions, workspaces, browsers | `references/api-v4.md` |
| Legacy v2 setup, pricing, FAQ | `references/quickstart.md` |
| v2 REST API: all 30 endpoints, cURL examples, schemas | `references/api-v2.md` |
| v3 BU Agent API: sessions, messages, files, workspaces | `references/api-v3.md` |
| Sessions, profiles, auth strategies, 1Password | `references/sessions.md` |
@@ -43,13 +44,17 @@ Read the relevant file based on what the user needs.
## Critical Notes
- Cloud API base URL: `https://api.browser-use.com/api/v2/` (v2) or `https://api.browser-use.com/api/v3` (v3)
- Use v4 for new hosted-agent integrations. Keep v2 or v3 only when maintaining an existing integration or using a resource not yet wrapped by the v4 SDK.
- Cloud API base URL: `https://api.browser-use.com/api/v2/` (v2), `https://api.browser-use.com/api/v3` (v3), or `https://api.browser-use.com/api/v4` (v4)
- Auth header: `X-Browser-Use-API-Key: <key>`
- Get API key: https://cloud.browser-use.com/new-api-key
- Set env var: `BROWSER_USE_API_KEY=<key>`
- Cloud SDK: `uv pip install browser-use-sdk` (Python) or `npm install browser-use-sdk` (TypeScript)
- Python v2: `from browser_use_sdk import AsyncBrowserUse`
- Python v3: `from browser_use_sdk.v3 import AsyncBrowserUse`
- Python v4: `from browser_use_sdk.v4 import BrowserUse` or `AsyncBrowserUse`
- TypeScript v2: `import { BrowserUse } from "browser-use-sdk"`
- TypeScript v3: `import { BrowserUse } from "browser-use-sdk/v3"`
- TypeScript v4: `import { BrowserUse } from "browser-use-sdk/v4"`
- Browser management is available at the v4 REST `/browsers` resource, but its SDK wrapper still uses the explicit v3 namespace. Always stop a browser explicitly; closing CDP does not stop billing.
- CDP WebSocket: `wss://connect.browser-use.com?apiKey=KEY&proxyCountryCode=us`
+134
View File
@@ -0,0 +1,134 @@
# API v4: Hosted Agent Runs
Use v4 for new hosted-agent integrations. A **run** is one agent turn, a
**session** is the conversation shared by follow-up runs, and a **workspace**
is the persistent filesystem that can be reused across sessions.
- REST base: `https://api.browser-use.com/api/v4`
- Auth header: `X-Browser-Use-API-Key: <key>`
- Python: `from browser_use_sdk.v4 import BrowserUse`
- TypeScript: `import { BrowserUse } from "browser-use-sdk/v4"`
## First Run
### Python
```python
from browser_use_sdk.v4 import BrowserUse
with BrowserUse() as client:
created = client.runs.create("Find the top Hacker News story")
run = client.runs.wait_for_completion(created.id)
print(run.result)
```
### TypeScript
```typescript
import { BrowserUse } from "browser-use-sdk/v4";
const client = new BrowserUse();
const created = await client.runs.create({
task: "Find the top Hacker News story",
});
const run = await client.runs.waitForCompletion(created.id);
console.log(run.result);
```
### REST
Create the run, poll the lightweight status route, then fetch the full result
only after the status is `completed`, `failed`, or `cancelled`:
```bash
curl -X POST https://api.browser-use.com/api/v4/runs \
-H "X-Browser-Use-API-Key: $BROWSER_USE_API_KEY" \
-H "Content-Type: application/json" \
-d '{"task":"Find the top Hacker News story"}'
curl https://api.browser-use.com/api/v4/runs/RUN_ID/status \
-H "X-Browser-Use-API-Key: $BROWSER_USE_API_KEY"
curl https://api.browser-use.com/api/v4/runs/RUN_ID \
-H "X-Browser-Use-API-Key: $BROWSER_USE_API_KEY"
```
Do not repeatedly poll the full run resource. The SDK wait helpers use the
status route and fetch the full run once at the end.
## Sessions and Follow-ups
Every new run implicitly creates a session. Reuse its session ID to continue
the same conversation:
```python
from browser_use_sdk.v4 import BrowserUse
with BrowserUse() as client:
first = client.runs.create("Open Hacker News")
client.runs.wait_for_completion(first.id)
follow_up = client.runs.create(
"Now summarize the top story",
session_id=first.session_id,
)
result = client.runs.wait_for_completion(follow_up.id)
```
For a busy session, queue a next turn with
`client.sessions.send_message(session_id, text)`. Pass `interrupt=True` only
when the active run should be cancelled so the new message can start. The REST
equivalent is `POST /sessions/{session_id}/queue` with `text` and optional
`interrupt`.
## Workspaces and Files
A workspace persists files independently of a session. Upload a local file,
then attach its returned file ID to a run:
```python
from browser_use_sdk.v4 import BrowserUse
with BrowserUse() as client:
workspace = client.workspaces.create(name="research")
uploaded = client.workspaces.upload(workspace.id, "people.csv")
run = client.runs.create(
"Read the CSV and save a report",
workspace_id=workspace.id,
attached_file_ids=[uploaded[0].id],
)
```
Attachments are run-scoped. Reusing a workspace does not automatically attach
every file in it. List generated files with `client.workspaces.files(workspace.id)`;
presigned download URLs expire after 60 seconds, so request them immediately
before downloading.
## Direct Browser Control
The v4 REST API can create a browser for direct CDP control:
1. `POST /browsers` returns the browser `id` (its session ID) and `cdpUrl`.
2. Connect Browser Use, Playwright, Puppeteer, or Selenium to `cdpUrl`.
3. `PATCH /browsers/{session_id}` with `{"action":"stop"}` stops the browser;
replace `session_id` with the returned `id`.
Closing a CDP client does not stop the cloud browser or its billing. The
browser-management SDK wrapper currently uses the explicit v3 namespace; use
`browser_use_sdk.v3` or `browser-use-sdk/v3` for that resource, or call the v4
REST endpoint directly.
## Resource Map
| Resource | Common operations |
|----------|-------------------|
| Runs | create, list, get, status, events, cancel, attachments |
| Sessions | list, get, queue messages, inspect/remove queued messages, purge |
| Workspaces | create, get, update, archive, size, upload/list/delete files |
| Browsers (REST) | create, inspect, stop |
For the complete current contract, use:
- Docs: https://docs.browser-use.com/cloud/api-v4
- OpenAPI: https://docs.browser-use.com/cloud/openapi/v4.json
@@ -30,7 +30,7 @@ Your agent already has tools (search, code execution, file I/O, etc.) and its ow
| Your agent type | Best approach | Control level |
|----------------|---------------|--------------|
| CLI coding agent in sandbox | [CLI commands](#shell-command-agents-cli) | Per-command |
| CLI coding agent in sandbox | [CLI 3.0 Python](#shell-command-agents-cli) | Per-call |
| TypeScript/JS | [CDP + Playwright](#typescriptjs-cdp--playwright) | Playwright API |
| MCP client (Claude Desktop, Cursor) | [Local MCP server](#mcp-native-agents) | MCP tools |
| Existing Playwright/Puppeteer/Selenium | [CDP WebSocket (stealth)](#existing-playwrightpuppeteerselenium) | Your existing API |
@@ -42,54 +42,54 @@ Your agent already has tools (search, code execution, file I/O, etc.) and its ow
**For:** Claude Code, Codex, OpenCode, Cline, Windsurf, Cursor background agents, Hermes, OpenClaw — any coding agent running in a VM/container with terminal access.
**Setup:** Install the CLI and load the browser-use SKILL.md into the agent's context. The agent calls browser commands as shell tool invocations.
**Setup:** Install the CLI and load the browser-use SKILL.md into the agent's context. CLI 3.0 runs Python from stdin. Browser helpers are already imported, and the browser stays alive between calls.
```bash
uv pip install 'browser-use[cli]'
```
**Core workflow** — the agent calls these commands one at a time, reading output between each:
For Browser Use Cloud, authenticate once and start a named remote browser:
```bash
# 1. Navigate
browser-use open https://example.com
browser-use auth login
# 2. Observe — ALWAYS run state first to get element indices
browser-use state
# Output: URL, title, list of clickable elements with indices
# e.g. [0] <input type="search" placeholder="Search...">
# [1] <button>Submit</button>
# [2] <a href="/about">About</a>
browser-use <<'PY'
start_remote_daemon("agent-1")
PY
```
# 3. Interact — use indices from state
browser-use input 0 "search query" # Type into element 0
browser-use click 1 # Click element 1
Use the same `BU_NAME` for every later call so the agent stays on that cloud browser:
# 4. Verify — re-run state to see result
browser-use state
```bash
# 1. Navigate and observe
BU_NAME=agent-1 browser-use <<'PY'
new_tab("https://html.duckduckgo.com/html/")
wait_for_load()
print(page_info())
PY
# 5. Extract data
browser-use get text 3 # Get element text
browser-use get html --selector "h1" # Get scoped HTML
browser-use eval "document.title" # Execute JavaScript
browser-use screenshot result.png # Capture visual state
# 2. Interact, then verify the result
BU_NAME=agent-1 browser-use <<'PY'
fill_input('input[name="q"]', "search query")
press_key("ENTER")
wait_for_load()
print(js("document.title"))
print(capture_screenshot())
PY
# 6. Wait for dynamic content
browser-use wait selector ".results" # Wait for element
browser-use wait text "Success" # Wait for text
# 7. Cleanup
browser-use close
# 3. Stop the cloud browser when the job is done
browser-use <<'PY'
stop_remote_daemon("agent-1")
PY
```
**Key details:**
- Background daemon keeps browser alive between commands (~50ms latency per call)
- Agent's reasoning loop decides which command to call next
- `state` output is the agent's "eyes" — it reads element indices and decides what to click
- Commands can be chained with `&&` when intermediate output isn't needed
- `--json` flag for machine-readable output
- `--headed` for visible browser (debugging)
- `--profile "Default"` for authenticated browsing with saved Chrome logins
- CLI 3.0 removed the old `open`, `state`, `click`, `eval`, `--json`, `--headed`, and `--profile` command surface.
- The agent writes Python with helpers such as `new_tab`, `page_info`, `fill_input`, `click_at_xy`, `js`, and `cdp`.
- The background daemon keeps the browser alive between calls. Printed Python values are the tool output.
- `BU_NAME` selects the named cloud browser. Without it, the CLI uses the default local browser.
- The first navigation is `new_tab(url)`. Use `goto_url(url)` only after a real tab exists.
- Remote browsers keep billing until they stop or time out. Always call `stop_remote_daemon(name)` after the job.
---
+6 -3
View File
@@ -1,5 +1,9 @@
# Cloud Quickstart, Pricing & FAQ
> This page keeps the v2 quickstart for existing integrations. For a new
> hosted-agent integration, use [API v4](api-v4.md). V4 is the current API and
> SDK path.
## Table of Contents
- [Overview](#overview)
- [Setup](#setup)
@@ -122,8 +126,7 @@ Typical task: 10 steps = ~$0.03 (with Browser Use LLM)
| BU Max (Claude Sonnet 4.6) | ~$3.60 | ~$18.00 |
### Browser Sessions
- PAYG: $0.06/hour
- Business: $0.03/hour
- All plans: $0.02/hour
- Billed upfront, proportional refund on stop. Min 1 minute.
### Skills
@@ -134,7 +137,7 @@ Typical task: 10 steps = ~$0.03 (with Browser Use LLM)
- PAYG: $10/GB, Business: $5/GB, Scaleup: $4/GB
### Tiers
- **Business**: 25% off per-step, 50% off sessions/skills/proxy
- **Business**: 25% off per-step, 50% off skills/proxy
- **Scaleup**: 50% off per-step, 60% off proxy
- **Enterprise**: Contact for ZDR, compliance, on-prem
+6 -5
View File
@@ -59,8 +59,8 @@ Optimized for browser automation — highest accuracy, fastest speed, lowest tok
```python
from browser_use import Agent, ChatBrowserUse
llm = ChatBrowserUse() # bu-latest (default)
llm = ChatBrowserUse(model='bu-2-0') # Premium model
llm = ChatBrowserUse() # bu-2-0 (default, 'bu-latest' tracks it)
llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Cheaper per token, opt-in while in preview
```
**Env:** `BROWSER_USE_API_KEY` — get at https://cloud.browser-use.com/new-api-key
@@ -68,8 +68,9 @@ llm = ChatBrowserUse(model='bu-2-0') # Premium model
**Models & Pricing (per 1M tokens):**
| Model | Input | Cached | Output |
|-------|-------|--------|--------|
| bu-1-0 / bu-latest (default) | $0.20 | $0.02 | $2.00 |
| bu-2-0 (premium) | $0.60 | $0.06 | $3.50 |
| bu-2-0 (default, premium) | $0.60 | $0.06 | $3.50 |
| bu-2-0-mini-preview (opt-in) | $0.15 | $0.15 | $1.50 |
| bu-1-0 (redirects to bu-2-0) | $0.60 | $0.06 | $3.50 |
| browser-use/bu-30b-a3b-preview (OSS) | — | — | — |
## OpenAI
@@ -149,7 +150,7 @@ Supports profiles, IAM roles, SSO via standard AWS credential chain. Install wit
```python
from browser_use import Agent, ChatDeepSeek
llm = ChatDeepSeek(model="deepseek-chat")
llm = ChatDeepSeek(model="deepseek-v4-flash")
```
**Env:** `DEEPSEEK_API_KEY` | [Available models](https://api-docs.deepseek.com/quick_start/pricing)
+55 -143
View File
@@ -1,179 +1,91 @@
---
name: remote-browser
description: Controls a local browser from a sandboxed remote machine. Use when the agent is running in a sandbox (no GUI) and needs to navigate websites, interact with web pages, fill forms, take screenshots, or expose local dev servers via tunnels.
description: Controls an isolated Browser Use Cloud browser from a sandboxed machine with the current Browser Use CLI.
allowed-tools: Bash(browser-use:*)
---
# Browser Automation for Sandboxed Agents
# Remote Browser
This skill is for agents running on **sandboxed remote machines** (cloud VMs, CI, coding agents) that need to control a headless browser.
Use this skill when an agent runs on a machine without a usable local Chrome and needs an isolated browser. The current Browser Use CLI runs Python from stdin. Do not use the removed `open`, `state`, `click`, `input`, `tab`, `cloud connect`, or `--connect` commands.
## Prerequisites
## Check the CLI
```bash
browser-use doctor # Verify installation
browser-use --doctor
browser-use skill show
```
For setup details, see https://github.com/browser-use/browser-use/blob/main/browser_use/skill_cli/README.md
If setup fails, follow the current [Browser Use skill](../browser-use/SKILL.md).
## Core Workflow
## Start an isolated browser
1. **Navigate**: `browser-use open <url>` — starts headless browser if needed
2. **Inspect**: `browser-use state` — returns clickable elements with indices
3. **Interact**: use indices from state (`browser-use click 5`, `browser-use input 3 "text"`)
4. **Verify**: `browser-use state` or `browser-use screenshot` to confirm
5. **Repeat**: browser stays open between commands
6. **Cleanup**: `browser-use close` when done
## Browser Modes
Authenticate once:
```bash
browser-use open <url> # Default: headless Chromium
browser-use cloud connect # Provision cloud browser and connect
browser-use --connect open <url> # Auto-discover running Chrome via CDP
browser-use --cdp-url ws://localhost:9222/... open <url> # Connect via CDP URL
browser-use auth login
```
## Commands
Pick a short unique name. `r7k2` below is only an example.
```bash
# Navigation
browser-use open <url> # Navigate to URL
browser-use back # Go back in history
browser-use scroll down # Scroll down (--amount N for pixels)
browser-use scroll up # Scroll up
browser-use tab list # List all tabs with lock status
browser-use tab new [url] # Open a new tab (blank or with URL)
browser-use tab switch <index> # Switch to tab by index
browser-use tab close <index> [index...] # Close one or more tabs
# Page State — always run state first to get element indices
browser-use state # URL, title, clickable elements with indices
browser-use screenshot [path.png] # Screenshot (base64 if no path, --full for full page)
# Interactions — use indices from state
browser-use click <index> # Click element by index
browser-use click <x> <y> # Click at pixel coordinates
browser-use type "text" # Type into focused element
browser-use input <index> "text" # Click element, then type
browser-use keys "Enter" # Send keyboard keys (also "Control+a", etc.)
browser-use select <index> "option" # Select dropdown option
browser-use upload <index> <path> # Upload file to file input
browser-use hover <index> # Hover over element
browser-use dblclick <index> # Double-click element
browser-use rightclick <index> # Right-click element
# Data Extraction
browser-use eval "js code" # Execute JavaScript, return result
browser-use get title # Page title
browser-use get html [--selector "h1"] # Page HTML (or scoped to selector)
browser-use get text <index> # Element text content
browser-use get value <index> # Input/textarea value
browser-use get attributes <index> # Element attributes
browser-use get bbox <index> # Bounding box (x, y, width, height)
# Wait
browser-use wait selector "css" # Wait for element (--state visible|hidden|attached|detached, --timeout ms)
browser-use wait text "text" # Wait for text to appear
# Cookies
browser-use cookies get [--url <url>] # Get cookies (optionally filtered)
browser-use cookies set <name> <value> # Set cookie (--domain, --secure, --http-only, --same-site, --expires)
browser-use cookies clear [--url <url>] # Clear cookies
browser-use cookies export <file> # Export to JSON
browser-use cookies import <file> # Import from JSON
# Python — persistent session with browser access
browser-use python "code" # Execute Python (variables persist across calls)
browser-use python --file script.py # Run file
browser-use python --vars # Show defined variables
browser-use python --reset # Clear namespace
# Session
browser-use close # Close browser and stop daemon
browser-use sessions # List active sessions
browser-use close --all # Close all sessions
browser-use <<'PY'
start_remote_daemon("r7k2")
PY
```
The Python `browser` object provides: `browser.url`, `browser.title`, `browser.html`, `browser.goto(url)`, `browser.back()`, `browser.click(index)`, `browser.type(text)`, `browser.input(index, text)`, `browser.keys(keys)`, `browser.upload(index, path)`, `browser.screenshot(path)`, `browser.scroll(direction, amount)`, `browser.wait(seconds)`.
## Tunnels
Expose local dev servers to the browser via Cloudflare tunnels.
Use the same name for every command in this browser:
```bash
browser-use tunnel <port> # Start tunnel (idempotent)
browser-use tunnel list # Show active tunnels
browser-use tunnel stop <port> # Stop tunnel
browser-use tunnel stop --all # Stop all tunnels
BU_NAME=r7k2 browser-use <<'PY'
new_tab("https://example.com")
wait_for_load()
print(page_info())
PY
```
## Command Chaining
Each remote daemon is a separate Browser Use Cloud browser. Use a different name for each parallel task. Remote browsers can bill until they stop or time out.
Commands can be chained with `&&`. The browser persists via the daemon, so chaining is safe and efficient.
## Inspect and interact
Helpers are pre-imported. Keep multi-step work in one heredoc when practical.
```bash
browser-use open https://example.com && browser-use state
browser-use input 5 "user@example.com" && browser-use input 6 "password" && browser-use click 7
BU_NAME=r7k2 browser-use <<'PY'
print(page_info())
print(js("document.title"))
fill_input('input[name="q"]', "browser automation")
press_key("Enter")
wait_for_load()
print(page_info())
PY
```
Chain when you don't need intermediate output. Run separately when you need to parse `state` to discover indices first.
Useful helpers:
## Common Workflows
- Navigate: `new_tab(url)`, `goto_url(url)`, `wait_for_load()`
- Inspect: `page_info()`, `js(code)`, `cdp(method, ...)`
- Interact: `click_at_xy(x, y)`, `type_text(text)`, `fill_input(selector, text)`, `press_key(key)`, `scroll(x, y)`
- Tabs: `list_tabs()`, `switch_tab(target)`, `close_tab(target)`
- Files and proof: `capture_screenshot()`, `wait_for_element(selector)`
### Exposing Local Dev Servers
Prefer the accessibility tree for element discovery:
```python
nodes = cdp("Accessibility.getFullAXTree")["nodes"]
```
Use a targeted `js(...)` query when the accessibility tree lacks the element. Verify each action with `page_info()`, a focused DOM check, or a screenshot.
## Stop the browser
When the work is done, stop the exact named browser:
```bash
python -m http.server 3000 & # Start dev server
browser-use tunnel 3000 # → https://abc.trycloudflare.com
browser-use open https://abc.trycloudflare.com # Browse the tunnel
browser-use <<'PY'
stop_remote_daemon("r7k2")
PY
```
Tunnels are independent of browser sessions and persist across `browser-use close`.
## Multi-Agent (--connect mode)
Multiple agents can share one browser via `--connect`. Each agent gets its own tab — other agents can't interfere.
**Setup**: Register once, then pass the index with every `--connect` command:
```bash
INDEX=$(browser-use register) # → prints "1"
browser-use --connect $INDEX open <url> # Navigate in agent's own tab
browser-use --connect $INDEX state # Get state from agent's tab
browser-use --connect $INDEX click <element> # Click in agent's tab
```
- **Tab locking**: When an agent mutates a tab (click, type, navigate), that tab is locked to it. Other agents get an error if they try to mutate the same tab.
- **Read-only access**: `state`, `screenshot`, `get`, and `wait` commands work on any tab regardless of locks.
- **Agent sessions expire** after 5 minutes of inactivity. Run `browser-use register` again to get a new index.
## Global Options
| Option | Description |
|--------|-------------|
| `--headed` | Show browser window |
| `--connect` | Auto-discover running Chrome via CDP |
| `--cdp-url <url>` | Connect via CDP URL (`http://` or `ws://`) |
| `--session NAME` | Target a named session (default: "default") |
| `--json` | Output as JSON |
## Tips
1. **Always run `state` first** to see available elements and their indices
2. **Sessions persist** — browser stays open between commands until you close it
3. **Tunnels are independent** — they persist across `browser-use close`
4. **`tunnel` is idempotent** — calling again for the same port returns the existing URL
## Troubleshooting
- **Browser won't start?** `browser-use close` then retry. Run `browser-use doctor` to check.
- **Element not found?** `browser-use scroll down` then `browser-use state`
- **Tunnel not working?** `which cloudflared` to check, `browser-use tunnel list` to see active tunnels
## Cleanup
```bash
browser-use close # Close browser session
browser-use tunnel stop --all # Stop tunnels (if any)
```
Do not leave an unused remote browser running.
+183
View File
@@ -5,6 +5,7 @@ import tempfile
from pathlib import Path
import pytest
from pypdf import PdfReader
from browser_use.filesystem.file_system import (
DEFAULT_FILE_SYSTEM_PATH,
@@ -16,9 +17,35 @@ from browser_use.filesystem.file_system import (
JsonlFile,
MarkdownFile,
TxtFile,
_markdown_inline_to_rml,
_split_heading,
)
def _extract_pdf_text(path: Path) -> tuple[str, list[str]]:
"""Extract a PDF as (whitespace-collapsed blob, non-empty stripped lines).
The blob tolerates line wrapping; the lines are what header markers would survive on.
"""
text = '\n'.join(page.extract_text() or '' for page in PdfReader(path).pages)
return ' '.join(text.split()), [line.strip() for line in text.splitlines() if line.strip()]
def _extract_pdf_fonts(path: Path) -> set[str]:
"""Return the PDF base fonts used to render non-empty text."""
fonts: set[str] = set()
def _record_font(text: str, _cm, _tm, font_dictionary, _font_size) -> None:
if text.strip() and font_dictionary is not None:
base_font = font_dictionary.get('/BaseFont')
if base_font is not None:
fonts.add(str(base_font))
for page in PdfReader(path).pages:
page.extract_text(visitor_text=_record_font)
return fonts
class TestBaseFile:
"""Test the BaseFile abstract base class and its implementations."""
@@ -33,6 +60,34 @@ class TestBaseFile:
assert md_file.get_size == 13
assert md_file.get_line_count == 1
def test_split_heading_levels(self):
"""PdfFile and DocxFile share one heading parser."""
assert _split_heading('# Title') == ('Title', 1)
assert _split_heading('## Section') == ('Section', 2)
assert _split_heading('### Notes') == ('Notes', 3)
assert _split_heading('#hashtag is not a header') == ('#hashtag is not a header', None)
assert _split_heading('plain') == ('plain', None)
assert _split_heading('#### too deep') == ('#### too deep', None)
def test_markdown_inline_glob_is_not_bold(self):
"""Recursive globs must not be treated as bold delimiters."""
assert _markdown_inline_to_rml('**/foo/**') == '**/foo/**'
assert _markdown_inline_to_rml('**bold** and **/foo/**') == '<b>bold</b> and **/foo/**'
def test_markdown_inline_code_protects_emphasis(self):
"""Emphasis markers inside backticks stay Courier, not italic/bold."""
assert _markdown_inline_to_rml('`*literal*`') == '<font face="Courier">*literal*</font>'
assert _markdown_inline_to_rml('`**bold**`') == '<font face="Courier">**bold**</font>'
def test_markdown_inline_code_token_cannot_replace_user_text(self):
"""A user-provided placeholder-like sequence must survive code-span restoration."""
literal = '\x00C0\x00'
assert _markdown_inline_to_rml(f'Keep {literal} and `code`.') == (f'Keep {literal} and <font face="Courier">code</font>.')
prefix_literal = '\x00BROWSER_USE_INLINE_CODE_0\x00'
assert _markdown_inline_to_rml(f'Keep {prefix_literal} and `code`.') == (
f'Keep {prefix_literal} and <font face="Courier">code</font>.'
)
def test_txt_file_creation(self):
"""Test TxtFile creation and basic properties."""
txt_file = TxtFile(name='notes', content='Hello\nWorld')
@@ -222,6 +277,122 @@ class TestFileSystem:
except Exception:
pass
async def test_write_pdf_renders_content_as_plain_text(self, empty_filesystem):
"""PDF content should be rendered as plain text, not ReportLab markup."""
# (markdown source, text expected in the PDF)
headers = [
('# Q&A: <b y vs 5', 'Q&A: <b y vs 5'),
('## Section 2 & 3', 'Section 2 & 3'),
('### Notes <i> #1', 'Notes <i> #1'),
]
body = ['Comparison: 2 <b y & 5 > 4', '#hashtag is not a header', 'O\'Brien said "hi" & left']
result = await empty_filesystem.write_file('comparison.pdf', '\n\n'.join([source for source, _ in headers] + body))
assert result == 'Data written to file comparison.pdf successfully.'
blob, lines = _extract_pdf_text(empty_filesystem.data_dir / 'comparison.pdf')
for _, rendered in headers:
assert rendered in blob
for line in body:
assert line in blob
# Content checks alone pass with a leftover marker, e.g. '# Notes <i> #1' contains 'Notes <i> #1'
assert [line for line in lines if line.startswith(('# ', '## ', '### '))] == []
async def test_append_pdf_escapes_both_halves(self, empty_filesystem):
"""Appending re-renders the whole PDF, so old and new content must both stay plain text."""
# The markup that breaks ReportLab goes in the appended half, so this fails on the append path alone
await empty_filesystem.write_file('report.pdf', 'First & half')
result = await empty_filesystem.append_file('report.pdf', '\nSecond <b half > 2')
assert result == 'Data appended to file report.pdf successfully.'
blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'report.pdf')
assert 'First & half' in blob
assert 'Second <b half > 2' in blob
async def test_write_pdf_renders_markdown_formatting(self, empty_filesystem):
"""PDF export honors bold, italic, inline code, and bullets."""
source = '\n'.join(
[
'This is **bold text** in a sentence.',
'This is *italic text* in a sentence.',
'Use `inline_code` here.',
'- first bullet item',
'- second bullet item',
'* star bullet item',
]
)
result = await empty_filesystem.write_file('formatted.pdf', source)
assert result == 'Data written to file formatted.pdf successfully.'
pdf_path = empty_filesystem.data_dir / 'formatted.pdf'
blob, lines = _extract_pdf_text(pdf_path)
fonts = _extract_pdf_fonts(pdf_path)
assert 'bold text' in blob
assert '**bold text**' not in blob
assert 'italic text' in blob
assert '*italic text*' not in blob
assert 'inline_code' in blob
assert '`inline_code`' not in blob
assert 'first bullet item' in blob
assert 'second bullet item' in blob
assert 'star bullet item' in blob
assert not any(line.startswith(('- ', '* ')) for line in lines)
assert {'/Helvetica-Bold', '/Helvetica-Oblique', '/Courier'} <= fonts
async def test_write_pdf_star_overload_is_not_emphasis(self, empty_filesystem):
"""Arithmetic and glob stars must not be treated as italic or bullets."""
source = '\n'.join(
[
'Arithmetic 2 * 3 * 4 stays literal',
'Glob pattern *.txt and **/*.py stay literal',
'Recursive glob **/foo/** stays literal',
]
)
result = await empty_filesystem.write_file('stars.pdf', source)
assert result == 'Data written to file stars.pdf successfully.'
blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'stars.pdf')
assert '2 * 3 * 4' in blob
assert '*.txt' in blob
assert '**/*.py' in blob
assert '**/foo/**' in blob
async def test_write_pdf_underscore_is_not_italic(self, empty_filesystem):
"""Underscores are identifiers, not emphasis."""
source = 'Keep snake_case_names and file_name_here unchanged.'
result = await empty_filesystem.write_file('names.pdf', source)
assert result == 'Data written to file names.pdf successfully.'
blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'names.pdf')
assert 'snake_case_names' in blob
assert 'file_name_here' in blob
async def test_write_pdf_fenced_backticks_stay_literal(self, empty_filesystem):
"""Inline code strips backticks; fenced shell snippets keep them."""
source = '\n'.join(
[
'Run `$(cmd)` inline.',
'```',
'echo `$(cmd)`',
'```',
]
)
result = await empty_filesystem.write_file('backticks.pdf', source)
assert result == 'Data written to file backticks.pdf successfully.'
blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'backticks.pdf')
assert 'Run $(cmd) inline.' in blob
assert 'echo `$(cmd)`' in blob
async def test_write_pdf_inline_code_protects_emphasis_markers(self, empty_filesystem):
"""Stars inside inline code stay literal instead of becoming italic/bold."""
source = 'Keep `*literal*` and `**bold**` as code.'
result = await empty_filesystem.write_file('code_stars.pdf', source)
assert result == 'Data written to file code_stars.pdf successfully.'
blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'code_stars.pdf')
assert '*literal*' in blob
assert '**bold**' in blob
assert '`*literal*`' not in blob
assert '`**bold**`' not in blob
def test_filesystem_initialization(self, temp_filesystem):
"""Test FileSystem initialization with default files."""
fs = temp_filesystem
@@ -478,6 +649,18 @@ class TestFileSystem:
assert 'not found' in result
assert 'auto-corrected' in result
async def test_replace_file_reports_missing_text(self, temp_filesystem):
"""Test that replacing absent text reports an error without changing the file."""
fs = temp_filesystem
original_content = '- [ ] First task\n- [ ] Second task'
await fs.write_file('todo.md', original_content)
result = await fs.replace_file_str('todo.md', '- [ ] Missing task', '- [x] Missing task')
assert result == 'Error: Could not find the specified text in file todo.md.'
assert fs.get_file('todo.md').content == original_content
assert (fs.data_dir / 'todo.md').read_text(encoding='utf-8') == original_content
async def test_append_json_file(self, temp_filesystem):
"""Test appending content to JSON files."""
fs = temp_filesystem
@@ -0,0 +1,14 @@
from browser_use.llm.aws.chat_anthropic import ChatAnthropicBedrock
def test_get_client_preserves_zero_max_retries() -> None:
llm = ChatAnthropicBedrock(
aws_access_key='test-access-key',
aws_secret_key='test-secret-key',
aws_region='us-east-1',
max_retries=0,
)
client = llm.get_client()
assert client.max_retries == 0
+32 -1
View File
@@ -26,12 +26,14 @@ async def test_browseruse_bu_latest(httpserver):
def test_default_model_is_bu_2_0():
"""The default must be a stable model - a preview is opt-in, never reached by omission."""
chat = ChatBrowserUse(api_key=TEST_API_KEY)
assert chat.model == 'bu-2-0'
assert chat.provider == 'browser-use'
assert 'preview' not in chat.model
@pytest.mark.parametrize('alias', ['bu-1-0', 'bu-2-0', 'bu-qa-1'])
@pytest.mark.parametrize('alias', ['bu-1-0', 'bu-2-0', 'bu-2-0-mini-preview', 'bu-qa-1'])
def test_bu_aliases_are_accepted(alias):
chat = ChatBrowserUse(model=alias, api_key=TEST_API_KEY)
assert chat.model == alias
@@ -40,9 +42,38 @@ def test_bu_aliases_are_accepted(alias):
def test_bu_latest_normalizes_to_bu_2_0():
"""'latest' tracks the stable premium line, which is also what omitting the model gives."""
chat = ChatBrowserUse(model='bu-latest', api_key=TEST_API_KEY)
assert chat.model == 'bu-2-0'
assert chat.name == 'bu-2-0'
assert ChatBrowserUse(api_key=TEST_API_KEY).model == chat.model
def test_bu_2_0_mini_preview_is_priced():
"""The default model must have a pricing entry, or cost tracking silently reports $0."""
from browser_use.tokens.custom_pricing import CUSTOM_MODEL_PRICING
pricing = CUSTOM_MODEL_PRICING['bu-2-0-mini-preview']
assert pricing['input_cost_per_token'] > 0
assert pricing['output_cost_per_token'] > 0
# bu-latest resolves to bu-2-0, not to the preview, so it keeps the premium pricing.
assert CUSTOM_MODEL_PRICING['bu-latest'] == CUSTOM_MODEL_PRICING['bu-2-0']
# bu-1-0 is redirected to bu-2-0 at the gateway, so it must be billed at bu-2-0 rates
# rather than the retired bu-1-0 rates.
assert CUSTOM_MODEL_PRICING['bu-1-0'] == CUSTOM_MODEL_PRICING['bu-2-0']
def test_llm_models_shortcut_resolves_mini_preview(monkeypatch):
"""`llm.bu_2_0_mini_preview` must map the underscored name back to the dashed model id."""
monkeypatch.setenv('BROWSER_USE_API_KEY', TEST_API_KEY)
from browser_use import llm
from browser_use.llm import models
# Go through the advertised attribute rather than the factory beneath it, so dropping the
# name from __all__ or breaking module __getattr__ fails here instead of passing silently.
assert llm.bu_2_0_mini_preview.name == 'bu-2-0-mini-preview'
assert models.bu_2_0_mini_preview.name == 'bu-2-0-mini-preview'
assert 'bu_2_0_mini_preview' in models.__all__
@pytest.mark.parametrize(
+116
View File
@@ -127,3 +127,119 @@ async def test_chat_google_temperature_fallback():
mock_models.generate_content.assert_called_once()
args, kwargs = mock_models.generate_content.call_args
assert kwargs['config']['temperature'] == 1.0
# A 1x1 PNG, small enough to inline and still be a real decodable image.
_PNG_1PX = 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=='
def _flatten(contents) -> list:
"""Flatten serialized Google contents into a single list of parts."""
return [part for content in contents for part in (content.parts or [])]
def _describe(contents) -> tuple[str, int]:
"""Summarise serialized Google contents as (all text, number of inline images)."""
parts = _flatten(contents)
return ''.join(part.text or '' for part in parts), sum(1 for part in parts if part.inline_data is not None)
def test_include_system_in_user_keeps_list_content_parts():
"""include_system_in_user must prepend the system text, not replace the user content.
With vision on, the agent's user message is a list of parts, and every part of it was
being dropped from the first user message.
"""
from browser_use.llm.google.serializer import GoogleMessageSerializer
from browser_use.llm.messages import (
ContentPartImageParam,
ContentPartTextParam,
ImageURL,
SystemMessage,
UserMessage,
)
messages = [
SystemMessage(content='You are a browser agent.'),
UserMessage(
content=[
ContentPartTextParam(text='<browser_state>the page and the task live here</browser_state>'),
ContentPartImageParam(image_url=ImageURL(url=f'data:image/png;base64,{_PNG_1PX}', media_type='image/png')),
]
),
]
contents, system_instruction = GoogleMessageSerializer.serialize_messages(messages, include_system_in_user=True)
assert system_instruction is None, 'system message should have moved into the user turn'
text, images = _describe(contents)
assert 'You are a browser agent.' in text
assert '<browser_state>the page and the task live here</browser_state>' in text, 'user text was dropped'
assert images == 1, 'screenshot was dropped'
def test_include_system_in_user_string_content_is_merged_into_one_part():
"""String content keeps the existing behaviour: system text and user text in a single part."""
from browser_use.llm.google.serializer import GoogleMessageSerializer
from browser_use.llm.messages import SystemMessage, UserMessage
messages = [
SystemMessage(content='You are a browser agent.'),
UserMessage(content='Buy a red stapler'),
]
contents, system_instruction = GoogleMessageSerializer.serialize_messages(messages, include_system_in_user=True)
assert system_instruction is None
parts = _flatten(contents)
assert len(parts) == 1
assert parts[0].text == 'You are a browser agent.\n\nBuy a red stapler'
def test_include_system_in_user_keeps_the_agent_state_message(tmp_path):
"""End-to-end shape: what the agent actually builds with vision on must survive serialization."""
from browser_use.agent.prompts import AgentMessagePrompt, SystemPrompt
from browser_use.agent.views import AgentStepInfo
from browser_use.browser.views import BrowserStateSummary, PageInfo, TabInfo
from browser_use.dom.views import SerializedDOMState
from browser_use.filesystem.file_system import FileSystem
from browser_use.llm.google.serializer import GoogleMessageSerializer
browser_state = BrowserStateSummary(
url='https://example.test/foo',
title='Test',
tabs=[TabInfo(target_id='abcd1234', url='https://example.test/foo', title='Test')],
page_info=PageInfo(
viewport_width=1280,
viewport_height=720,
page_width=1280,
page_height=1440,
scroll_x=0,
scroll_y=0,
pixels_above=0,
pixels_below=720,
pixels_left=0,
pixels_right=0,
),
dom_state=SerializedDOMState(_root=None, selector_map={}),
is_pdf_viewer=False,
recent_events=None,
closed_popup_messages=[],
screenshot=_PNG_1PX,
)
user_message = AgentMessagePrompt(
browser_state_summary=browser_state,
file_system=FileSystem(base_dir=str(tmp_path), create_default_files=False),
agent_history_description='<step>existing history</step>',
task='Buy a red stapler',
step_info=AgentStepInfo(step_number=1, max_steps=50),
screenshots=[_PNG_1PX],
).get_user_message(use_vision=True)
assert isinstance(user_message.content, list), 'vision messages are lists of parts'
messages = [SystemPrompt(max_actions_per_step=5).get_system_message(), user_message]
contents, _ = GoogleMessageSerializer.serialize_messages(messages, include_system_in_user=True)
text, images = _describe(contents)
assert 'Buy a red stapler' in text, 'the task never reached the model'
assert images == 1, 'the screenshot never reached the model'
+71
View File
@@ -0,0 +1,71 @@
"""Tests for ChatOllama option handling and structured-output parsing."""
from unittest.mock import AsyncMock, MagicMock, patch
import pytest
from pydantic import BaseModel
from browser_use.llm.exceptions import ModelProviderError
from browser_use.llm.messages import UserMessage
from browser_use.llm.ollama.chat import ChatOllama
class Answer(BaseModel):
answer: str
def _client_returning(content: str) -> MagicMock:
client = MagicMock()
client.chat = AsyncMock(return_value=MagicMock(message=MagicMock(content=content)))
return client
async def test_splits_top_level_chat_parameters_from_ollama_options():
"""Top-level chat parameters must not be sent inside model options (#5017)."""
client = _client_returning('{"answer": "ok"}')
llm = ChatOllama(
model='test-model',
ollama_options={
'think': False,
'logprobs': True,
'top_logprobs': 3,
'keep_alive': '10m',
'format': 'json',
'stream': False,
'num_ctx': 2048,
},
)
with patch.object(ChatOllama, 'get_client', return_value=client):
result = await llm.ainvoke([UserMessage(content='hi')], output_format=Answer)
assert result.completion.answer == 'ok'
kwargs = client.chat.await_args.kwargs
assert kwargs['options'] == {'num_ctx': 2048}
assert kwargs['think'] is False
assert kwargs['logprobs'] is True
assert kwargs['top_logprobs'] == 3
assert kwargs['keep_alive'] == '10m'
assert kwargs['format'] == Answer.model_json_schema()
assert kwargs.get('stream') is None
@pytest.mark.parametrize('fence', ['```json', '```JSON', '``` json', '```'])
async def test_parses_json_wrapped_in_markdown_fences(fence: str):
client = _client_returning(f'{fence}\n{{"answer": "ok"}}\n```')
llm = ChatOllama(model='test-model')
with patch.object(ChatOllama, 'get_client', return_value=client):
result = await llm.ainvoke([UserMessage(content='hi')], output_format=Answer)
assert result.completion.answer == 'ok'
async def test_truncated_json_raises_model_provider_error():
client = _client_returning('{\n')
llm = ChatOllama(model='test-model')
with patch.object(ChatOllama, 'get_client', return_value=client), pytest.raises(ModelProviderError) as exc_info:
await llm.ainvoke([UserMessage(content='hi')], output_format=Answer)
assert 'Invalid JSON' in exc_info.value.message or 'invalid JSON' in exc_info.value.message.lower()
+51
View File
@@ -1,9 +1,30 @@
"""Test OpenAI model button click."""
from types import SimpleNamespace
import pytest
from browser_use.llm.base import is_reasoning_model
from browser_use.llm.messages import UserMessage
from browser_use.llm.openai.chat import ChatOpenAI
from tests.ci.models.model_test_helper import run_model_button_click_test
@pytest.mark.parametrize(
('model', 'reasoning_models', 'expected'),
[
('gpt-4.1', [''], False),
('gpt-4.1', [' ', ''], False),
('o3-mini', ['', 'o3'], True),
('o3-mini', [' o3'], False),
('gpt-4.1', None, False),
],
)
def test_reasoning_model_matching_ignores_empty_patterns(model, reasoning_models, expected):
"""Empty patterns must not match every model name."""
assert is_reasoning_model(model, reasoning_models) is expected
async def test_openai_gpt_4_1_mini(httpserver):
"""Test OpenAI gpt-4.1-mini can click a button."""
await run_model_button_click_test(
@@ -13,3 +34,33 @@ async def test_openai_gpt_4_1_mini(httpserver):
extra_kwargs={},
httpserver=httpserver,
)
@pytest.mark.parametrize('reasoning_models', [[], [''], [' ', '', '']])
async def test_openai_empty_reasoning_model_patterns_preserve_sampling_parameters(monkeypatch, reasoning_models):
"""Empty reasoning patterns must not classify a regular model as reasoning."""
captured: dict[str, object] = {}
class FakeCompletions:
async def create(self, **kwargs):
captured.update(kwargs)
return SimpleNamespace(
choices=[SimpleNamespace(message=SimpleNamespace(content='ok'), finish_reason='stop')],
usage=None,
)
fake_client = SimpleNamespace(chat=SimpleNamespace(completions=FakeCompletions()))
llm = ChatOpenAI(
model='gpt-4.1',
api_key='test-key',
temperature=0.7,
frequency_penalty=0.4,
reasoning_models=reasoning_models,
)
monkeypatch.setattr(llm, 'get_client', lambda: fake_client)
await llm.ainvoke([UserMessage(content='hello')])
assert captured['temperature'] == 0.7
assert captured['frequency_penalty'] == 0.4
assert 'reasoning_effort' not in captured
+22 -1
View File
@@ -3,7 +3,7 @@ Tests for the SchemaOptimizer to ensure it correctly processes and
optimizes the schemas for agent actions without losing information.
"""
from pydantic import BaseModel
from pydantic import BaseModel, Field
from browser_use.agent.views import AgentOutput
from browser_use.llm.schema import SchemaOptimizer
@@ -74,3 +74,24 @@ def test_gemini_schema_retains_required_fields():
required_fields = set(schema['required'])
assert {'price', 'title'}.issubset(required_fields), 'Mandatory fields must stay required for Gemini.'
def test_optimizer_treats_property_names_as_data_not_schema_keywords():
"""Nested fields named after schema keywords must still have their refs flattened."""
class Details(BaseModel):
summary: str
class Article(BaseModel):
description: Details
properties: Details
additional_properties: Details = Field(alias='additionalProperties')
defs: Details = Field(alias='$defs')
schema = SchemaOptimizer.create_optimized_json_schema(Article)
assert '$defs' not in schema
for field_name in ('description', 'properties', 'additionalProperties', '$defs'):
field_schema = schema['properties'][field_name]
assert '$ref' not in field_schema
assert field_schema['properties']['summary']['type'] == 'string'
+35
View File
@@ -18,6 +18,18 @@ class SensitiveParams(BaseModel):
text: str = Field(description='Text with sensitive data placeholders')
class TupleSensitiveParams(BaseModel):
"""Test parameter model for tuple-based sensitive data placeholders."""
items: tuple[str, ...] = Field(description='Tuple with sensitive data placeholders')
class NestedTupleSensitiveParams(BaseModel):
"""Test parameter model for nested tuple-based sensitive data placeholders."""
payload: tuple[dict[str, tuple[str, list[str]]], ...] = Field(description='Nested tuple with sensitive data placeholders')
@pytest.fixture
def registry():
return Registry()
@@ -72,6 +84,29 @@ def test_replace_sensitive_data_with_missing_keys(registry, caplog):
assert '<secret>password</secret>' in result.text # Empty value's tag remains
def test_replace_sensitive_data_inside_tuple(registry):
"""Test that _replace_sensitive_data replaces placeholders inside tuple fields."""
params = TupleSensitiveParams(items=('<secret>api_key</secret>', 'username', 'unchanged'))
sensitive_data = {'api_key': 'sk-replaced', 'username': 'admin_user'}
result = registry._replace_sensitive_data(params, sensitive_data)
assert result.items == ('sk-replaced', 'admin_user', 'unchanged')
assert isinstance(result.items, tuple)
def test_replace_sensitive_data_inside_nested_tuple(registry):
"""Test that _replace_sensitive_data replaces placeholders inside nested tuples."""
params = NestedTupleSensitiveParams(payload=({'credentials': ('<secret>token</secret>', ['<secret>username</secret>'])},))
sensitive_data = {'token': 'token-replaced', 'username': 'admin_user'}
result = registry._replace_sensitive_data(params, sensitive_data)
assert result.payload == ({'credentials': ('token-replaced', ['admin_user'])},)
assert isinstance(result.payload, tuple)
assert isinstance(result.payload[0]['credentials'], tuple)
def test_simple_domain_specific_sensitive_data(registry, caplog):
"""Test the basic functionality of domain-specific sensitive data replacement"""
# Create a simple Pydantic model with sensitive data placeholders
+13
View File
@@ -4297,6 +4297,10 @@ def test_beta_agent_exposes_task_helper_methods():
assert 'Expected output format: Answer' in enhanced
assert '"answer"' in enhanced
assert agent._extract_start_url('Open example.com and report the title.') == 'https://example.com'
assert agent._extract_start_url('Open this notable site: https://example.com') == 'https://example.com'
assert browser_use_agent._extract_start_url('Open this notable site: https://example.com') == 'https://example.com'
assert agent._extract_start_url('Do not open https://example.com.') is None
assert browser_use_agent._extract_start_url('Do not open https://example.com.') is None
assert agent._extract_start_url('Email support@example.com only.') is None
assert agent._extract_start_url('Open https://example.com/report.pdf and summarize it.') is None
assert agent._extract_start_url('Use https://XXX.XX as a placeholder in the table.') is None
@@ -4305,6 +4309,15 @@ def test_beta_agent_exposes_task_helper_methods():
assert agent._extract_start_url(numbered_task) == 'https://elibrary.ferc.gov/eLibrary/search'
assert browser_use_agent._extract_start_url(numbered_task) == 'https://elibrary.ferc.gov/eLibrary/search'
# A closing bracket the URL opened itself is part of the path, not prose.
wikipedia_task = 'Summarize https://en.wikipedia.org/wiki/Python_(programming_language)'
assert agent._extract_start_url(wikipedia_task) == 'https://en.wikipedia.org/wiki/Python_(programming_language)'
assert browser_use_agent._extract_start_url(wikipedia_task) == 'https://en.wikipedia.org/wiki/Python_(programming_language)'
assert agent._extract_start_url('Check https://example.com/a[1] please') == 'https://example.com/a[1]'
# ...but one the prose opened is still dropped, however many there are.
assert agent._extract_start_url('See the docs (https://example.com/guide) for details.') == 'https://example.com/guide'
assert agent._extract_start_url('Read ((see https://example.com/a)) now') == 'https://example.com/a'
def test_beta_agent_exposes_url_text_helper_methods():
from browser_use.beta import Agent
+43
View File
@@ -0,0 +1,43 @@
import asyncio
from browser_use.browser.profile import BrowserProfile
from browser_use.browser.session import BrowserSession
class MessageHandlerClient:
def __init__(self, task: asyncio.Task) -> None:
self._message_handler_task = task
async def test_ws_drop_during_reconnect_triggers_follow_up_attempt(monkeypatch) -> None:
session = BrowserSession(browser_profile=BrowserProfile(headless=True, user_data_dir=None, cdp_url='ws://127.0.0.1:9222'))
reconnect_started = asyncio.Event()
allow_reconnect = asyncio.Event()
reconnect_attempts = 0
async def reconnect(self: BrowserSession) -> None:
nonlocal reconnect_attempts
reconnect_attempts += 1
if reconnect_attempts == 1:
reconnect_started.set()
await allow_reconnect.wait()
monkeypatch.setattr(BrowserSession, 'reconnect', reconnect)
connection_closed = asyncio.get_running_loop().create_future()
task = asyncio.ensure_future(connection_closed)
session._cdp_client_root = MessageHandlerClient(task) # type: ignore[assignment]
initial_reconnect = asyncio.create_task(session._auto_reconnect(max_attempts=1))
await reconnect_started.wait()
session._attach_ws_drop_callback()
connection_closed.set_exception(ConnectionResetError('ws dropped again'))
await asyncio.sleep(0)
connection_closed.exception()
allow_reconnect.set()
await initial_reconnect
assert session._reconnect_task is not None
await session._reconnect_task
assert reconnect_attempts == 2
await session.event_bus.stop(clear=True, timeout=5)
@@ -1,4 +1,5 @@
import os
import re
import subprocess
import sys
from pathlib import Path
@@ -12,6 +13,7 @@ EXPECTED_SKILL_INSTALL_PATHS = (
Path('.copilot') / 'skills' / 'browser-use' / 'SKILL.md',
Path('.cursor') / 'skills' / 'browser-use' / 'SKILL.md',
Path('.gemini') / 'skills' / 'browser-use' / 'SKILL.md',
Path('.openclaw') / 'skills' / 'browser-use' / 'SKILL.md',
Path('.config') / 'opencode' / 'skills' / 'browser-use' / 'SKILL.md',
)
@@ -56,6 +58,50 @@ def test_docs_install_browser_use_skill_from_package_alias():
assert 'raw.githubusercontent.com/browser-use/browser-harness/main/SKILL.md' not in readme
def test_cloud_v4_reference_scopes_workspace_file_listing():
api_v4 = (ROOT / 'skills' / 'cloud' / 'references' / 'api-v4.md').read_text(encoding='utf-8')
assert 'client.workspaces.files(workspace.id)' in api_v4
assert 'client.workspaces.files()' not in api_v4
python_examples = re.findall(r'```python\n(.*?)```', api_v4, flags=re.DOTALL)
assert python_examples
assert all('BrowserUse' in example for example in python_examples if 'client.' in example)
def test_remote_browser_skill_uses_current_cli():
remote_skill = (ROOT / 'skills' / 'remote-browser' / 'SKILL.md').read_text(encoding='utf-8')
for removed_command in (
'browser-use open',
'browser-use state',
'browser-use click',
'browser-use input',
'browser-use tab',
'browser-use screenshot',
'browser-use eval',
'browser-use cookies',
'browser-use close',
'browser-use sessions',
'browser-use tunnel',
'browser-use wait',
'browser-use register',
'browser-use cloud connect',
'browser-use --connect',
'browser_use/skill_cli/README.md',
):
assert removed_command not in remote_skill
for current_command in (
"browser-use <<'PY'",
'start_remote_daemon("r7k2")',
'BU_NAME=r7k2 browser-use',
'new_tab("https://example.com")',
'print(page_info())',
'stop_remote_daemon("r7k2")',
):
assert current_command in remote_skill
def test_browser_use_cli_installs_browser_harness_package_skill(tmp_path):
bin_dir = _fake_browser_harness_tools(tmp_path, '---\nname: browser-harness\n---\n\n# Browser Harness\n')
@@ -86,6 +132,24 @@ def test_browser_use_cli_installs_browser_harness_package_skill(tmp_path):
'---\n'
'name: browser-use\n'
'description: "Direct browser control via CDP for web interaction: automation, scraping, testing, screenshots, and site/app work."\n'
'homepage: https://browser-use.com\n'
'metadata:\n'
' {\n'
' "openclaw":\n'
' {\n'
' "requires": { "bins": ["browser-use"] },\n'
' "install":\n'
' [\n'
' {\n'
' "id": "uv",\n'
' "kind": "uv",\n'
' "package": "browser-use",\n'
' "bins": ["browser-use"],\n'
' "label": "Install Browser Use CLI (uv)",\n'
' },\n'
' ],\n'
' },\n'
' }\n'
'---\n\n'
'# Browser Use\n'
)
+101 -108
View File
@@ -1,123 +1,116 @@
"""Tests for extension configuration environment variables."""
import os
"""Tests for browser configuration environment variables."""
import pytest
from browser_use.browser import BrowserSession
from browser_use.browser.profile import (
BrowserProfile,
_get_enable_default_extensions_default,
_get_headless_default,
)
class TestDisableExtensionsEnvVar:
"""Test BROWSER_USE_DISABLE_EXTENSIONS environment variable."""
TRUTHY_STRINGS = ['true', 'True', 'TRUE', '1', 'yes', 'on']
FALSY_STRINGS = ['false', 'False', 'FALSE', '0', 'no', 'off', '']
def test_default_value_is_true(self):
"""Without env var set, enable_default_extensions should default to True."""
# Clear the env var if it exists
original = os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None)
try:
# Import fresh to get the default
from browser_use.browser.profile import _get_enable_default_extensions_default
assert _get_enable_default_extensions_default() is True
finally:
if original is not None:
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original
class TestConfigEnvVars:
"""Tests for browser profile env var configuration."""
def test_default_values_without_env(self, monkeypatch: pytest.MonkeyPatch):
"""Verify default values when environment variables are unset."""
monkeypatch.delenv('BROWSER_USE_DISABLE_EXTENSIONS', raising=False)
monkeypatch.delenv('BROWSER_USE_HEADLESS', raising=False)
assert _get_enable_default_extensions_default() is True
assert _get_headless_default() is None
@pytest.mark.parametrize(
'env_value,expected_enabled',
'env_var,getter,expected',
[
# Truthy values for DISABLE = extensions disabled (False)
('true', False),
('True', False),
('TRUE', False),
('1', False),
('yes', False),
('on', False),
# Falsy values for DISABLE = extensions enabled (True)
('false', True),
('False', True),
('FALSE', True),
('0', True),
('no', True),
('off', True),
('', True),
('BROWSER_USE_DISABLE_EXTENSIONS', _get_enable_default_extensions_default, False),
('BROWSER_USE_HEADLESS', _get_headless_default, True),
],
)
def test_env_var_values(self, env_value: str, expected_enabled: bool):
"""Test various env var values are parsed correctly."""
original = os.environ.get('BROWSER_USE_DISABLE_EXTENSIONS')
try:
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = env_value
from browser_use.browser.profile import _get_enable_default_extensions_default
def test_env_var_truthy_values(
self,
monkeypatch: pytest.MonkeyPatch,
env_var: str,
getter,
expected: bool,
):
"""Test truthy env var values are parsed correctly."""
for val in TRUTHY_STRINGS:
monkeypatch.setenv(env_var, val)
assert getter() is expected, f'Failed for {env_var}={val}'
result = _get_enable_default_extensions_default()
assert result is expected_enabled, (
f"Expected enable_default_extensions={expected_enabled} for DISABLE_EXTENSIONS='{env_value}', got {result}"
)
finally:
if original is not None:
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original
else:
os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None)
@pytest.mark.parametrize(
'env_var,getter,expected',
[
('BROWSER_USE_DISABLE_EXTENSIONS', _get_enable_default_extensions_default, True),
('BROWSER_USE_HEADLESS', _get_headless_default, False),
],
)
def test_env_var_falsy_values(
self,
monkeypatch: pytest.MonkeyPatch,
env_var: str,
getter,
expected: bool,
):
"""Test falsy env var values are parsed correctly."""
for val in FALSY_STRINGS:
monkeypatch.setenv(env_var, val)
assert getter() is expected, f'Failed for {env_var}={val}'
def test_browser_profile_uses_env_var(self):
"""Test that BrowserProfile picks up the env var."""
original = os.environ.get('BROWSER_USE_DISABLE_EXTENSIONS')
try:
# Test with env var set to true (disable extensions)
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = 'true'
@pytest.mark.parametrize(
'env_var,attr_name,truthy_val,falsy_val',
[
('BROWSER_USE_DISABLE_EXTENSIONS', 'enable_default_extensions', False, True),
('BROWSER_USE_HEADLESS', 'headless', True, False),
],
)
def test_browser_profile_and_session_env_var(
self,
monkeypatch: pytest.MonkeyPatch,
env_var: str,
attr_name: str,
truthy_val: bool,
falsy_val: bool,
):
"""Test that BrowserProfile and BrowserSession pick up env vars."""
# Test truthy env value
monkeypatch.setenv(env_var, 'true')
profile = BrowserProfile()
assert getattr(profile, attr_name) is truthy_val
session = BrowserSession()
assert getattr(session.browser_profile, attr_name) is truthy_val
from browser_use.browser.profile import BrowserProfile
# Test falsy env value
monkeypatch.setenv(env_var, 'false')
profile_falsy = BrowserProfile()
assert getattr(profile_falsy, attr_name) is falsy_val
session_falsy = BrowserSession()
assert getattr(session_falsy.browser_profile, attr_name) is falsy_val
profile = BrowserProfile(headless=True)
assert profile.enable_default_extensions is False, (
'BrowserProfile should disable extensions when BROWSER_USE_DISABLE_EXTENSIONS=true'
)
# Test with env var set to false (enable extensions)
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = 'false'
profile2 = BrowserProfile(headless=True)
assert profile2.enable_default_extensions is True, (
'BrowserProfile should enable extensions when BROWSER_USE_DISABLE_EXTENSIONS=false'
)
finally:
if original is not None:
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original
else:
os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None)
def test_explicit_param_overrides_env_var(self):
"""Test that explicit enable_default_extensions parameter overrides env var."""
original = os.environ.get('BROWSER_USE_DISABLE_EXTENSIONS')
try:
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = 'true'
from browser_use.browser.profile import BrowserProfile
# Explicitly set to True should override env var
profile = BrowserProfile(headless=True, enable_default_extensions=True)
assert profile.enable_default_extensions is True, 'Explicit param should override env var'
finally:
if original is not None:
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original
else:
os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None)
def test_browser_session_uses_env_var(self):
"""Test that BrowserSession picks up the env var via BrowserProfile."""
original = os.environ.get('BROWSER_USE_DISABLE_EXTENSIONS')
try:
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = '1'
from browser_use.browser import BrowserSession
session = BrowserSession(headless=True)
assert session.browser_profile.enable_default_extensions is False, (
'BrowserSession should disable extensions when BROWSER_USE_DISABLE_EXTENSIONS=1'
)
finally:
if original is not None:
os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original
else:
os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None)
@pytest.mark.parametrize(
'env_var,attr_name,env_val,explicit_arg,expected',
[
('BROWSER_USE_DISABLE_EXTENSIONS', 'enable_default_extensions', 'true', {'enable_default_extensions': True}, True),
('BROWSER_USE_DISABLE_EXTENSIONS', 'enable_default_extensions', 'false', {'enable_default_extensions': False}, False),
('BROWSER_USE_HEADLESS', 'headless', 'true', {'headless': False}, False),
('BROWSER_USE_HEADLESS', 'headless', 'false', {'headless': True}, True),
],
)
def test_explicit_parameter_overrides_env_var(
self,
monkeypatch: pytest.MonkeyPatch,
env_var: str,
attr_name: str,
env_val: str,
explicit_arg: dict,
expected: bool,
):
"""Test that explicit constructor parameters override env vars."""
monkeypatch.setenv(env_var, env_val)
profile = BrowserProfile(**explicit_arg)
assert getattr(profile, attr_name) is expected
@@ -0,0 +1,193 @@
from browser_use.dom.serializer.serializer import DOMTreeSerializer
from browser_use.dom.views import (
DEFAULT_INCLUDE_ATTRIBUTES,
DOMRect,
EnhancedDOMTreeNode,
EnhancedSnapshotNode,
NodeType,
SimplifiedNode,
)
def _make_element_node(
backend_node_id: int,
tag_name: str,
attributes: dict[str, str],
x: float,
y: float,
width: float = 64,
height: float = 64,
parent: EnhancedDOMTreeNode | None = None,
) -> EnhancedDOMTreeNode:
bounds = DOMRect(x=x, y=y, width=width, height=height)
return EnhancedDOMTreeNode(
node_id=backend_node_id,
backend_node_id=backend_node_id,
node_type=NodeType.ELEMENT_NODE,
node_name=tag_name.upper(),
node_value='',
attributes=attributes,
is_scrollable=None,
is_visible=True,
absolute_position=bounds,
target_id='target-1',
frame_id=None,
session_id=None,
content_document=None,
shadow_root_type=None,
shadow_roots=None,
parent_node=parent,
children_nodes=None,
ax_node=None,
snapshot_node=EnhancedSnapshotNode(
is_clickable=tag_name in {'a', 'button'},
cursor_style='pointer' if tag_name in {'a', 'button'} else None,
bounds=bounds,
clientRects=bounds,
scrollRects=None,
computed_styles=None,
paint_order=None,
stacking_contexts=None,
),
)
def test_image_only_interactive_parent_includes_child_image_context_in_llm_dom():
"""Image-only clickable cards should expose child image context."""
link = _make_element_node(201, 'a', {'href': '/select-payment-method'}, x=10, y=10, width=80, height=80)
image = _make_element_node(
202,
'img',
{'src': 'https://cdn.example.test/logos/acme-bank-primary-card.png'},
x=18,
y=18,
width=64,
height=64,
parent=link,
)
link.children_nodes = [image]
llm_dom = DOMTreeSerializer.serialize_tree(
SimplifiedNode(
original_node=link,
children=[SimplifiedNode(original_node=image, children=[])],
is_interactive=True,
selector_index=201,
),
DEFAULT_INCLUDE_ATTRIBUTES,
)
assert '[201]<a' in llm_dom
assert 'acme-bank-primary-card.png' in llm_dom
def _simplified_image(backend_node_id: int, attributes: dict[str, str]) -> SimplifiedNode:
image = _make_element_node(backend_node_id, 'img', attributes, x=18, y=18)
return SimplifiedNode(original_node=image, children=[])
class _NoEagerReverseList(list[SimplifiedNode]):
def __reversed__(self):
raise AssertionError('child lists must be traversed lazily')
def test_child_image_context_finds_images_below_non_interactive_wrappers():
parent = _make_element_node(251, 'a', {'href': '/cards'}, x=10, y=10)
wrapper = _make_element_node(252, 'div', {}, x=12, y=12)
node = SimplifiedNode(
original_node=parent,
children=[
SimplifiedNode(
original_node=wrapper,
children=[_simplified_image(253, {'src': '/nested-card.png'})],
),
],
)
context = DOMTreeSerializer._get_child_image_context(node)
assert context == 'image_src=nested-card.png'
def test_child_image_context_strips_query_and_fragment_without_leaking_query_only_sources():
parent = _make_element_node(301, 'a', {'href': '/cards'}, x=10, y=10)
node = SimplifiedNode(
original_node=parent,
children=[
_simplified_image(302, {'src': 'https://cdn.test/card.png?token=secret#preview'}),
_simplified_image(303, {'src': '?token=must-not-leak'}),
],
)
context = DOMTreeSerializer._get_child_image_context(node)
assert context == 'image_src=card.png'
assert 'secret' not in context
assert 'must-not-leak' not in context
def test_child_image_context_ignores_data_src_but_keeps_accessible_attributes():
parent = _make_element_node(401, 'button', {}, x=10, y=10)
node = SimplifiedNode(
original_node=parent,
children=[
_simplified_image(
402,
{
'src': 'data:image/png;base64,private-payload',
'alt': 'Payment card',
'title': 'Choose card',
'aria-label': 'Primary payment method',
},
),
],
)
context = DOMTreeSerializer._get_child_image_context(node)
assert context == 'image_alt=Payment card image_title=Choose card image_label=Primary payment method'
assert 'data:' not in context
assert 'private-payload' not in context
def test_child_image_context_limits_returned_images():
parent = _make_element_node(501, 'a', {'href': '/gallery'}, x=10, y=10)
node = SimplifiedNode(
original_node=parent,
children=[_simplified_image(502 + index, {'src': f'/image-{index}.png'}) for index in range(4)],
)
context = DOMTreeSerializer._get_child_image_context(node)
assert 'image-0.png' in context
assert 'image-1.png' in context
assert 'image-2.png' in context
assert 'image-3.png' not in context
def test_child_image_context_caps_traversal_even_when_images_have_no_context():
parent = _make_element_node(601, 'a', {'href': '/gallery'}, x=10, y=10)
ignored_images = [
_simplified_image(602 + index, {'src': 'data:image/png;base64,ignored'})
for index in range(DOMTreeSerializer.MAX_CHILD_IMAGE_DESCENDANTS)
]
node = SimplifiedNode(
original_node=parent,
children=[*ignored_images, _simplified_image(999, {'src': '/too-deep.png'})],
)
context = DOMTreeSerializer._get_child_image_context(node)
assert context == ''
def test_child_image_context_does_not_copy_wide_child_lists_before_traversal():
parent = _make_element_node(701, 'a', {'href': '/gallery'}, x=10, y=10)
node = SimplifiedNode(original_node=parent, children=[])
node.children = _NoEagerReverseList(
_simplified_image(702 + index, {'src': 'data:image/png;base64,ignored'}) for index in range(200)
)
context = DOMTreeSerializer._get_child_image_context(node)
assert context == ''