From 2c3162069e5a07d62f72509b3dd9ac2643a3099d Mon Sep 17 00:00:00 2001 From: Shantanav Date: Mon, 3 Aug 2026 23:20:34 +0530 Subject: [PATCH 01/69] Fix reconnect drop handling during reconnect Record WebSocket drops that occur during an in-flight reconnect and schedule a retry after the reconnect finishes.\n\nFixes #5366 --- browser_use/browser/session.py | 18 ++++++++++++++++-- tests/ci/test_browser_session.py | 23 +++++++++++++++++++++++ 2 files changed, 39 insertions(+), 2 deletions(-) create mode 100644 tests/ci/test_browser_session.py diff --git a/browser_use/browser/session.py b/browser_use/browser/session.py index baef63f48..c0c7f755e 100644 --- a/browser_use/browser/session.py +++ b/browser_use/browser/session.py @@ -590,6 +590,7 @@ class BrowserSession(BaseModel): _reconnect_event: asyncio.Event = PrivateAttr(default_factory=asyncio.Event) _reconnect_lock: asyncio.Lock = PrivateAttr(default_factory=asyncio.Lock) _reconnect_task: asyncio.Task | None = PrivateAttr(default=None) + _reconnect_pending: bool = PrivateAttr(default=False) _intentional_stop: bool = PrivateAttr(default=False) _logger: Any = PrivateAttr(default=None) @@ -634,6 +635,7 @@ class BrowserSession(BaseModel): if self._reconnect_task and not self._reconnect_task.done(): self._reconnect_task.cancel() self._reconnect_task = None + self._reconnect_pending = False self._reconnecting = False self._reconnect_event.set() # unblock any waiters @@ -2291,9 +2293,18 @@ class BrowserSession(BaseModel): ) ) finally: + reconnect_pending = self._reconnect_pending + self._reconnect_pending = False self._reconnecting = False self._reconnect_event.set() # wake up all waiters regardless of outcome + if reconnect_pending and not self._intentional_stop and self.cdp_url: + try: + loop = asyncio.get_running_loop() + self._reconnect_task = loop.create_task(self._auto_reconnect()) + except RuntimeError: + self.logger.error('๐Ÿ”Œ No event loop available for pending auto-reconnect') + def _attach_ws_drop_callback(self) -> None: """Attach a done callback to the CDPClient's message handler task to detect WS drops.""" if not self._cdp_client_root or not hasattr(self._cdp_client_root, '_message_handler_task'): @@ -2304,8 +2315,11 @@ class BrowserSession(BaseModel): return def _on_message_handler_done(fut: asyncio.Future) -> None: - # Guard: skip if intentionally stopped, already reconnecting, or no cdp_url - if self._intentional_stop or self._reconnecting or not self.cdp_url: + # Guard: skip if intentionally stopped or no cdp_url + if self._intentional_stop or not self.cdp_url: + return + if self._reconnecting: + self._reconnect_pending = True return # The message handler task exiting means the WS connection dropped diff --git a/tests/ci/test_browser_session.py b/tests/ci/test_browser_session.py new file mode 100644 index 000000000..e07c81a3f --- /dev/null +++ b/tests/ci/test_browser_session.py @@ -0,0 +1,23 @@ +import asyncio + +from browser_use.browser.profile import BrowserProfile +from browser_use.browser.session import BrowserSession + + +class MessageHandlerClient: + def __init__(self, task: asyncio.Task) -> None: + self._message_handler_task = task + + +async def test_ws_drop_during_reconnect_is_recorded_for_retry() -> None: + session = BrowserSession(browser_profile=BrowserProfile(headless=True, user_data_dir=None, cdp_url='ws://127.0.0.1:9222')) + + session._reconnecting = True + connection_closed = asyncio.get_running_loop().create_future() + task = asyncio.ensure_future(connection_closed) + session._cdp_client_root = MessageHandlerClient(task) # type: ignore[assignment] + session._attach_ws_drop_callback() + connection_closed.set_exception(ConnectionResetError('ws dropped again')) + await asyncio.sleep(0) + connection_closed.exception() + assert session._reconnect_pending From 97c25801d6e75051fd35275d1e94f631f0e5bb18 Mon Sep 17 00:00:00 2001 From: Shantanav Date: Mon, 3 Aug 2026 23:54:23 +0530 Subject: [PATCH 02/69] test: cover reconnect retry after websocket drop --- tests/ci/test_browser_session.py | 26 +++++++++++++++++++++++--- 1 file changed, 23 insertions(+), 3 deletions(-) diff --git a/tests/ci/test_browser_session.py b/tests/ci/test_browser_session.py index e07c81a3f..0fe4bb978 100644 --- a/tests/ci/test_browser_session.py +++ b/tests/ci/test_browser_session.py @@ -9,15 +9,35 @@ class MessageHandlerClient: self._message_handler_task = task -async def test_ws_drop_during_reconnect_is_recorded_for_retry() -> None: +async def test_ws_drop_during_reconnect_triggers_follow_up_attempt(monkeypatch) -> None: session = BrowserSession(browser_profile=BrowserProfile(headless=True, user_data_dir=None, cdp_url='ws://127.0.0.1:9222')) - session._reconnecting = True + reconnect_started = asyncio.Event() + allow_reconnect = asyncio.Event() + reconnect_attempts = 0 + + async def reconnect(self: BrowserSession) -> None: + nonlocal reconnect_attempts + reconnect_attempts += 1 + if reconnect_attempts == 1: + reconnect_started.set() + await allow_reconnect.wait() + + monkeypatch.setattr(BrowserSession, 'reconnect', reconnect) + connection_closed = asyncio.get_running_loop().create_future() task = asyncio.ensure_future(connection_closed) session._cdp_client_root = MessageHandlerClient(task) # type: ignore[assignment] + initial_reconnect = asyncio.create_task(session._auto_reconnect(max_attempts=1)) + await reconnect_started.wait() session._attach_ws_drop_callback() connection_closed.set_exception(ConnectionResetError('ws dropped again')) await asyncio.sleep(0) connection_closed.exception() - assert session._reconnect_pending + allow_reconnect.set() + await initial_reconnect + + assert session._reconnect_task is not None + await session._reconnect_task + assert reconnect_attempts == 2 + await session.event_bus.stop(clear=True, timeout=5) From 9b897408c47e8f6757647c9fb38593628d242efe Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Mon, 3 Aug 2026 16:10:13 -0700 Subject: [PATCH 03/69] Bump dependency versions --- pyproject.toml | 24 ++++++++++++------------ 1 file changed, 12 insertions(+), 12 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 92f970dea..b7b2842c9 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -11,7 +11,7 @@ classifiers = [ "Operating System :: OS Independent", ] dependencies = [ - "aiohttp==3.13.4", + "aiohttp==3.14.1", "anyio==4.12.1", "bubus==1.5.6", "click==8.3.1", @@ -29,24 +29,24 @@ dependencies = [ "typing-extensions==4.15.0", "uuid7==0.1.0", "google-genai==1.65.0", - "openai==2.16.0", + "openai==2.26.0", "anthropic==0.76.0", "groq==1.0.0", "ollama==0.6.1", "google-api-python-client==2.188.0", "google-auth==2.48.0", "google-auth-oauthlib==1.2.4", - "mcp==1.26.0", - "pypdf==6.10.2", + "mcp==1.28.1", + "pypdf==6.14.2", "reportlab==4.4.9", "cdp-use==1.4.5", "pyotp==2.9.0", - "pillow==12.2.0", + "pillow==12.3.0", "cloudpickle==3.1.2", "markdownify==1.2.2", "python-docx==1.2.0", "browser-use-sdk==3.4.2", - "browser-harness==0.1.8", + "browser-harness==0.1.9", ] # google-api-core: only used for Google LLM APIs # pyperclip: only used for examples that use copy/paste @@ -60,11 +60,11 @@ dependencies = [ [project.optional-dependencies] cli = [] core = [ - "browser-use-core==0.13.2; sys_platform == 'darwin' and platform_machine == 'arm64'", - "browser-use-core==0.13.2; sys_platform == 'darwin' and platform_machine == 'x86_64'", - "browser-use-core==0.13.2; sys_platform == 'linux' and platform_machine == 'x86_64'", - "browser-use-core==0.13.2; sys_platform == 'linux' and platform_machine == 'aarch64'", - "browser-use-core==0.13.2; sys_platform == 'win32' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')", + "browser-use-core==0.13.3; sys_platform == 'darwin' and platform_machine == 'arm64'", + "browser-use-core==0.13.3; sys_platform == 'darwin' and platform_machine == 'x86_64'", + "browser-use-core==0.13.3; sys_platform == 'linux' and platform_machine == 'x86_64'", + "browser-use-core==0.13.3; sys_platform == 'linux' and platform_machine == 'aarch64'", + "browser-use-core==0.13.3; sys_platform == 'win32' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')", ] aws = ["boto3==1.42.37"] oci = ["oci==2.166.0"] @@ -76,7 +76,7 @@ examples = [ "imgcat==0.6.0", # "stagehand-py>=0.3.6", # "browserbase>=0.4.0", - "langchain-openai==1.1.7", + "langchain-openai==1.1.14", ] eval = [ "lmnr[all]==0.7.42", From ca19dd1626f3ca38c18dfab85964805647f04443 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 13 Aug 2026 10:05:08 -0700 Subject: [PATCH 04/69] docs: refresh README demo embeds --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 9dfdae628..e311fde73 100644 --- a/README.md +++ b/README.md @@ -55,7 +55,7 @@ Browser Use lets an AI agent use a web browser the same way you do โ€” it opens ### ๐ŸŽ Extract data #### Task: "Extract structured data about my followers and export it as a CSV." -https://github.com/user-attachments/assets/93714c75-98f4-4cfc-add1-69c38b5138b5 +![Social Data Extraction Demo](https://github.com/user-attachments/assets/93714c75-98f4-4cfc-add1-69c38b5138b5) [Browser Use Cloud Docs โ†—](https://docs.browser-use.com/cloud/quickstart) From f1f9ed85a3680f526d0fd744955fbc486eab3c00 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 13 Aug 2026 10:06:48 -0700 Subject: [PATCH 05/69] docs: describe human browser use --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index e311fde73..07d164b7e 100644 --- a/README.md +++ b/README.md @@ -41,7 +41,7 @@ # What can Browser Use do? -Browser Use lets an AI agent use a web browser the same way you do โ€” it opens pages, clicks buttons, types, and fills in forms. You describe the task, and it completes it. For example, you can have it: +Browser Use lets an AI agent use a web browser the same way humans do โ€” it opens pages, clicks buttons, types, and fills in forms. You describe the task, and it completes it. For example, you can have it: ### ๐Ÿ“‹ Fill Forms From b0afe30990464f9dd9c3530049cfec8c88fb0775 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 13 Aug 2026 10:07:19 -0700 Subject: [PATCH 06/69] docs: remove QA automation demo --- README.md | 8 -------- 1 file changed, 8 deletions(-) diff --git a/README.md b/README.md index 07d164b7e..18d43a864 100644 --- a/README.md +++ b/README.md @@ -60,14 +60,6 @@ Browser Use lets an AI agent use a web browser the same way humans do โ€” it ope [Browser Use Cloud Docs โ†—](https://docs.browser-use.com/cloud/quickstart) -### ๐Ÿ’ป QA Automation -#### Task: "QA test my local website and report any bugs, usability issues, and visual inconsistencies." - -qa-demo-small - -[Browser Use CLI โ†—](https://docs.browser-use.com/open-source/browser-use-cli) - -
# Quickstart From a8a45554308a003da2e64aa49b00d9c891a3bfdc Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 13 Aug 2026 10:22:39 -0700 Subject: [PATCH 07/69] docs: use trimmed README demo assets --- README.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 18d43a864..833b6832d 100644 --- a/README.md +++ b/README.md @@ -47,7 +47,7 @@ Browser Use lets an AI agent use a web browser the same way humans do โ€” it ope ### ๐Ÿ“‹ Fill Forms #### Task: "Fill in this job application with my resume and information." -![Job Application Demo](https://github.com/user-attachments/assets/57865ee6-6004-49d5-b2c2-6dff39ec2ba9) +![Job Application Demo](https://github.com/user-attachments/assets/57611d8e-0474-4de6-84b7-37a0c0cd27e7) [Example code โ†—](https://github.com/browser-use/browser-use/blob/main/examples/use-cases/apply_to_job.py) @@ -55,7 +55,7 @@ Browser Use lets an AI agent use a web browser the same way humans do โ€” it ope ### ๐ŸŽ Extract data #### Task: "Extract structured data about my followers and export it as a CSV." -![Social Data Extraction Demo](https://github.com/user-attachments/assets/93714c75-98f4-4cfc-add1-69c38b5138b5) +![Social Data Extraction Demo](https://github.com/user-attachments/assets/485fd3ec-61b9-4afc-9e86-ee9b85acb592) [Browser Use Cloud Docs โ†—](https://docs.browser-use.com/cloud/quickstart) From 55bc98dfcb14cca97c6e5811c1018ea321cbe0cc Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 13 Aug 2026 10:33:29 -0700 Subject: [PATCH 08/69] docs: preserve social video preview --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 833b6832d..6d89895ee 100644 --- a/README.md +++ b/README.md @@ -55,7 +55,7 @@ Browser Use lets an AI agent use a web browser the same way humans do โ€” it ope ### ๐ŸŽ Extract data #### Task: "Extract structured data about my followers and export it as a CSV." -![Social Data Extraction Demo](https://github.com/user-attachments/assets/485fd3ec-61b9-4afc-9e86-ee9b85acb592) +https://github.com/user-attachments/assets/485fd3ec-61b9-4afc-9e86-ee9b85acb592 [Browser Use Cloud Docs โ†—](https://docs.browser-use.com/cloud/quickstart) From 9be28bcea151f572462c0fcb7a9bc13755421029 Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Thu, 13 Aug 2026 10:51:12 -0700 Subject: [PATCH 09/69] feat(llm): default ChatBrowserUse to bu-2-0-mini-preview Adds bu-2-0-mini-preview as an accepted model id and makes it the constructor default, so a bare ChatBrowserUse() now routes there. bu-2-0 is unchanged: still accepted, still documented as the premium option, and bu-latest still resolves to it. Keeping 'latest' on the stable line means existing callers pinned to that alias do not silently move onto a preview model; only the bare-constructor default moves. Pricing is registered alongside the model so cost tracking does not silently report $0 for what is now the default. This model has no cache discount, so cached reads bill at the input rate. Examples, README and the model reference are updated to the new default. The model reference also claimed bu-latest resolved to bu-1-0, which has not been true since bu-2-0 shipped; corrected here. --- README.md | 2 +- browser_use/llm/browser_use/chat.py | 10 ++++--- browser_use/llm/models.py | 4 ++- browser_use/tokens/custom_pricing.py | 10 +++++++ examples/beta_agent/basic.py | 2 +- examples/browser/cloud_browser.py | 4 +-- examples/browser/custom_headers.py | 2 +- examples/demo_mode_example.py | 2 +- examples/features/csv_file_generation.py | 2 +- examples/features/judge_trace.py | 2 +- examples/features/rerun_history.py | 2 +- examples/features/save_as_pdf.py | 2 +- examples/getting_started/01_basic_search.py | 2 +- examples/getting_started/02_form_filling.py | 2 +- .../getting_started/03_data_extraction.py | 2 +- .../getting_started/04_multi_step_task.py | 2 +- examples/integrations/agentmail/2fa.py | 2 +- examples/models/browser_use_llm.py | 10 ++++--- examples/sandbox/example.py | 2 +- examples/sandbox/structured_output.py | 2 +- examples/simple.py | 2 +- examples/use-cases/buy_groceries.py | 2 +- examples/use-cases/pcpartpicker.py | 2 +- examples/use-cases/phone_comparison.py | 2 +- skills/open-source/references/models.md | 8 +++-- tests/ci/models/test_llm_browseruse.py | 29 ++++++++++++++++--- 26 files changed, 76 insertions(+), 37 deletions(-) diff --git a/README.md b/README.md index 9dfdae628..31286952f 100644 --- a/README.md +++ b/README.md @@ -113,7 +113,7 @@ async def main(): agent = Agent( task="Find the number of stars of the browser-use repo", llm=ChatBrowserUse(model='openai/gpt-5.5'), - # llm=ChatBrowserUse(model='bu-2-0'), # Browser Use's optimized model + # llm=ChatBrowserUse(model='bu-2-0-mini-preview'), # Browser Use's optimized model # llm=ChatOpenAI(model='gpt-5.5'), # llm=ChatAnthropic(model='claude-opus-4-8'), # Sonnet also works well ) diff --git a/browser_use/llm/browser_use/chat.py b/browser_use/llm/browser_use/chat.py index 6f0d0c4b6..25e108f8b 100644 --- a/browser_use/llm/browser_use/chat.py +++ b/browser_use/llm/browser_use/chat.py @@ -44,7 +44,7 @@ class ChatBrowserUse(BaseChatModel): def __init__( self, - model: str = 'bu-2-0', + model: str = 'bu-2-0-mini-preview', api_key: str | None = None, base_url: str | None = None, timeout: float = 120.0, @@ -58,7 +58,8 @@ class ChatBrowserUse(BaseChatModel): Args: model: Model name to use. Options: - - 'bu-2-0' or 'bu-latest': Default model (latest premium) + - 'bu-2-0-mini-preview': Default model (fast + cheap, preview) + - 'bu-2-0' or 'bu-latest': Premium model - 'bu-1-0': Previous generation model - 'bu-qa-1': Website QA model (tests a site and scores functionality/aesthetics) - 'browser-use/bu-30b-a3b-preview': Browser Use Open Source Model @@ -73,7 +74,7 @@ class ChatBrowserUse(BaseChatModel): """ # Accept 'bu-*' aliases and any provider-prefixed id; the gateway resolves the # latter (anthropic/*, openai/*, google/*, browser-use/*), so we don't enumerate them. - bu_aliases = ['bu-latest', 'bu-1-0', 'bu-2-0', 'bu-qa-1'] + bu_aliases = ['bu-latest', 'bu-1-0', 'bu-2-0', 'bu-2-0-mini-preview', 'bu-qa-1'] is_valid = model in bu_aliases or '/' in model if not is_valid: raise ValueError( @@ -82,7 +83,8 @@ class ChatBrowserUse(BaseChatModel): "'openai/gpt-5.5', or 'google/gemini-3-pro'." ) - # Normalize bu-latest to the current latest model + # Normalize bu-latest to the current latest model. Deliberately not the constructor + # default: 'latest' tracks the stable premium line, not the preview. if model == 'bu-latest': self.model = 'bu-2-0' else: diff --git a/browser_use/llm/models.py b/browser_use/llm/models.py index 81a277731..81edf9945 100644 --- a/browser_use/llm/models.py +++ b/browser_use/llm/models.py @@ -8,7 +8,7 @@ Usage: model = llm.azure_gpt_4_1_mini model = llm.openai_gpt_4o model = llm.google_gemini_2_5_pro - model = llm.bu_latest # or bu_1_0, bu_2_0 + model = llm.bu_latest # or bu_2_0_mini_preview, bu_2_0, bu_1_0 """ import os @@ -83,6 +83,7 @@ cerebras_gemma_4_31b: 'BaseChatModel' bu_latest: 'BaseChatModel' bu_1_0: 'BaseChatModel' bu_2_0: 'BaseChatModel' +bu_2_0_mini_preview: 'BaseChatModel' def get_llm_by_name(model_name: str): @@ -319,6 +320,7 @@ __all__ += [ 'bu_latest', 'bu_1_0', 'bu_2_0', + 'bu_2_0_mini_preview', ] # NOTE: OCI backend is optional. The try/except ImportError and conditional __all__ are required diff --git a/browser_use/tokens/custom_pricing.py b/browser_use/tokens/custom_pricing.py index 3bb00c905..b625ea758 100644 --- a/browser_use/tokens/custom_pricing.py +++ b/browser_use/tokens/custom_pricing.py @@ -27,6 +27,16 @@ CUSTOM_MODEL_PRICING: dict[str, dict[str, Any]] = { 'max_input_tokens': None, # Not specified 'max_output_tokens': None, # Not specified }, + 'bu-2-0-mini-preview': { + 'input_cost_per_token': 0.15 / 1_000_000, # $0.15 per 1M tokens + 'output_cost_per_token': 1.50 / 1_000_000, # $1.50 per 1M tokens + # No cache discount on this model: cached reads bill at the input rate. + 'cache_read_input_token_cost': 0.15 / 1_000_000, # $0.15 per 1M tokens + 'cache_creation_input_token_cost': None, # Not specified + 'max_tokens': None, # Not specified + 'max_input_tokens': None, # Not specified + 'max_output_tokens': None, # Not specified + }, 'claude-sonnet-4-6': { 'input_cost_per_token': 3.00 / 1_000_000, 'output_cost_per_token': 15.00 / 1_000_000, diff --git a/examples/beta_agent/basic.py b/examples/beta_agent/basic.py index 1eed65ca4..bb7f7a8e0 100644 --- a/examples/beta_agent/basic.py +++ b/examples/beta_agent/basic.py @@ -23,7 +23,7 @@ async def main() -> None: agent = Agent( task=task, llm=ChatBrowserUse(model='openai/gpt-5.5'), - # llm=ChatBrowserUse(), # Browser Use's own optimized model (bu-2-0) + # llm=ChatBrowserUse(), # Browser Use's own optimized model (bu-2-0-mini-preview) # llm=ChatOpenAI(model='gpt-5.5'), # llm=ChatGoogle(model='gemini-3.1-pro-preview'), # llm=ChatAnthropic(model='claude-opus-4-8'), # Sonnet also works well. diff --git a/examples/browser/cloud_browser.py b/examples/browser/cloud_browser.py index 72eebb9a9..f7bce8520 100644 --- a/examples/browser/cloud_browser.py +++ b/examples/browser/cloud_browser.py @@ -21,7 +21,7 @@ async def basic(): agent = Agent( task='Go to github.com/browser-use/browser-use and tell me the star count', - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), browser=browser, ) @@ -39,7 +39,7 @@ async def full_config(): agent = Agent( task='go and check my ip address and the location', - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), browser=browser, ) diff --git a/examples/browser/custom_headers.py b/examples/browser/custom_headers.py index d1f4b5cd5..7f6ff4636 100644 --- a/examples/browser/custom_headers.py +++ b/examples/browser/custom_headers.py @@ -90,7 +90,7 @@ async def main(): 'Open https://httpbin.org/headers in two different tabs and extract the full JSON response. ' 'Look for the custom headers X-Custom-Auth, X-Request-Source, and X-Trace-Id in the output and compare the results.' ), - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), browser=browser, ) diff --git a/examples/demo_mode_example.py b/examples/demo_mode_example.py index b6c540ba3..b0bbe546e 100644 --- a/examples/demo_mode_example.py +++ b/examples/demo_mode_example.py @@ -6,7 +6,7 @@ from browser_use import Agent, ChatBrowserUse async def main() -> None: agent = Agent( task='Please find the latest commit on browser-use/browser-use repo and tell me the commit message. Please summarize what it is about.', - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), demo_mode=True, ) await agent.run(max_steps=5) diff --git a/examples/features/csv_file_generation.py b/examples/features/csv_file_generation.py index efecc5480..7119a36c2 100644 --- a/examples/features/csv_file_generation.py +++ b/examples/features/csv_file_generation.py @@ -33,7 +33,7 @@ async def main(): 'Create a CSV file called "top_cities.csv" with columns: rank, city name, country, population. ' 'Make sure to include all cities even if some data is missing โ€” leave those cells empty.' ), - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), ) history = await agent.run() diff --git a/examples/features/judge_trace.py b/examples/features/judge_trace.py index c20030041..3826422f3 100644 --- a/examples/features/judge_trace.py +++ b/examples/features/judge_trace.py @@ -27,7 +27,7 @@ Round your result to the nearest 1000 hours and do not use any comma separators async def main(): - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') agent = Agent( task=task, llm=llm, diff --git a/examples/features/rerun_history.py b/examples/features/rerun_history.py index af30cd1c1..6f8fbaa52 100644 --- a/examples/features/rerun_history.py +++ b/examples/features/rerun_history.py @@ -59,7 +59,7 @@ async def main(): # Example task to demonstrate history saving and rerunning history_file = Path('agent_history.json') task = 'Go to https://browser-use.github.io/stress-tests/challenges/reference-number-form.html and fill the form with example data and submit and extract the refernence number.' - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Optional: Use custom LLMs for AI features during rerun # Uncomment to use a custom LLM: diff --git a/examples/features/save_as_pdf.py b/examples/features/save_as_pdf.py index fd5f71658..9b86e024b 100644 --- a/examples/features/save_as_pdf.py +++ b/examples/features/save_as_pdf.py @@ -34,7 +34,7 @@ async def main(): 'Go to https://news.ycombinator.com and save the front page as a PDF named "hackernews". ' 'Then go to https://en.wikipedia.org/wiki/Web_browser and save just that article as a PDF in A4 format.' ), - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), ) history = await agent.run() diff --git a/examples/getting_started/01_basic_search.py b/examples/getting_started/01_basic_search.py index a916287ab..ba4f26f4c 100644 --- a/examples/getting_started/01_basic_search.py +++ b/examples/getting_started/01_basic_search.py @@ -19,7 +19,7 @@ from browser_use import Agent, ChatBrowserUse async def main(): - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') task = "Search Google for 'what is browser automation' and tell me the top 3 results" agent = Agent(task=task, llm=llm) await agent.run() diff --git a/examples/getting_started/02_form_filling.py b/examples/getting_started/02_form_filling.py index 97e44d78d..22b59cfb4 100644 --- a/examples/getting_started/02_form_filling.py +++ b/examples/getting_started/02_form_filling.py @@ -30,7 +30,7 @@ from browser_use import Agent, ChatBrowserUse async def main(): # Initialize the model - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Define a form filling task task = """ diff --git a/examples/getting_started/03_data_extraction.py b/examples/getting_started/03_data_extraction.py index 18c8b51cd..c204967f3 100644 --- a/examples/getting_started/03_data_extraction.py +++ b/examples/getting_started/03_data_extraction.py @@ -30,7 +30,7 @@ from browser_use import Agent, ChatBrowserUse async def main(): # Initialize the model - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Define a data extraction task task = """ diff --git a/examples/getting_started/04_multi_step_task.py b/examples/getting_started/04_multi_step_task.py index 5855f9087..2bba002d9 100644 --- a/examples/getting_started/04_multi_step_task.py +++ b/examples/getting_started/04_multi_step_task.py @@ -30,7 +30,7 @@ from browser_use import Agent, ChatBrowserUse async def main(): # Initialize the model - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Define a multi-step task task = """ diff --git a/examples/integrations/agentmail/2fa.py b/examples/integrations/agentmail/2fa.py index 60c1ff83f..3130ad25a 100644 --- a/examples/integrations/agentmail/2fa.py +++ b/examples/integrations/agentmail/2fa.py @@ -28,7 +28,7 @@ async def main(): tools = EmailTools(email_client=email_client, inbox=inbox) # Initialize the LLM for browser-use agent - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Set your local browser path browser = Browser(executable_path='/Applications/Google Chrome.app/Contents/MacOS/Google Chrome') diff --git a/examples/models/browser_use_llm.py b/examples/models/browser_use_llm.py index 4e38698e9..3cb441039 100644 --- a/examples/models/browser_use_llm.py +++ b/examples/models/browser_use_llm.py @@ -20,12 +20,14 @@ if not os.getenv('BROWSER_USE_API_KEY'): async def main(): - # `bu-2-0` is the optimized default. ChatBrowserUse can also route to - # provider-prefixed models (e.g. 'anthropic/claude-sonnet-4-6', 'openai/gpt-5.5', - # 'google/gemini-3-pro') through the same gateway - see browser_use_provider_models.py. + # `bu-2-0-mini-preview` is the optimized default - what a bare `ChatBrowserUse()` gives you. + # For the larger premium model pass `model='bu-2-0'` (or 'bu-latest', which tracks it). + # ChatBrowserUse can also route to provider-prefixed models (e.g. 'anthropic/claude-sonnet-4-6', + # 'openai/gpt-5.5', 'google/gemini-3-pro') through the same gateway - see + # browser_use_provider_models.py. agent = Agent( task='Find the number of stars of the browser-use repo', - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), ) # Run the agent diff --git a/examples/sandbox/example.py b/examples/sandbox/example.py index 565a70035..7182bd2bc 100644 --- a/examples/sandbox/example.py +++ b/examples/sandbox/example.py @@ -37,7 +37,7 @@ async def pydantic_example(browser: Browser): agent = Agent( """go and check my ip address and the location. return the result in json format""", browser=browser, - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), ) res = await agent.run() diff --git a/examples/sandbox/structured_output.py b/examples/sandbox/structured_output.py index 14a8c70f5..a4e2fae78 100644 --- a/examples/sandbox/structured_output.py +++ b/examples/sandbox/structured_output.py @@ -29,7 +29,7 @@ async def get_ip_location(browser: Browser) -> AgentHistoryList: agent = Agent( task='Go to ipinfo.io and extract my IP address and location details (country, city, region)', browser=browser, - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), output_model_schema=IPLocation, ) return await agent.run(max_steps=10) diff --git a/examples/simple.py b/examples/simple.py index 449237819..ec8b9176c 100644 --- a/examples/simple.py +++ b/examples/simple.py @@ -12,6 +12,6 @@ load_dotenv() agent = Agent( task='Find the number of stars of the following repos: browser-use, playwright, stagehand, react, nextjs', - llm=ChatBrowserUse(model='bu-2-0'), + llm=ChatBrowserUse(model='bu-2-0-mini-preview'), ) agent.run_sync() diff --git a/examples/use-cases/buy_groceries.py b/examples/use-cases/buy_groceries.py index d2cb6852a..39e4ed8cc 100644 --- a/examples/use-cases/buy_groceries.py +++ b/examples/use-cases/buy_groceries.py @@ -24,7 +24,7 @@ class GroceryCart(BaseModel): async def add_to_cart(items: list[str] = ['milk', 'eggs', 'bread']): browser = Browser(cdp_url='http://localhost:9222') - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Task prompt task = f""" diff --git a/examples/use-cases/pcpartpicker.py b/examples/use-cases/pcpartpicker.py index c332c150f..d6752f04d 100644 --- a/examples/use-cases/pcpartpicker.py +++ b/examples/use-cases/pcpartpicker.py @@ -6,7 +6,7 @@ from browser_use import Agent, Browser, ChatBrowserUse, Tools async def main(): browser = Browser(cdp_url='http://localhost:9222') - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') tools = Tools() diff --git a/examples/use-cases/phone_comparison.py b/examples/use-cases/phone_comparison.py index 65157d1c2..a52fa6d25 100644 --- a/examples/use-cases/phone_comparison.py +++ b/examples/use-cases/phone_comparison.py @@ -34,7 +34,7 @@ async def find(item: str = 'Used iPhone 12'): """ browser = Browser(cdp_url='http://localhost:9222') - llm = ChatBrowserUse(model='bu-2-0') + llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Task prompt task = f""" diff --git a/skills/open-source/references/models.md b/skills/open-source/references/models.md index ad277d76b..8193b731d 100644 --- a/skills/open-source/references/models.md +++ b/skills/open-source/references/models.md @@ -59,8 +59,9 @@ Optimized for browser automation โ€” highest accuracy, fastest speed, lowest tok ```python from browser_use import Agent, ChatBrowserUse -llm = ChatBrowserUse() # bu-latest (default) -llm = ChatBrowserUse(model='bu-2-0') # Premium model +llm = ChatBrowserUse() # bu-2-0-mini-preview (default) +llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Default, named explicitly +llm = ChatBrowserUse(model='bu-2-0') # Premium model ('bu-latest' tracks this) ``` **Env:** `BROWSER_USE_API_KEY` โ€” get at https://cloud.browser-use.com/new-api-key @@ -68,7 +69,8 @@ llm = ChatBrowserUse(model='bu-2-0') # Premium model **Models & Pricing (per 1M tokens):** | Model | Input | Cached | Output | |-------|-------|--------|--------| -| bu-1-0 / bu-latest (default) | $0.20 | $0.02 | $2.00 | +| bu-2-0-mini-preview (default) | $0.15 | $0.15 | $1.50 | +| bu-1-0 | $0.20 | $0.02 | $2.00 | | bu-2-0 (premium) | $0.60 | $0.06 | $3.50 | | browser-use/bu-30b-a3b-preview (OSS) | โ€” | โ€” | โ€” | diff --git a/tests/ci/models/test_llm_browseruse.py b/tests/ci/models/test_llm_browseruse.py index 6e602fbeb..bf1aa8c31 100644 --- a/tests/ci/models/test_llm_browseruse.py +++ b/tests/ci/models/test_llm_browseruse.py @@ -25,13 +25,13 @@ async def test_browseruse_bu_latest(httpserver): # --- Model validation ------------------------------------------------------- -def test_default_model_is_bu_2_0(): +def test_default_model_is_bu_2_0_mini_preview(): chat = ChatBrowserUse(api_key=TEST_API_KEY) - assert chat.model == 'bu-2-0' + assert chat.model == 'bu-2-0-mini-preview' assert chat.provider == 'browser-use' -@pytest.mark.parametrize('alias', ['bu-1-0', 'bu-2-0', 'bu-qa-1']) +@pytest.mark.parametrize('alias', ['bu-1-0', 'bu-2-0', 'bu-2-0-mini-preview', 'bu-qa-1']) def test_bu_aliases_are_accepted(alias): chat = ChatBrowserUse(model=alias, api_key=TEST_API_KEY) assert chat.model == alias @@ -39,10 +39,31 @@ def test_bu_aliases_are_accepted(alias): assert chat.provider == 'browser-use' -def test_bu_latest_normalizes_to_bu_2_0(): +def test_bu_latest_still_normalizes_to_bu_2_0(): + """'latest' tracks the stable premium line, so it does NOT follow the preview default.""" chat = ChatBrowserUse(model='bu-latest', api_key=TEST_API_KEY) assert chat.model == 'bu-2-0' assert chat.name == 'bu-2-0' + assert ChatBrowserUse(api_key=TEST_API_KEY).model != chat.model + + +def test_bu_2_0_mini_preview_is_priced(): + """The default model must have a pricing entry, or cost tracking silently reports $0.""" + from browser_use.tokens.custom_pricing import CUSTOM_MODEL_PRICING + + pricing = CUSTOM_MODEL_PRICING['bu-2-0-mini-preview'] + assert pricing['input_cost_per_token'] > 0 + assert pricing['output_cost_per_token'] > 0 + # bu-latest resolves to bu-2-0, not to the preview, so it keeps the premium pricing. + assert CUSTOM_MODEL_PRICING['bu-latest'] == CUSTOM_MODEL_PRICING['bu-2-0'] + + +def test_llm_models_shortcut_resolves_mini_preview(monkeypatch): + """`llm.bu_2_0_mini_preview` must map the underscored name back to the dashed model id.""" + monkeypatch.setenv('BROWSER_USE_API_KEY', TEST_API_KEY) + from browser_use.llm.models import get_llm_by_name + + assert get_llm_by_name('bu_2_0_mini_preview').name == 'bu-2-0-mini-preview' @pytest.mark.parametrize( From b1d1794a9303df4cf92a8a2ccce6c4ceb77bf27e Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Thu, 13 Aug 2026 11:16:30 -0700 Subject: [PATCH 10/69] fix(tokens): bill bu-1-0 at bu-2-0 rates bu-1-0 is redirected to bu-2-0 at the gateway, so requests naming it are served and billed as bu-2-0. The pricing table still carried the retired bu-1-0 rates, which under-reported cost by 3x on input and 3x on cached reads for anyone still passing that id. Aliases it to the bu-2-0 entry rather than duplicating the numbers, same as bu-latest, so the three cannot drift apart. --- browser_use/llm/browser_use/chat.py | 2 +- browser_use/tokens/custom_pricing.py | 12 +++--------- skills/open-source/references/models.md | 2 +- tests/ci/models/test_llm_browseruse.py | 3 +++ 4 files changed, 8 insertions(+), 11 deletions(-) diff --git a/browser_use/llm/browser_use/chat.py b/browser_use/llm/browser_use/chat.py index 25e108f8b..b014b0f32 100644 --- a/browser_use/llm/browser_use/chat.py +++ b/browser_use/llm/browser_use/chat.py @@ -60,7 +60,7 @@ class ChatBrowserUse(BaseChatModel): model: Model name to use. Options: - 'bu-2-0-mini-preview': Default model (fast + cheap, preview) - 'bu-2-0' or 'bu-latest': Premium model - - 'bu-1-0': Previous generation model + - 'bu-1-0': Previous generation model, redirected to bu-2-0 at the gateway - 'bu-qa-1': Website QA model (tests a site and scores functionality/aesthetics) - 'browser-use/bu-30b-a3b-preview': Browser Use Open Source Model - Provider-prefixed ids resolved by the gateway, e.g. 'anthropic/claude-sonnet-4-6', diff --git a/browser_use/tokens/custom_pricing.py b/browser_use/tokens/custom_pricing.py index b625ea758..c053a8748 100644 --- a/browser_use/tokens/custom_pricing.py +++ b/browser_use/tokens/custom_pricing.py @@ -9,15 +9,6 @@ from typing import Any # Custom model pricing data # Format matches LiteLLM's model_prices_and_context_window.json structure CUSTOM_MODEL_PRICING: dict[str, dict[str, Any]] = { - 'bu-1-0': { - 'input_cost_per_token': 0.2 / 1_000_000, # $0.20 per 1M tokens - 'output_cost_per_token': 2.00 / 1_000_000, # $2.00 per 1M tokens - 'cache_read_input_token_cost': 0.02 / 1_000_000, # $0.02 per 1M tokens - 'cache_creation_input_token_cost': None, # Not specified - 'max_tokens': None, # Not specified - 'max_input_tokens': None, # Not specified - 'max_output_tokens': None, # Not specified - }, 'bu-2-0': { 'input_cost_per_token': 0.60 / 1_000_000, # $0.60 per 1M tokens 'output_cost_per_token': 3.50 / 1_000_000, # $3.50 per 1M tokens @@ -100,4 +91,7 @@ CUSTOM_MODEL_PRICING: dict[str, dict[str, Any]] = { } CUSTOM_MODEL_PRICING['bu-latest'] = CUSTOM_MODEL_PRICING['bu-2-0'] +# bu-1-0 is redirected to bu-2-0 at the gateway, so it bills at bu-2-0 rates. +CUSTOM_MODEL_PRICING['bu-1-0'] = CUSTOM_MODEL_PRICING['bu-2-0'] + CUSTOM_MODEL_PRICING['smart'] = CUSTOM_MODEL_PRICING['bu-2-0'] diff --git a/skills/open-source/references/models.md b/skills/open-source/references/models.md index 8193b731d..590e7df2f 100644 --- a/skills/open-source/references/models.md +++ b/skills/open-source/references/models.md @@ -70,8 +70,8 @@ llm = ChatBrowserUse(model='bu-2-0') # Premium model ('bu-latest' | Model | Input | Cached | Output | |-------|-------|--------|--------| | bu-2-0-mini-preview (default) | $0.15 | $0.15 | $1.50 | -| bu-1-0 | $0.20 | $0.02 | $2.00 | | bu-2-0 (premium) | $0.60 | $0.06 | $3.50 | +| bu-1-0 (redirects to bu-2-0) | $0.60 | $0.06 | $3.50 | | browser-use/bu-30b-a3b-preview (OSS) | โ€” | โ€” | โ€” | ## OpenAI diff --git a/tests/ci/models/test_llm_browseruse.py b/tests/ci/models/test_llm_browseruse.py index bf1aa8c31..e073c04a4 100644 --- a/tests/ci/models/test_llm_browseruse.py +++ b/tests/ci/models/test_llm_browseruse.py @@ -56,6 +56,9 @@ def test_bu_2_0_mini_preview_is_priced(): assert pricing['output_cost_per_token'] > 0 # bu-latest resolves to bu-2-0, not to the preview, so it keeps the premium pricing. assert CUSTOM_MODEL_PRICING['bu-latest'] == CUSTOM_MODEL_PRICING['bu-2-0'] + # bu-1-0 is redirected to bu-2-0 at the gateway, so it must be billed at bu-2-0 rates + # rather than the retired bu-1-0 rates. + assert CUSTOM_MODEL_PRICING['bu-1-0'] == CUSTOM_MODEL_PRICING['bu-2-0'] def test_llm_models_shortcut_resolves_mini_preview(monkeypatch): From d762a8359eb14420b397fe2f9db96c11a1ca853c Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Thu, 13 Aug 2026 11:33:53 -0700 Subject: [PATCH 11/69] test: cover the llm.bu_2_0_mini_preview attribute, not just the factory The shortcut test called get_llm_by_name directly, which bypasses the module __getattr__ and __all__ that the shortcut actually goes through - so dropping the name from __all__ would not have failed anything. --- tests/ci/models/test_llm_browseruse.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/tests/ci/models/test_llm_browseruse.py b/tests/ci/models/test_llm_browseruse.py index e073c04a4..166a176fe 100644 --- a/tests/ci/models/test_llm_browseruse.py +++ b/tests/ci/models/test_llm_browseruse.py @@ -64,9 +64,14 @@ def test_bu_2_0_mini_preview_is_priced(): def test_llm_models_shortcut_resolves_mini_preview(monkeypatch): """`llm.bu_2_0_mini_preview` must map the underscored name back to the dashed model id.""" monkeypatch.setenv('BROWSER_USE_API_KEY', TEST_API_KEY) - from browser_use.llm.models import get_llm_by_name + from browser_use import llm + from browser_use.llm import models - assert get_llm_by_name('bu_2_0_mini_preview').name == 'bu-2-0-mini-preview' + # Go through the advertised attribute rather than the factory beneath it, so dropping the + # name from __all__ or breaking module __getattr__ fails here instead of passing silently. + assert llm.bu_2_0_mini_preview.name == 'bu-2-0-mini-preview' + assert models.bu_2_0_mini_preview.name == 'bu-2-0-mini-preview' + assert 'bu_2_0_mini_preview' in models.__all__ @pytest.mark.parametrize( From 5ec35bea5615e7603ed0f6b03b95f795fc22266a Mon Sep 17 00:00:00 2001 From: cosin2077 Date: Mon, 3 Aug 2026 11:31:53 +0800 Subject: [PATCH 12/69] fix(llm): preserve zero retries for Anthropic Bedrock --- browser_use/llm/aws/chat_anthropic.py | 3 +-- .../test_chat_anthropic_bedrock_client_config.py | 14 ++++++++++++++ 2 files changed, 15 insertions(+), 2 deletions(-) create mode 100644 tests/ci/models/test_chat_anthropic_bedrock_client_config.py diff --git a/browser_use/llm/aws/chat_anthropic.py b/browser_use/llm/aws/chat_anthropic.py index 48f21dc9a..4b673b8a1 100644 --- a/browser_use/llm/aws/chat_anthropic.py +++ b/browser_use/llm/aws/chat_anthropic.py @@ -89,8 +89,7 @@ class ChatAnthropicBedrock(ChatAWSBedrock): client_params['aws_session_token'] = self.aws_session_token # Add optional parameters - if self.max_retries: - client_params['max_retries'] = self.max_retries + client_params['max_retries'] = self.max_retries if self.default_headers: client_params['default_headers'] = self.default_headers if self.default_query: diff --git a/tests/ci/models/test_chat_anthropic_bedrock_client_config.py b/tests/ci/models/test_chat_anthropic_bedrock_client_config.py new file mode 100644 index 000000000..9657c68b8 --- /dev/null +++ b/tests/ci/models/test_chat_anthropic_bedrock_client_config.py @@ -0,0 +1,14 @@ +from browser_use.llm.aws.chat_anthropic import ChatAnthropicBedrock + + +def test_get_client_preserves_zero_max_retries() -> None: + llm = ChatAnthropicBedrock( + aws_access_key='test-access-key', + aws_secret_key='test-secret-key', + aws_region='us-east-1', + max_retries=0, + ) + + client = llm.get_client() + + assert client.max_retries == 0 From cb6f2968b5dcf1a6f88a84293e21017d32780a6e Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sat, 15 Aug 2026 08:48:37 -0700 Subject: [PATCH 13/69] fix(deps): bump rich to 14.3.3 --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 158eaea76..c3c3073a8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -16,7 +16,7 @@ dependencies = [ "bubus==1.5.6", "click==8.3.1", "InquirerPy==0.3.4", - "rich==14.3.1", + "rich==14.3.3", "google-api-core==2.29.0", "httpx==0.28.1", "posthog==7.7.0", From 1f6b19488cf5ae4eda71df591b828a52150481ad Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sat, 15 Aug 2026 09:38:38 -0700 Subject: [PATCH 14/69] chore: sync browser-use skill copies --- browser_use/skills/browser-use/SKILL.md | 13 ++++++++++--- skills/browser-use/SKILL.md | 13 ++++++++++--- 2 files changed, 20 insertions(+), 6 deletions(-) diff --git a/browser_use/skills/browser-use/SKILL.md b/browser_use/skills/browser-use/SKILL.md index 039209a21..ce9ebb645 100644 --- a/browser_use/skills/browser-use/SKILL.md +++ b/browser_use/skills/browser-use/SKILL.md @@ -36,7 +36,7 @@ If the daemon cannot connect, run diagnostics: browser-use --doctor ``` -If Chrome is not running at all, the harness launches it automatically and retries โ€” no user action needed beyond clicking Allow if a permission popup appears. +If Chrome is not running at all, the harness launches it automatically and retries. If Chrome is running but remote debugging is not enabled, the harness opens: @@ -44,7 +44,14 @@ If Chrome is running but remote debugging is not enabled, the harness opens: chrome://inspect/#remote-debugging ``` -Ask the user to tick "Allow remote debugging for this browser instance" and click Allow if Chrome shows a permission popup. Then retry the same `browser-use` command. +On macOS, when Chrome asks for remote-debugging permission, run: + +```text +browser-use mac-approve +``` + +Continue browser work when it returns `ready`; otherwise follow its printed +instruction. ## Remote Browsers @@ -160,7 +167,7 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro ## Gotchas - `chrome://inspect/#remote-debugging` must be enabled for local Chrome control. -- Chrome may show an "Allow remote debugging?" popup; wait for the user to click Allow. Do not retry in a loop โ€” Chrome pops a fresh dialog for every new connection, and the daemon's single held connection is what makes this a one-time click. +- On macOS, if Chrome shows an "Allow remote debugging?" popup, run `browser-use mac-approve`. Do not poll in a loop โ€” the daemon holds one connection. - Omnibox popups are not real work tabs. - CDP target order is not Chrome's visible tab-strip order. - `BU_CDP_URL` is an HTTP DevTools endpoint; the daemon resolves it to WebSocket. diff --git a/skills/browser-use/SKILL.md b/skills/browser-use/SKILL.md index 039209a21..ce9ebb645 100644 --- a/skills/browser-use/SKILL.md +++ b/skills/browser-use/SKILL.md @@ -36,7 +36,7 @@ If the daemon cannot connect, run diagnostics: browser-use --doctor ``` -If Chrome is not running at all, the harness launches it automatically and retries โ€” no user action needed beyond clicking Allow if a permission popup appears. +If Chrome is not running at all, the harness launches it automatically and retries. If Chrome is running but remote debugging is not enabled, the harness opens: @@ -44,7 +44,14 @@ If Chrome is running but remote debugging is not enabled, the harness opens: chrome://inspect/#remote-debugging ``` -Ask the user to tick "Allow remote debugging for this browser instance" and click Allow if Chrome shows a permission popup. Then retry the same `browser-use` command. +On macOS, when Chrome asks for remote-debugging permission, run: + +```text +browser-use mac-approve +``` + +Continue browser work when it returns `ready`; otherwise follow its printed +instruction. ## Remote Browsers @@ -160,7 +167,7 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro ## Gotchas - `chrome://inspect/#remote-debugging` must be enabled for local Chrome control. -- Chrome may show an "Allow remote debugging?" popup; wait for the user to click Allow. Do not retry in a loop โ€” Chrome pops a fresh dialog for every new connection, and the daemon's single held connection is what makes this a one-time click. +- On macOS, if Chrome shows an "Allow remote debugging?" popup, run `browser-use mac-approve`. Do not poll in a loop โ€” the daemon holds one connection. - Omnibox popups are not real work tabs. - CDP target order is not Chrome's visible tab-strip order. - `BU_CDP_URL` is an HTTP DevTools endpoint; the daemon resolves it to WebSocket. From cebb51c15820d253232b72cf784e63ea7e96aa7b Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sat, 15 Aug 2026 07:48:53 -0700 Subject: [PATCH 15/69] feat: add OpenClaw skill support --- browser_use/skills/browser-use/SKILL.md | 19 +++++++++++++++++++ browser_use/skills/install.py | 1 + skills/browser-use/SKILL.md | 19 +++++++++++++++++++ .../ci/test_browser_use_skill_install_docs.py | 1 + 4 files changed, 40 insertions(+) diff --git a/browser_use/skills/browser-use/SKILL.md b/browser_use/skills/browser-use/SKILL.md index ce9ebb645..e779fad1b 100644 --- a/browser_use/skills/browser-use/SKILL.md +++ b/browser_use/skills/browser-use/SKILL.md @@ -1,6 +1,25 @@ --- name: browser-use description: "Direct browser control via CDP for web interaction: automation, scraping, testing, screenshots, and site/app work." +homepage: https://browser-use.com +metadata: + { + "openclaw": + { + "emoji": "๐ŸŒ", + "requires": { "bins": ["browser-use"] }, + "install": + [ + { + "id": "uv", + "kind": "uv", + "package": "browser-use==0.13.7", + "bins": ["browser-use"], + "label": "Install Browser Use CLI (uv)", + }, + ], + }, + } --- # Browser Use diff --git a/browser_use/skills/install.py b/browser_use/skills/install.py index feed6f8f1..5997e12ac 100644 --- a/browser_use/skills/install.py +++ b/browser_use/skills/install.py @@ -29,6 +29,7 @@ TARGET_DIR_BUILDERS = { 'copilot': lambda: _home_skill_dir('copilot'), 'cursor': lambda: _home_skill_dir('cursor'), 'gemini': lambda: _home_skill_dir('gemini'), + 'openclaw': lambda: _home_skill_dir('openclaw'), 'opencode': lambda: _xdg_config_home() / 'opencode' / 'skills' / SKILL_NAME, } diff --git a/skills/browser-use/SKILL.md b/skills/browser-use/SKILL.md index ce9ebb645..e779fad1b 100644 --- a/skills/browser-use/SKILL.md +++ b/skills/browser-use/SKILL.md @@ -1,6 +1,25 @@ --- name: browser-use description: "Direct browser control via CDP for web interaction: automation, scraping, testing, screenshots, and site/app work." +homepage: https://browser-use.com +metadata: + { + "openclaw": + { + "emoji": "๐ŸŒ", + "requires": { "bins": ["browser-use"] }, + "install": + [ + { + "id": "uv", + "kind": "uv", + "package": "browser-use==0.13.7", + "bins": ["browser-use"], + "label": "Install Browser Use CLI (uv)", + }, + ], + }, + } --- # Browser Use diff --git a/tests/ci/test_browser_use_skill_install_docs.py b/tests/ci/test_browser_use_skill_install_docs.py index 66dbf6512..d7f1246aa 100644 --- a/tests/ci/test_browser_use_skill_install_docs.py +++ b/tests/ci/test_browser_use_skill_install_docs.py @@ -12,6 +12,7 @@ EXPECTED_SKILL_INSTALL_PATHS = ( Path('.copilot') / 'skills' / 'browser-use' / 'SKILL.md', Path('.cursor') / 'skills' / 'browser-use' / 'SKILL.md', Path('.gemini') / 'skills' / 'browser-use' / 'SKILL.md', + Path('.openclaw') / 'skills' / 'browser-use' / 'SKILL.md', Path('.config') / 'opencode' / 'skills' / 'browser-use' / 'SKILL.md', ) From a5d57d48dd3b41ed422871e5053708b3670a1a5c Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sat, 15 Aug 2026 08:05:41 -0700 Subject: [PATCH 16/69] fix: generate OpenClaw skill metadata --- browser_use/skills/browser-use/SKILL.md | 2 +- browser_use/skills/browser_use.py | 25 +++++++++++++++ browser_use/skills/install.py | 14 +++++++- skills/browser-use/SKILL.md | 2 +- .../ci/test_browser_use_skill_install_docs.py | 32 +++++++++++++++++++ 5 files changed, 72 insertions(+), 3 deletions(-) diff --git a/browser_use/skills/browser-use/SKILL.md b/browser_use/skills/browser-use/SKILL.md index e779fad1b..a0aa3e585 100644 --- a/browser_use/skills/browser-use/SKILL.md +++ b/browser_use/skills/browser-use/SKILL.md @@ -13,7 +13,7 @@ metadata: { "id": "uv", "kind": "uv", - "package": "browser-use==0.13.7", + "package": "browser-use", "bins": ["browser-use"], "label": "Install Browser Use CLI (uv)", }, diff --git a/browser_use/skills/browser_use.py b/browser_use/skills/browser_use.py index 9b364adca..03001fced 100644 --- a/browser_use/skills/browser_use.py +++ b/browser_use/skills/browser_use.py @@ -6,6 +6,27 @@ import re from importlib import resources from pathlib import Path +OPENCLAW_METADATA = ( + 'metadata:\n' + ' {\n' + ' "openclaw":\n' + ' {\n' + ' "emoji": "๐ŸŒ",\n' + ' "requires": { "bins": ["browser-use"] },\n' + ' "install":\n' + ' [\n' + ' {\n' + ' "id": "uv",\n' + ' "kind": "uv",\n' + ' "package": "browser-use",\n' + ' "bins": ["browser-use"],\n' + ' "label": "Install Browser Use CLI (uv)",\n' + ' },\n' + ' ],\n' + ' },\n' + ' }' +) + def as_browser_use_skill(text: str) -> str: """Expose the Browser Harness skill under the Browser Use skill identity.""" @@ -39,6 +60,10 @@ def as_browser_use_skill(text: str) -> str: 1, 'description: "Direct browser control via CDP for web interaction: automation, scraping, testing, screenshots, and site/app work."', ) + if not any(line.startswith('homepage:') for line in lines): + lines.append('homepage: https://browser-use.com') + if not any(line.startswith('metadata:') for line in lines): + lines.extend(OPENCLAW_METADATA.splitlines()) body = body.replace('# browser-harness', '# Browser Use', 1).replace('# Browser Harness', '# Browser Use', 1) # Rebrand every mention except repo URLs (github.com/browser-use/browser-harness/...) diff --git a/browser_use/skills/install.py b/browser_use/skills/install.py index 5997e12ac..6b44d806f 100644 --- a/browser_use/skills/install.py +++ b/browser_use/skills/install.py @@ -148,6 +148,18 @@ def _validate_output_paths(output_paths: list[Path]) -> None: raise RuntimeError(f'{ancestor} is not a directory.') +def _print_skill_text(text: str) -> None: + # Windows consoles commonly default to cp1252, which cannot encode the emoji + # in OpenClaw skill metadata. Prefer UTF-8 when stdout supports reconfiguration. + reconfigure = getattr(sys.stdout, 'reconfigure', None) + if callable(reconfigure): + try: + reconfigure(encoding='utf-8') + except OSError: + pass + print(text, end='') + + def handle(argv: list[str]) -> int: parser = _build_parser() args = parser.parse_args(argv) @@ -160,7 +172,7 @@ def handle(argv: list[str]) -> int: except RuntimeError as exc: print(f'Error: {exc}', file=sys.stderr) return 1 - print(text, end='') + _print_skill_text(text) return 0 if command == 'install': diff --git a/skills/browser-use/SKILL.md b/skills/browser-use/SKILL.md index e779fad1b..a0aa3e585 100644 --- a/skills/browser-use/SKILL.md +++ b/skills/browser-use/SKILL.md @@ -13,7 +13,7 @@ metadata: { "id": "uv", "kind": "uv", - "package": "browser-use==0.13.7", + "package": "browser-use", "bins": ["browser-use"], "label": "Install Browser Use CLI (uv)", }, diff --git a/tests/ci/test_browser_use_skill_install_docs.py b/tests/ci/test_browser_use_skill_install_docs.py index d7f1246aa..562fc9c09 100644 --- a/tests/ci/test_browser_use_skill_install_docs.py +++ b/tests/ci/test_browser_use_skill_install_docs.py @@ -1,6 +1,7 @@ import os import subprocess import sys +from io import BytesIO, TextIOWrapper from pathlib import Path ROOT = Path(__file__).resolve().parents[2] @@ -87,6 +88,25 @@ def test_browser_use_cli_installs_browser_harness_package_skill(tmp_path): '---\n' 'name: browser-use\n' 'description: "Direct browser control via CDP for web interaction: automation, scraping, testing, screenshots, and site/app work."\n' + 'homepage: https://browser-use.com\n' + 'metadata:\n' + ' {\n' + ' "openclaw":\n' + ' {\n' + ' "emoji": "๐ŸŒ",\n' + ' "requires": { "bins": ["browser-use"] },\n' + ' "install":\n' + ' [\n' + ' {\n' + ' "id": "uv",\n' + ' "kind": "uv",\n' + ' "package": "browser-use",\n' + ' "bins": ["browser-use"],\n' + ' "label": "Install Browser Use CLI (uv)",\n' + ' },\n' + ' ],\n' + ' },\n' + ' }\n' '---\n\n' '# Browser Use\n' ) @@ -118,3 +138,15 @@ def test_browser_use_cli_validates_destination_before_installing_harness(tmp_pat assert result.returncode == 1 assert 'is not a directory' in result.stderr assert not uv_args.exists() + + +def test_skill_show_reconfigures_windows_stdout_to_utf8(monkeypatch): + from browser_use.skills import install + + buffer = BytesIO() + windows_stdout = TextIOWrapper(buffer, encoding='cp1252') + monkeypatch.setattr(sys, 'stdout', windows_stdout) + install._print_skill_text('Browser Use ๐ŸŒ\n') + windows_stdout.flush() + + assert buffer.getvalue().decode('utf-8') == 'Browser Use ๐ŸŒ\n' From 872ca53d04bbf0f530d515b7f980fd40ee90fad5 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sat, 15 Aug 2026 08:23:21 -0700 Subject: [PATCH 17/69] fix: harden generated skill output --- browser_use/skills/browser_use.py | 52 +++++++++++-------- browser_use/skills/install.py | 9 +++- .../ci/test_browser_use_skill_install_docs.py | 38 ++++++++++++++ 3 files changed, 75 insertions(+), 24 deletions(-) diff --git a/browser_use/skills/browser_use.py b/browser_use/skills/browser_use.py index 03001fced..1a1bb76b7 100644 --- a/browser_use/skills/browser_use.py +++ b/browser_use/skills/browser_use.py @@ -6,26 +6,34 @@ import re from importlib import resources from pathlib import Path -OPENCLAW_METADATA = ( - 'metadata:\n' - ' {\n' - ' "openclaw":\n' - ' {\n' - ' "emoji": "๐ŸŒ",\n' - ' "requires": { "bins": ["browser-use"] },\n' - ' "install":\n' - ' [\n' - ' {\n' - ' "id": "uv",\n' - ' "kind": "uv",\n' - ' "package": "browser-use",\n' - ' "bins": ["browser-use"],\n' - ' "label": "Install Browser Use CLI (uv)",\n' - ' },\n' - ' ],\n' - ' },\n' - ' }' -) + +def _canonical_skill_path() -> Path: + return Path(__file__).resolve().parent / 'browser-use' / 'SKILL.md' + + +def _canonical_metadata_lines() -> list[str]: + """Read the metadata block from the canonical Browser Use skill.""" + skill_path = _canonical_skill_path() + if not skill_path.exists(): + return [] + + try: + _, frontmatter, _ = skill_path.read_text(encoding='utf-8').split('---\n', 2) + except ValueError: + return [] + + lines = frontmatter.splitlines() + try: + start = lines.index('metadata:') + except ValueError: + return [] + + end = len(lines) + for index in range(start + 1, len(lines)): + if lines[index] and not lines[index][0].isspace(): + end = index + break + return lines[start:end] def as_browser_use_skill(text: str) -> str: @@ -63,7 +71,7 @@ def as_browser_use_skill(text: str) -> str: if not any(line.startswith('homepage:') for line in lines): lines.append('homepage: https://browser-use.com') if not any(line.startswith('metadata:') for line in lines): - lines.extend(OPENCLAW_METADATA.splitlines()) + lines.extend(_canonical_metadata_lines()) body = body.replace('# browser-harness', '# Browser Use', 1).replace('# Browser Harness', '# Browser Use', 1) # Rebrand every mention except repo URLs (github.com/browser-use/browser-harness/...) @@ -75,7 +83,7 @@ def as_browser_use_skill(text: str) -> str: def skill_text() -> str: """Return the canonical Browser Use skill.""" - skill_path = Path(__file__).resolve().parent / 'browser-use' / 'SKILL.md' + skill_path = _canonical_skill_path() if skill_path.exists(): return skill_path.read_text(encoding='utf-8') diff --git a/browser_use/skills/install.py b/browser_use/skills/install.py index 6b44d806f..9d67ff3ad 100644 --- a/browser_use/skills/install.py +++ b/browser_use/skills/install.py @@ -152,11 +152,16 @@ def _print_skill_text(text: str) -> None: # Windows consoles commonly default to cp1252, which cannot encode the emoji # in OpenClaw skill metadata. Prefer UTF-8 when stdout supports reconfiguration. reconfigure = getattr(sys.stdout, 'reconfigure', None) + reconfigured = False if callable(reconfigure): try: - reconfigure(encoding='utf-8') - except OSError: + reconfigure(encoding='utf-8', errors='replace') + reconfigured = True + except (OSError, TypeError, ValueError): pass + if not reconfigured: + encoding = getattr(sys.stdout, 'encoding', None) or 'utf-8' + text = text.encode(encoding, errors='replace').decode(encoding) print(text, end='') diff --git a/tests/ci/test_browser_use_skill_install_docs.py b/tests/ci/test_browser_use_skill_install_docs.py index 562fc9c09..f83e6c3e6 100644 --- a/tests/ci/test_browser_use_skill_install_docs.py +++ b/tests/ci/test_browser_use_skill_install_docs.py @@ -150,3 +150,41 @@ def test_skill_show_reconfigures_windows_stdout_to_utf8(monkeypatch): windows_stdout.flush() assert buffer.getvalue().decode('utf-8') == 'Browser Use ๐ŸŒ\n' + + +def test_skill_show_replaces_unsupported_characters_without_reconfigure(monkeypatch): + from browser_use.skills import install + + class LegacyStdout: + encoding = 'cp1252' + + def __init__(self): + self.buffer = BytesIO() + + def write(self, text: str) -> int: + data = text.encode(self.encoding) + self.buffer.write(data) + return len(text) + + class FailingReconfigureStdout(LegacyStdout): + def reconfigure(self, **kwargs): + raise OSError('legacy stream cannot be reconfigured') + + for legacy_stdout in (LegacyStdout(), FailingReconfigureStdout()): + monkeypatch.setattr(sys, 'stdout', legacy_stdout) + install._print_skill_text('Browser Use ๐ŸŒ\n') + + assert legacy_stdout.buffer.getvalue().decode('cp1252') == 'Browser Use ?\n' + + +def test_browser_use_alias_sources_metadata_from_canonical_skill(): + from browser_use.skills import browser_use + + canonical = browser_use.skill_text() + _, canonical_frontmatter, _ = canonical.split('---\n', 2) + metadata = canonical_frontmatter[canonical_frontmatter.index('metadata:') :] + + converted = browser_use.as_browser_use_skill('---\nname: browser-harness\n---\n\n# Browser Harness\n') + _, converted_frontmatter, _ = converted.split('---\n', 2) + + assert metadata in converted_frontmatter From dee2d16bef1ef0ff5e7e7717500264096afc3913 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sat, 15 Aug 2026 10:04:17 -0700 Subject: [PATCH 18/69] fix: simplify OpenClaw skill metadata --- browser_use/skills/browser-use/SKILL.md | 1 - browser_use/skills/browser_use.py | 53 ++++++++----------- browser_use/skills/install.py | 19 +------ skills/browser-use/SKILL.md | 1 - .../ci/test_browser_use_skill_install_docs.py | 52 ------------------ 5 files changed, 24 insertions(+), 102 deletions(-) diff --git a/browser_use/skills/browser-use/SKILL.md b/browser_use/skills/browser-use/SKILL.md index a0aa3e585..0c8f6cc22 100644 --- a/browser_use/skills/browser-use/SKILL.md +++ b/browser_use/skills/browser-use/SKILL.md @@ -6,7 +6,6 @@ metadata: { "openclaw": { - "emoji": "๐ŸŒ", "requires": { "bins": ["browser-use"] }, "install": [ diff --git a/browser_use/skills/browser_use.py b/browser_use/skills/browser_use.py index 1a1bb76b7..d78f0c194 100644 --- a/browser_use/skills/browser_use.py +++ b/browser_use/skills/browser_use.py @@ -6,34 +6,27 @@ import re from importlib import resources from pathlib import Path - -def _canonical_skill_path() -> Path: - return Path(__file__).resolve().parent / 'browser-use' / 'SKILL.md' - - -def _canonical_metadata_lines() -> list[str]: - """Read the metadata block from the canonical Browser Use skill.""" - skill_path = _canonical_skill_path() - if not skill_path.exists(): - return [] - - try: - _, frontmatter, _ = skill_path.read_text(encoding='utf-8').split('---\n', 2) - except ValueError: - return [] - - lines = frontmatter.splitlines() - try: - start = lines.index('metadata:') - except ValueError: - return [] - - end = len(lines) - for index in range(start + 1, len(lines)): - if lines[index] and not lines[index][0].isspace(): - end = index - break - return lines[start:end] +# Browser Use-only frontmatter added while generating both checked-in SKILL.md copies. +# Keep this as the source of truth; scripts/sync_browser_harness_skill.py verifies the outputs. +OPENCLAW_METADATA_LINES = ( + 'metadata:', + ' {', + ' "openclaw":', + ' {', + ' "requires": { "bins": ["browser-use"] },', + ' "install":', + ' [', + ' {', + ' "id": "uv",', + ' "kind": "uv",', + ' "package": "browser-use",', + ' "bins": ["browser-use"],', + ' "label": "Install Browser Use CLI (uv)",', + ' },', + ' ],', + ' },', + ' }', +) def as_browser_use_skill(text: str) -> str: @@ -71,7 +64,7 @@ def as_browser_use_skill(text: str) -> str: if not any(line.startswith('homepage:') for line in lines): lines.append('homepage: https://browser-use.com') if not any(line.startswith('metadata:') for line in lines): - lines.extend(_canonical_metadata_lines()) + lines.extend(OPENCLAW_METADATA_LINES) body = body.replace('# browser-harness', '# Browser Use', 1).replace('# Browser Harness', '# Browser Use', 1) # Rebrand every mention except repo URLs (github.com/browser-use/browser-harness/...) @@ -83,7 +76,7 @@ def as_browser_use_skill(text: str) -> str: def skill_text() -> str: """Return the canonical Browser Use skill.""" - skill_path = _canonical_skill_path() + skill_path = Path(__file__).resolve().parent / 'browser-use' / 'SKILL.md' if skill_path.exists(): return skill_path.read_text(encoding='utf-8') diff --git a/browser_use/skills/install.py b/browser_use/skills/install.py index 9d67ff3ad..5997e12ac 100644 --- a/browser_use/skills/install.py +++ b/browser_use/skills/install.py @@ -148,23 +148,6 @@ def _validate_output_paths(output_paths: list[Path]) -> None: raise RuntimeError(f'{ancestor} is not a directory.') -def _print_skill_text(text: str) -> None: - # Windows consoles commonly default to cp1252, which cannot encode the emoji - # in OpenClaw skill metadata. Prefer UTF-8 when stdout supports reconfiguration. - reconfigure = getattr(sys.stdout, 'reconfigure', None) - reconfigured = False - if callable(reconfigure): - try: - reconfigure(encoding='utf-8', errors='replace') - reconfigured = True - except (OSError, TypeError, ValueError): - pass - if not reconfigured: - encoding = getattr(sys.stdout, 'encoding', None) or 'utf-8' - text = text.encode(encoding, errors='replace').decode(encoding) - print(text, end='') - - def handle(argv: list[str]) -> int: parser = _build_parser() args = parser.parse_args(argv) @@ -177,7 +160,7 @@ def handle(argv: list[str]) -> int: except RuntimeError as exc: print(f'Error: {exc}', file=sys.stderr) return 1 - _print_skill_text(text) + print(text, end='') return 0 if command == 'install': diff --git a/skills/browser-use/SKILL.md b/skills/browser-use/SKILL.md index a0aa3e585..0c8f6cc22 100644 --- a/skills/browser-use/SKILL.md +++ b/skills/browser-use/SKILL.md @@ -6,7 +6,6 @@ metadata: { "openclaw": { - "emoji": "๐ŸŒ", "requires": { "bins": ["browser-use"] }, "install": [ diff --git a/tests/ci/test_browser_use_skill_install_docs.py b/tests/ci/test_browser_use_skill_install_docs.py index f83e6c3e6..41da21a81 100644 --- a/tests/ci/test_browser_use_skill_install_docs.py +++ b/tests/ci/test_browser_use_skill_install_docs.py @@ -1,7 +1,6 @@ import os import subprocess import sys -from io import BytesIO, TextIOWrapper from pathlib import Path ROOT = Path(__file__).resolve().parents[2] @@ -93,7 +92,6 @@ def test_browser_use_cli_installs_browser_harness_package_skill(tmp_path): ' {\n' ' "openclaw":\n' ' {\n' - ' "emoji": "๐ŸŒ",\n' ' "requires": { "bins": ["browser-use"] },\n' ' "install":\n' ' [\n' @@ -138,53 +136,3 @@ def test_browser_use_cli_validates_destination_before_installing_harness(tmp_pat assert result.returncode == 1 assert 'is not a directory' in result.stderr assert not uv_args.exists() - - -def test_skill_show_reconfigures_windows_stdout_to_utf8(monkeypatch): - from browser_use.skills import install - - buffer = BytesIO() - windows_stdout = TextIOWrapper(buffer, encoding='cp1252') - monkeypatch.setattr(sys, 'stdout', windows_stdout) - install._print_skill_text('Browser Use ๐ŸŒ\n') - windows_stdout.flush() - - assert buffer.getvalue().decode('utf-8') == 'Browser Use ๐ŸŒ\n' - - -def test_skill_show_replaces_unsupported_characters_without_reconfigure(monkeypatch): - from browser_use.skills import install - - class LegacyStdout: - encoding = 'cp1252' - - def __init__(self): - self.buffer = BytesIO() - - def write(self, text: str) -> int: - data = text.encode(self.encoding) - self.buffer.write(data) - return len(text) - - class FailingReconfigureStdout(LegacyStdout): - def reconfigure(self, **kwargs): - raise OSError('legacy stream cannot be reconfigured') - - for legacy_stdout in (LegacyStdout(), FailingReconfigureStdout()): - monkeypatch.setattr(sys, 'stdout', legacy_stdout) - install._print_skill_text('Browser Use ๐ŸŒ\n') - - assert legacy_stdout.buffer.getvalue().decode('cp1252') == 'Browser Use ?\n' - - -def test_browser_use_alias_sources_metadata_from_canonical_skill(): - from browser_use.skills import browser_use - - canonical = browser_use.skill_text() - _, canonical_frontmatter, _ = canonical.split('---\n', 2) - metadata = canonical_frontmatter[canonical_frontmatter.index('metadata:') :] - - converted = browser_use.as_browser_use_skill('---\nname: browser-harness\n---\n\n# Browser Harness\n') - _, converted_frontmatter, _ = converted.split('---\n', 2) - - assert metadata in converted_frontmatter From 610c9614a9078031546e2907fbb793e13924d0bd Mon Sep 17 00:00:00 2001 From: Aneesh Sharma Date: Sat, 15 Aug 2026 17:47:27 +0530 Subject: [PATCH 19/69] fix(browser): honor BROWSER_USE_HEADLESS in BrowserProfile (#5420) - Read BROWSER_USE_HEADLESS env var via default_factory in BrowserProfile - Preserve fallback to display detection when env var is unset - Add unit tests covering env var parsing and overrides Fixes #5420 --- browser_use/browser/profile.py | 13 +++- tests/ci/test_extension_config.py | 113 ++++++++++++++++++++++++++++++ 2 files changed, 125 insertions(+), 1 deletion(-) diff --git a/browser_use/browser/profile.py b/browser_use/browser/profile.py index 4b5e4a9bb..1e0e82fb3 100644 --- a/browser_use/browser/profile.py +++ b/browser_use/browser/profile.py @@ -25,6 +25,14 @@ def _get_enable_default_extensions_default() -> bool: return True +def _get_headless_default() -> bool | None: + """Get the default value for headless from BROWSER_USE_HEADLESS env var, or None to fall back to display detection.""" + env_val = os.getenv('BROWSER_USE_HEADLESS') + if env_val is not None: + return env_val.lower() not in ('0', 'false', 'no', 'off', '') + return None + + CHROME_DEBUG_PORT = 9242 # use a non-default port to avoid conflicts with other tools / devs using 9222 DOMAIN_OPTIMIZATION_THRESHOLD = 100 # Convert domain lists to sets for O(1) lookup when >= this size CHROME_PROFILE_TRANSIENT_FILE_PATTERNS = ( @@ -419,7 +427,10 @@ class BrowserLaunchArgs(BaseModel): validation_alias=AliasChoices('browser_binary_path', 'chrome_binary_path'), description='Path to the chromium-based browser executable to use.', ) - headless: bool | None = Field(default=None, description='Whether to run the browser in headless or windowed mode.') + headless: bool | None = Field( + default_factory=_get_headless_default, + description='Whether to run the browser in headless or windowed mode. Can be set via BROWSER_USE_HEADLESS environment variable.', + ) args: list[CliArgStr] = Field( default_factory=list, description='List of *extra* CLI args to pass to the browser when launching.' ) diff --git a/tests/ci/test_extension_config.py b/tests/ci/test_extension_config.py index 0493019e6..4eae8c724 100644 --- a/tests/ci/test_extension_config.py +++ b/tests/ci/test_extension_config.py @@ -121,3 +121,116 @@ class TestDisableExtensionsEnvVar: os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original else: os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None) + + +class TestHeadlessEnvVar: + """Test BROWSER_USE_HEADLESS environment variable.""" + + def test_default_value_is_none(self): + """Without env var set, headless default should be None.""" + original = os.environ.pop('BROWSER_USE_HEADLESS', None) + try: + from browser_use.browser.profile import _get_headless_default + + assert _get_headless_default() is None + finally: + if original is not None: + os.environ['BROWSER_USE_HEADLESS'] = original + + @pytest.mark.parametrize( + 'env_value,expected_headless', + [ + # Truthy values for HEADLESS = headless enabled (True) + ('true', True), + ('True', True), + ('TRUE', True), + ('1', True), + ('yes', True), + ('on', True), + # Falsy values for HEADLESS = headless disabled (False) + ('false', False), + ('False', False), + ('FALSE', False), + ('0', False), + ('no', False), + ('off', False), + ('', False), + ], + ) + def test_env_var_values(self, env_value: str, expected_headless: bool): + """Test various env var values are parsed correctly.""" + original = os.environ.get('BROWSER_USE_HEADLESS') + try: + os.environ['BROWSER_USE_HEADLESS'] = env_value + from browser_use.browser.profile import _get_headless_default + + result = _get_headless_default() + assert result is expected_headless, ( + f"Expected headless={expected_headless} for BROWSER_USE_HEADLESS='{env_value}', got {result}" + ) + finally: + if original is not None: + os.environ['BROWSER_USE_HEADLESS'] = original + else: + os.environ.pop('BROWSER_USE_HEADLESS', None) + + def test_browser_profile_uses_env_var(self): + """Test that BrowserProfile picks up the BROWSER_USE_HEADLESS env var.""" + original = os.environ.get('BROWSER_USE_HEADLESS') + try: + # Test with env var set to true + os.environ['BROWSER_USE_HEADLESS'] = 'true' + + from browser_use.browser.profile import BrowserProfile + + profile = BrowserProfile() + assert profile.headless is True, 'BrowserProfile should have headless=True when BROWSER_USE_HEADLESS=true' + + # Test with env var set to false + os.environ['BROWSER_USE_HEADLESS'] = 'false' + profile2 = BrowserProfile() + assert profile2.headless is False, 'BrowserProfile should have headless=False when BROWSER_USE_HEADLESS=false' + finally: + if original is not None: + os.environ['BROWSER_USE_HEADLESS'] = original + else: + os.environ.pop('BROWSER_USE_HEADLESS', None) + + def test_explicit_param_overrides_env_var(self): + """Test that explicit headless parameter overrides env var.""" + original = os.environ.get('BROWSER_USE_HEADLESS') + try: + os.environ['BROWSER_USE_HEADLESS'] = 'true' + + from browser_use.browser.profile import BrowserProfile + + # Explicitly set to False should override env var + profile = BrowserProfile(headless=False) + assert profile.headless is False, 'Explicit param should override env var' + + os.environ['BROWSER_USE_HEADLESS'] = 'false' + profile2 = BrowserProfile(headless=True) + assert profile2.headless is True, 'Explicit param should override env var' + finally: + if original is not None: + os.environ['BROWSER_USE_HEADLESS'] = original + else: + os.environ.pop('BROWSER_USE_HEADLESS', None) + + def test_browser_session_uses_env_var(self): + """Test that BrowserSession picks up the env var via BrowserProfile.""" + original = os.environ.get('BROWSER_USE_HEADLESS') + try: + os.environ['BROWSER_USE_HEADLESS'] = 'true' + + from browser_use.browser import BrowserSession + + session = BrowserSession() + assert session.browser_profile.headless is True, ( + 'BrowserSession should have headless=True when BROWSER_USE_HEADLESS=true' + ) + finally: + if original is not None: + os.environ['BROWSER_USE_HEADLESS'] = original + else: + os.environ.pop('BROWSER_USE_HEADLESS', None) From 1e9ee3e64b5701b653d6a08ac4c5480fdccdb1a7 Mon Sep 17 00:00:00 2001 From: Aneesh Sharma Date: Sat, 15 Aug 2026 17:58:44 +0530 Subject: [PATCH 20/69] test: deduplicate browser config env var tests using pytest parametrization --- tests/ci/test_extension_config.py | 316 ++++++++++-------------------- 1 file changed, 99 insertions(+), 217 deletions(-) diff --git a/tests/ci/test_extension_config.py b/tests/ci/test_extension_config.py index 4eae8c724..2f8e17631 100644 --- a/tests/ci/test_extension_config.py +++ b/tests/ci/test_extension_config.py @@ -1,236 +1,118 @@ -"""Tests for extension configuration environment variables.""" - -import os +"""Tests for browser configuration environment variables.""" import pytest +from browser_use.browser import BrowserSession +from browser_use.browser.profile import ( + BrowserProfile, + _get_enable_default_extensions_default, + _get_headless_default, +) -class TestDisableExtensionsEnvVar: - """Test BROWSER_USE_DISABLE_EXTENSIONS environment variable.""" +TRUTHY_STRINGS = ['true', 'True', 'TRUE', '1', 'yes', 'on'] +FALSY_STRINGS = ['false', 'False', 'FALSE', '0', 'no', 'off', ''] - def test_default_value_is_true(self): - """Without env var set, enable_default_extensions should default to True.""" - # Clear the env var if it exists - original = os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None) - try: - # Import fresh to get the default - from browser_use.browser.profile import _get_enable_default_extensions_default - assert _get_enable_default_extensions_default() is True - finally: - if original is not None: - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original +class TestConfigEnvVars: + """Tests for browser profile env var configuration.""" + + def test_default_values_without_env(self, monkeypatch: pytest.MonkeyPatch): + """Verify default values when environment variables are unset.""" + monkeypatch.delenv('BROWSER_USE_DISABLE_EXTENSIONS', raising=False) + monkeypatch.delenv('BROWSER_USE_HEADLESS', raising=False) + + assert _get_enable_default_extensions_default() is True + assert _get_headless_default() is None @pytest.mark.parametrize( - 'env_value,expected_enabled', + 'env_var,getter,truthy_expected,falsy_expected', [ - # Truthy values for DISABLE = extensions disabled (False) - ('true', False), - ('True', False), - ('TRUE', False), - ('1', False), - ('yes', False), - ('on', False), - # Falsy values for DISABLE = extensions enabled (True) - ('false', True), - ('False', True), - ('FALSE', True), - ('0', True), - ('no', True), - ('off', True), - ('', True), + ('BROWSER_USE_DISABLE_EXTENSIONS', _get_enable_default_extensions_default, False, True), + ('BROWSER_USE_HEADLESS', _get_headless_default, True, False), ], ) - def test_env_var_values(self, env_value: str, expected_enabled: bool): - """Test various env var values are parsed correctly.""" - original = os.environ.get('BROWSER_USE_DISABLE_EXTENSIONS') - try: - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = env_value - from browser_use.browser.profile import _get_enable_default_extensions_default - - result = _get_enable_default_extensions_default() - assert result is expected_enabled, ( - f"Expected enable_default_extensions={expected_enabled} for DISABLE_EXTENSIONS='{env_value}', got {result}" - ) - finally: - if original is not None: - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original - else: - os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None) - - def test_browser_profile_uses_env_var(self): - """Test that BrowserProfile picks up the env var.""" - original = os.environ.get('BROWSER_USE_DISABLE_EXTENSIONS') - try: - # Test with env var set to true (disable extensions) - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = 'true' - - from browser_use.browser.profile import BrowserProfile - - profile = BrowserProfile(headless=True) - assert profile.enable_default_extensions is False, ( - 'BrowserProfile should disable extensions when BROWSER_USE_DISABLE_EXTENSIONS=true' - ) - - # Test with env var set to false (enable extensions) - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = 'false' - profile2 = BrowserProfile(headless=True) - assert profile2.enable_default_extensions is True, ( - 'BrowserProfile should enable extensions when BROWSER_USE_DISABLE_EXTENSIONS=false' - ) - - finally: - if original is not None: - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original - else: - os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None) - - def test_explicit_param_overrides_env_var(self): - """Test that explicit enable_default_extensions parameter overrides env var.""" - original = os.environ.get('BROWSER_USE_DISABLE_EXTENSIONS') - try: - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = 'true' - - from browser_use.browser.profile import BrowserProfile - - # Explicitly set to True should override env var - profile = BrowserProfile(headless=True, enable_default_extensions=True) - assert profile.enable_default_extensions is True, 'Explicit param should override env var' - - finally: - if original is not None: - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original - else: - os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None) - - def test_browser_session_uses_env_var(self): - """Test that BrowserSession picks up the env var via BrowserProfile.""" - original = os.environ.get('BROWSER_USE_DISABLE_EXTENSIONS') - try: - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = '1' - - from browser_use.browser import BrowserSession - - session = BrowserSession(headless=True) - assert session.browser_profile.enable_default_extensions is False, ( - 'BrowserSession should disable extensions when BROWSER_USE_DISABLE_EXTENSIONS=1' - ) - - finally: - if original is not None: - os.environ['BROWSER_USE_DISABLE_EXTENSIONS'] = original - else: - os.environ.pop('BROWSER_USE_DISABLE_EXTENSIONS', None) - - -class TestHeadlessEnvVar: - """Test BROWSER_USE_HEADLESS environment variable.""" - - def test_default_value_is_none(self): - """Without env var set, headless default should be None.""" - original = os.environ.pop('BROWSER_USE_HEADLESS', None) - try: - from browser_use.browser.profile import _get_headless_default - - assert _get_headless_default() is None - finally: - if original is not None: - os.environ['BROWSER_USE_HEADLESS'] = original + def test_env_var_truthy_values( + self, + monkeypatch: pytest.MonkeyPatch, + env_var: str, + getter, + truthy_expected: bool, + falsy_expected: bool, + ): + """Test truthy env var values are parsed correctly.""" + for val in TRUTHY_STRINGS: + monkeypatch.setenv(env_var, val) + assert getter() is truthy_expected, f'Failed for {env_var}={val}' @pytest.mark.parametrize( - 'env_value,expected_headless', + 'env_var,getter,truthy_expected,falsy_expected', [ - # Truthy values for HEADLESS = headless enabled (True) - ('true', True), - ('True', True), - ('TRUE', True), - ('1', True), - ('yes', True), - ('on', True), - # Falsy values for HEADLESS = headless disabled (False) - ('false', False), - ('False', False), - ('FALSE', False), - ('0', False), - ('no', False), - ('off', False), - ('', False), + ('BROWSER_USE_DISABLE_EXTENSIONS', _get_enable_default_extensions_default, False, True), + ('BROWSER_USE_HEADLESS', _get_headless_default, True, False), ], ) - def test_env_var_values(self, env_value: str, expected_headless: bool): - """Test various env var values are parsed correctly.""" - original = os.environ.get('BROWSER_USE_HEADLESS') - try: - os.environ['BROWSER_USE_HEADLESS'] = env_value - from browser_use.browser.profile import _get_headless_default + def test_env_var_falsy_values( + self, + monkeypatch: pytest.MonkeyPatch, + env_var: str, + getter, + truthy_expected: bool, + falsy_expected: bool, + ): + """Test falsy env var values are parsed correctly.""" + for val in FALSY_STRINGS: + monkeypatch.setenv(env_var, val) + assert getter() is falsy_expected, f'Failed for {env_var}={val}' - result = _get_headless_default() - assert result is expected_headless, ( - f"Expected headless={expected_headless} for BROWSER_USE_HEADLESS='{env_value}', got {result}" - ) - finally: - if original is not None: - os.environ['BROWSER_USE_HEADLESS'] = original - else: - os.environ.pop('BROWSER_USE_HEADLESS', None) + @pytest.mark.parametrize( + 'env_var,attr_name,truthy_val,falsy_val', + [ + ('BROWSER_USE_DISABLE_EXTENSIONS', 'enable_default_extensions', False, True), + ('BROWSER_USE_HEADLESS', 'headless', True, False), + ], + ) + def test_browser_profile_and_session_env_var( + self, + monkeypatch: pytest.MonkeyPatch, + env_var: str, + attr_name: str, + truthy_val: bool, + falsy_val: bool, + ): + """Test that BrowserProfile and BrowserSession pick up env vars.""" + # Test truthy env value + monkeypatch.setenv(env_var, 'true') + profile = BrowserProfile() + assert getattr(profile, attr_name) is truthy_val + session = BrowserSession() + assert getattr(session.browser_profile, attr_name) is truthy_val - def test_browser_profile_uses_env_var(self): - """Test that BrowserProfile picks up the BROWSER_USE_HEADLESS env var.""" - original = os.environ.get('BROWSER_USE_HEADLESS') - try: - # Test with env var set to true - os.environ['BROWSER_USE_HEADLESS'] = 'true' + # Test falsy env value + monkeypatch.setenv(env_var, 'false') + profile_falsy = BrowserProfile() + assert getattr(profile_falsy, attr_name) is falsy_val + session_falsy = BrowserSession() + assert getattr(session_falsy.browser_profile, attr_name) is falsy_val - from browser_use.browser.profile import BrowserProfile - - profile = BrowserProfile() - assert profile.headless is True, 'BrowserProfile should have headless=True when BROWSER_USE_HEADLESS=true' - - # Test with env var set to false - os.environ['BROWSER_USE_HEADLESS'] = 'false' - profile2 = BrowserProfile() - assert profile2.headless is False, 'BrowserProfile should have headless=False when BROWSER_USE_HEADLESS=false' - finally: - if original is not None: - os.environ['BROWSER_USE_HEADLESS'] = original - else: - os.environ.pop('BROWSER_USE_HEADLESS', None) - - def test_explicit_param_overrides_env_var(self): - """Test that explicit headless parameter overrides env var.""" - original = os.environ.get('BROWSER_USE_HEADLESS') - try: - os.environ['BROWSER_USE_HEADLESS'] = 'true' - - from browser_use.browser.profile import BrowserProfile - - # Explicitly set to False should override env var - profile = BrowserProfile(headless=False) - assert profile.headless is False, 'Explicit param should override env var' - - os.environ['BROWSER_USE_HEADLESS'] = 'false' - profile2 = BrowserProfile(headless=True) - assert profile2.headless is True, 'Explicit param should override env var' - finally: - if original is not None: - os.environ['BROWSER_USE_HEADLESS'] = original - else: - os.environ.pop('BROWSER_USE_HEADLESS', None) - - def test_browser_session_uses_env_var(self): - """Test that BrowserSession picks up the env var via BrowserProfile.""" - original = os.environ.get('BROWSER_USE_HEADLESS') - try: - os.environ['BROWSER_USE_HEADLESS'] = 'true' - - from browser_use.browser import BrowserSession - - session = BrowserSession() - assert session.browser_profile.headless is True, ( - 'BrowserSession should have headless=True when BROWSER_USE_HEADLESS=true' - ) - finally: - if original is not None: - os.environ['BROWSER_USE_HEADLESS'] = original - else: - os.environ.pop('BROWSER_USE_HEADLESS', None) + @pytest.mark.parametrize( + 'env_var,attr_name,env_val,explicit_arg,expected', + [ + ('BROWSER_USE_DISABLE_EXTENSIONS', 'enable_default_extensions', 'true', {'enable_default_extensions': True}, True), + ('BROWSER_USE_DISABLE_EXTENSIONS', 'enable_default_extensions', 'false', {'enable_default_extensions': False}, False), + ('BROWSER_USE_HEADLESS', 'headless', 'true', {'headless': False}, False), + ('BROWSER_USE_HEADLESS', 'headless', 'false', {'headless': True}, True), + ], + ) + def test_explicit_parameter_overrides_env_var( + self, + monkeypatch: pytest.MonkeyPatch, + env_var: str, + attr_name: str, + env_val: str, + explicit_arg: dict, + expected: bool, + ): + """Test that explicit constructor parameters override env vars.""" + monkeypatch.setenv(env_var, env_val) + profile = BrowserProfile(**explicit_arg) + assert getattr(profile, attr_name) is expected From 7c6ac585f1438c0c9825f6426421a04f150183df Mon Sep 17 00:00:00 2001 From: Aneesh Sharma Date: Sat, 15 Aug 2026 18:56:52 +0530 Subject: [PATCH 21/69] test: remove unused parameters in env var truthy/falsy test signatures --- tests/ci/test_extension_config.py | 22 ++++++++++------------ 1 file changed, 10 insertions(+), 12 deletions(-) diff --git a/tests/ci/test_extension_config.py b/tests/ci/test_extension_config.py index 2f8e17631..6bb1e7d96 100644 --- a/tests/ci/test_extension_config.py +++ b/tests/ci/test_extension_config.py @@ -25,10 +25,10 @@ class TestConfigEnvVars: assert _get_headless_default() is None @pytest.mark.parametrize( - 'env_var,getter,truthy_expected,falsy_expected', + 'env_var,getter,expected', [ - ('BROWSER_USE_DISABLE_EXTENSIONS', _get_enable_default_extensions_default, False, True), - ('BROWSER_USE_HEADLESS', _get_headless_default, True, False), + ('BROWSER_USE_DISABLE_EXTENSIONS', _get_enable_default_extensions_default, False), + ('BROWSER_USE_HEADLESS', _get_headless_default, True), ], ) def test_env_var_truthy_values( @@ -36,19 +36,18 @@ class TestConfigEnvVars: monkeypatch: pytest.MonkeyPatch, env_var: str, getter, - truthy_expected: bool, - falsy_expected: bool, + expected: bool, ): """Test truthy env var values are parsed correctly.""" for val in TRUTHY_STRINGS: monkeypatch.setenv(env_var, val) - assert getter() is truthy_expected, f'Failed for {env_var}={val}' + assert getter() is expected, f'Failed for {env_var}={val}' @pytest.mark.parametrize( - 'env_var,getter,truthy_expected,falsy_expected', + 'env_var,getter,expected', [ - ('BROWSER_USE_DISABLE_EXTENSIONS', _get_enable_default_extensions_default, False, True), - ('BROWSER_USE_HEADLESS', _get_headless_default, True, False), + ('BROWSER_USE_DISABLE_EXTENSIONS', _get_enable_default_extensions_default, True), + ('BROWSER_USE_HEADLESS', _get_headless_default, False), ], ) def test_env_var_falsy_values( @@ -56,13 +55,12 @@ class TestConfigEnvVars: monkeypatch: pytest.MonkeyPatch, env_var: str, getter, - truthy_expected: bool, - falsy_expected: bool, + expected: bool, ): """Test falsy env var values are parsed correctly.""" for val in FALSY_STRINGS: monkeypatch.setenv(env_var, val) - assert getter() is falsy_expected, f'Failed for {env_var}={val}' + assert getter() is expected, f'Failed for {env_var}={val}' @pytest.mark.parametrize( 'env_var,attr_name,truthy_val,falsy_val', From fa57d340e2bf771286db67c6754fbda50a4f0cf8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gregor=20=C5=BDuni=C4=8D?= <36313686+gregpr07@users.noreply.github.com> Date: Sun, 16 Aug 2026 11:24:15 -0700 Subject: [PATCH 22/69] Release 0.13.8 with Browser Harness 0.1.9 --- browser_use/skills/browser-use/SKILL.md | 4 ++++ pyproject.toml | 16 ++++++++-------- skills/browser-use/SKILL.md | 4 ++++ 3 files changed, 16 insertions(+), 8 deletions(-) diff --git a/browser_use/skills/browser-use/SKILL.md b/browser_use/skills/browser-use/SKILL.md index 0c8f6cc22..356e12fe9 100644 --- a/browser_use/skills/browser-use/SKILL.md +++ b/browser_use/skills/browser-use/SKILL.md @@ -44,6 +44,10 @@ PY - Invoke as `browser-use`. Use heredocs for multi-line commands. - Helpers are pre-imported. `run.py` calls `ensure_daemon()` before `exec`. - First navigation is `new_tab(url)`, not `goto_url(url)`. +- `new_tab()` and `switch_tab()` attach and move the horse marker without + changing Chrome's visible tab. Screenshots and normal CDP input work in the + background; call `activate_tab(target)` only when the user explicitly asks + or a page demonstrably pauses rendering while hidden. - The normal local flow attaches to the running Chrome/Chromium CDP endpoint. No browser ids or local profile selection. ## Local Chrome diff --git a/pyproject.toml b/pyproject.toml index c3c3073a8..ba89a26ad 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -2,7 +2,7 @@ name = "browser-use" description = "Make websites accessible for AI agents" authors = [{ name = "Gregor Zunic" }] -version = "0.13.7" +version = "0.13.8" readme = "README.md" requires-python = ">=3.11,<4.0" classifiers = [ @@ -41,12 +41,12 @@ dependencies = [ "reportlab==4.4.9", "cdp-use==1.4.5", "pyotp==2.9.0", - "pillow==12.2.0", + "pillow==12.3.0", "cloudpickle==3.1.2", "markdownify==1.2.2", "python-docx==1.2.0", "browser-use-sdk==3.4.2", - "browser-harness==0.1.8", + "browser-harness==0.1.9", ] # google-api-core: only used for Google LLM APIs # pyperclip: only used for examples that use copy/paste @@ -60,11 +60,11 @@ dependencies = [ [project.optional-dependencies] cli = [] core = [ - "browser-use-core==0.13.2; sys_platform == 'darwin' and platform_machine == 'arm64'", - "browser-use-core==0.13.2; sys_platform == 'darwin' and platform_machine == 'x86_64'", - "browser-use-core==0.13.2; sys_platform == 'linux' and platform_machine == 'x86_64'", - "browser-use-core==0.13.2; sys_platform == 'linux' and platform_machine == 'aarch64'", - "browser-use-core==0.13.2; sys_platform == 'win32' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')", + "browser-use-core==0.13.3; sys_platform == 'darwin' and platform_machine == 'arm64'", + "browser-use-core==0.13.3; sys_platform == 'darwin' and platform_machine == 'x86_64'", + "browser-use-core==0.13.3; sys_platform == 'linux' and platform_machine == 'x86_64'", + "browser-use-core==0.13.3; sys_platform == 'linux' and platform_machine == 'aarch64'", + "browser-use-core==0.13.3; sys_platform == 'win32' and (platform_machine == 'AMD64' or platform_machine == 'x86_64')", ] aws = ["boto3==1.42.37"] oci = ["oci==2.166.0"] diff --git a/skills/browser-use/SKILL.md b/skills/browser-use/SKILL.md index 0c8f6cc22..356e12fe9 100644 --- a/skills/browser-use/SKILL.md +++ b/skills/browser-use/SKILL.md @@ -44,6 +44,10 @@ PY - Invoke as `browser-use`. Use heredocs for multi-line commands. - Helpers are pre-imported. `run.py` calls `ensure_daemon()` before `exec`. - First navigation is `new_tab(url)`, not `goto_url(url)`. +- `new_tab()` and `switch_tab()` attach and move the horse marker without + changing Chrome's visible tab. Screenshots and normal CDP input work in the + background; call `activate_tab(target)` only when the user explicitly asks + or a page demonstrably pauses rendering while hidden. - The normal local flow attaches to the running Chrome/Chromium CDP endpoint. No browser ids or local profile selection. ## Local Chrome From 5953df7d2f9d4432964534af570758aa6e56e3a1 Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Mon, 17 Aug 2026 09:19:26 -0700 Subject: [PATCH 23/69] fix(llm): make bu-2-0 the default again, keep mini opt-in 0.13.8 shipped bu-2-0-mini-preview as the ChatBrowserUse default, which means Agent(task=...) with no llm - and every bare ChatBrowserUse() - silently moved onto a preview model on upgrade, with no code change on the caller's side. Two problems with that. A preview id can change behaviour or be renamed, so it is the wrong thing to reach by omission. And per-token price is the wrong yardstick for an agent: total cost is tokens-per-step times steps, and steps is a function of model quality, so a cheaper-per-token model that needs more steps to finish can cost more and take longer. We do not yet have a per-task benchmark number for mini to say which way that goes. bu-2-0-mini-preview stays a first-class option: still accepted, still priced, still what the examples demonstrate. It is just opted into by name now rather than landed on by default. Revisit once the benchmark number exists. --- browser_use/llm/browser_use/chat.py | 10 +++++----- examples/beta_agent/basic.py | 2 +- examples/models/browser_use_llm.py | 5 +++-- skills/open-source/references/models.md | 9 ++++----- tests/ci/models/test_llm_browseruse.py | 12 +++++++----- 5 files changed, 20 insertions(+), 18 deletions(-) diff --git a/browser_use/llm/browser_use/chat.py b/browser_use/llm/browser_use/chat.py index b014b0f32..c2f29da07 100644 --- a/browser_use/llm/browser_use/chat.py +++ b/browser_use/llm/browser_use/chat.py @@ -44,7 +44,7 @@ class ChatBrowserUse(BaseChatModel): def __init__( self, - model: str = 'bu-2-0-mini-preview', + model: str = 'bu-2-0', api_key: str | None = None, base_url: str | None = None, timeout: float = 120.0, @@ -58,8 +58,8 @@ class ChatBrowserUse(BaseChatModel): Args: model: Model name to use. Options: - - 'bu-2-0-mini-preview': Default model (fast + cheap, preview) - - 'bu-2-0' or 'bu-latest': Premium model + - 'bu-2-0' or 'bu-latest': Default model (premium) + - 'bu-2-0-mini-preview': Cheaper and faster per token, opt-in while in preview - 'bu-1-0': Previous generation model, redirected to bu-2-0 at the gateway - 'bu-qa-1': Website QA model (tests a site and scores functionality/aesthetics) - 'browser-use/bu-30b-a3b-preview': Browser Use Open Source Model @@ -83,8 +83,8 @@ class ChatBrowserUse(BaseChatModel): "'openai/gpt-5.5', or 'google/gemini-3-pro'." ) - # Normalize bu-latest to the current latest model. Deliberately not the constructor - # default: 'latest' tracks the stable premium line, not the preview. + # Normalize bu-latest to the current latest model, which is also the default: a + # preview model is opt-in, never something a caller lands on by omission. if model == 'bu-latest': self.model = 'bu-2-0' else: diff --git a/examples/beta_agent/basic.py b/examples/beta_agent/basic.py index bb7f7a8e0..1eed65ca4 100644 --- a/examples/beta_agent/basic.py +++ b/examples/beta_agent/basic.py @@ -23,7 +23,7 @@ async def main() -> None: agent = Agent( task=task, llm=ChatBrowserUse(model='openai/gpt-5.5'), - # llm=ChatBrowserUse(), # Browser Use's own optimized model (bu-2-0-mini-preview) + # llm=ChatBrowserUse(), # Browser Use's own optimized model (bu-2-0) # llm=ChatOpenAI(model='gpt-5.5'), # llm=ChatGoogle(model='gemini-3.1-pro-preview'), # llm=ChatAnthropic(model='claude-opus-4-8'), # Sonnet also works well. diff --git a/examples/models/browser_use_llm.py b/examples/models/browser_use_llm.py index 3cb441039..1ad1e9e33 100644 --- a/examples/models/browser_use_llm.py +++ b/examples/models/browser_use_llm.py @@ -20,8 +20,9 @@ if not os.getenv('BROWSER_USE_API_KEY'): async def main(): - # `bu-2-0-mini-preview` is the optimized default - what a bare `ChatBrowserUse()` gives you. - # For the larger premium model pass `model='bu-2-0'` (or 'bu-latest', which tracks it). + # A bare `ChatBrowserUse()` gives you `bu-2-0`, the premium default (as does 'bu-latest'). + # `bu-2-0-mini-preview`, used below, is cheaper and faster per token but is in preview, so + # you opt into it by name rather than getting it by default. # ChatBrowserUse can also route to provider-prefixed models (e.g. 'anthropic/claude-sonnet-4-6', # 'openai/gpt-5.5', 'google/gemini-3-pro') through the same gateway - see # browser_use_provider_models.py. diff --git a/skills/open-source/references/models.md b/skills/open-source/references/models.md index 590e7df2f..30438aec7 100644 --- a/skills/open-source/references/models.md +++ b/skills/open-source/references/models.md @@ -59,9 +59,8 @@ Optimized for browser automation โ€” highest accuracy, fastest speed, lowest tok ```python from browser_use import Agent, ChatBrowserUse -llm = ChatBrowserUse() # bu-2-0-mini-preview (default) -llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Default, named explicitly -llm = ChatBrowserUse(model='bu-2-0') # Premium model ('bu-latest' tracks this) +llm = ChatBrowserUse() # bu-2-0 (default, 'bu-latest' tracks it) +llm = ChatBrowserUse(model='bu-2-0-mini-preview') # Cheaper per token, opt-in while in preview ``` **Env:** `BROWSER_USE_API_KEY` โ€” get at https://cloud.browser-use.com/new-api-key @@ -69,8 +68,8 @@ llm = ChatBrowserUse(model='bu-2-0') # Premium model ('bu-latest' **Models & Pricing (per 1M tokens):** | Model | Input | Cached | Output | |-------|-------|--------|--------| -| bu-2-0-mini-preview (default) | $0.15 | $0.15 | $1.50 | -| bu-2-0 (premium) | $0.60 | $0.06 | $3.50 | +| bu-2-0 (default, premium) | $0.60 | $0.06 | $3.50 | +| bu-2-0-mini-preview (opt-in) | $0.15 | $0.15 | $1.50 | | bu-1-0 (redirects to bu-2-0) | $0.60 | $0.06 | $3.50 | | browser-use/bu-30b-a3b-preview (OSS) | โ€” | โ€” | โ€” | diff --git a/tests/ci/models/test_llm_browseruse.py b/tests/ci/models/test_llm_browseruse.py index 166a176fe..b29d8d17c 100644 --- a/tests/ci/models/test_llm_browseruse.py +++ b/tests/ci/models/test_llm_browseruse.py @@ -25,10 +25,12 @@ async def test_browseruse_bu_latest(httpserver): # --- Model validation ------------------------------------------------------- -def test_default_model_is_bu_2_0_mini_preview(): +def test_default_model_is_bu_2_0(): + """The default must be a stable model - a preview is opt-in, never reached by omission.""" chat = ChatBrowserUse(api_key=TEST_API_KEY) - assert chat.model == 'bu-2-0-mini-preview' + assert chat.model == 'bu-2-0' assert chat.provider == 'browser-use' + assert 'preview' not in chat.model @pytest.mark.parametrize('alias', ['bu-1-0', 'bu-2-0', 'bu-2-0-mini-preview', 'bu-qa-1']) @@ -39,12 +41,12 @@ def test_bu_aliases_are_accepted(alias): assert chat.provider == 'browser-use' -def test_bu_latest_still_normalizes_to_bu_2_0(): - """'latest' tracks the stable premium line, so it does NOT follow the preview default.""" +def test_bu_latest_normalizes_to_bu_2_0(): + """'latest' tracks the stable premium line, which is also what omitting the model gives.""" chat = ChatBrowserUse(model='bu-latest', api_key=TEST_API_KEY) assert chat.model == 'bu-2-0' assert chat.name == 'bu-2-0' - assert ChatBrowserUse(api_key=TEST_API_KEY).model != chat.model + assert ChatBrowserUse(api_key=TEST_API_KEY).model == chat.model def test_bu_2_0_mini_preview_is_priced(): From f9c933ed5e4d942c468725f71a34895b09b09ddf Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Mon, 17 Aug 2026 17:21:34 -0700 Subject: [PATCH 24/69] ci: drop the dead cloud_evals image-build trigger MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The only job in this workflow dispatched a `trigger-workflow` event to browser-use/cloud using TRIGGER_CLOUD_BUILD_GH_KEY. That token stopped working after 2026-04-25 โ€” every run since has failed with a 401 "Bad credentials" on POST /repos/browser-use/cloud/dispatches, so the cloud eval image has not actually been built from this trigger in ~4 months. Nothing else references the workflow and it is not a required status check on main. --- .github/workflows/cloud_evals.yml | 35 ------------------------------- 1 file changed, 35 deletions(-) delete mode 100644 .github/workflows/cloud_evals.yml diff --git a/.github/workflows/cloud_evals.yml b/.github/workflows/cloud_evals.yml deleted file mode 100644 index 9dd97f482..000000000 --- a/.github/workflows/cloud_evals.yml +++ /dev/null @@ -1,35 +0,0 @@ -name: cloud_evals - -# Cancel in-progress runs when a new commit is pushed to the same branch/PR -concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} - cancel-in-progress: true - -on: - push: - branches: - - main - - 'releases/*' - workflow_dispatch: - inputs: - commit_hash: - description: Commit hash of the library to build the Cloud eval image for - required: false - -permissions: {} - -jobs: - trigger_cloud_eval_image_build: - runs-on: ubuntu-latest - steps: - - uses: actions/github-script@v7 - with: - github-token: ${{ secrets.TRIGGER_CLOUD_BUILD_GH_KEY }} - script: | - const result = await github.rest.repos.createDispatchEvent({ - owner: 'browser-use', - repo: 'cloud', - event_type: 'trigger-workflow', - client_payload: {"commit_hash": "${{ github.event.inputs.commit_hash || github.sha }}"} - }) - console.log(result) From 3648bbad7f2aa9e8447ff796a54ffbde840a789d Mon Sep 17 00:00:00 2001 From: uczltw6 <62146150+uczltw6@users.noreply.github.com> Date: Wed, 19 Aug 2026 10:55:52 +0800 Subject: [PATCH 25/69] fix(filesystem): report missing replacement text Return an error when the target text is absent instead of reporting a successful no-op. Assisted-by: OpenAI Codex --- browser_use/filesystem/file_system.py | 2 ++ tests/ci/infrastructure/test_filesystem.py | 12 ++++++++++++ 2 files changed, 14 insertions(+) diff --git a/browser_use/filesystem/file_system.py b/browser_use/filesystem/file_system.py index 103b4b7da..097eca84c 100644 --- a/browser_use/filesystem/file_system.py +++ b/browser_use/filesystem/file_system.py @@ -792,6 +792,8 @@ class FileSystem: try: content = file_obj.read() + if old_str not in content: + return f'Error: Could not find the specified text in file {full_filename}.' content = content.replace(old_str, new_str) await file_obj.write(content, self.data_dir) sanitize_note = f" (auto-corrected from '{original_filename}')" if was_sanitized else '' diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index 273be8a0d..d8196f379 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -478,6 +478,18 @@ class TestFileSystem: assert 'not found' in result assert 'auto-corrected' in result + async def test_replace_file_reports_missing_text(self, temp_filesystem): + """Test that replacing absent text reports an error without changing the file.""" + fs = temp_filesystem + original_content = '- [ ] First task\n- [ ] Second task' + await fs.write_file('todo.md', original_content) + + result = await fs.replace_file_str('todo.md', '- [ ] Missing task', '- [x] Missing task') + + assert result == 'Error: Could not find the specified text in file todo.md.' + assert fs.get_file('todo.md').content == original_content + assert (fs.data_dir / 'todo.md').read_text(encoding='utf-8') == original_content + async def test_append_json_file(self, temp_filesystem): """Test appending content to JSON files.""" fs = temp_filesystem From 8e880168a94119422fbbccd6f9753e97c104235b Mon Sep 17 00:00:00 2001 From: JayLay0310 <2024020752@bistu.edu.cn> Date: Thu, 20 Aug 2026 09:34:18 +0800 Subject: [PATCH 26/69] fix(agent): match URL negations as whole words --- browser_use/agent/service.py | 18 +++++------------- browser_use/beta/service.py | 5 ++--- browser_use/utils.py | 6 ++++++ tests/ci/test_beta_agent.py | 4 ++++ 4 files changed, 17 insertions(+), 16 deletions(-) diff --git a/browser_use/agent/service.py b/browser_use/agent/service.py index 3e0970ddd..304f5765a 100644 --- a/browser_use/agent/service.py +++ b/browser_use/agent/service.py @@ -77,6 +77,7 @@ from browser_use.utils import ( _log_pretty_path, check_latest_browser_use_version, get_browser_use_version, + has_url_negation, is_placeholder_url, sanitize_url_candidate, time_execution_async, @@ -2367,13 +2368,6 @@ class Agent(Generic[Context, AgentStructuredOutput]): 'polynomial', } - excluded_words = { - 'never', - 'dont', - 'not', - "don't", - } - found_urls = [] matched_spans: list[tuple[int, int]] = [] for pattern in patterns: @@ -2411,16 +2405,14 @@ class Agent(Generic[Context, AgentStructuredOutput]): self.logger.debug(f'Excluding URL with file extension from auto-navigation: {url}') continue - # If in the 20 characters before the url position is a word in excluded_words skip to avoid "Never go to this url" + # Skip URLs explicitly negated by nearby prose, such as "Never go to this URL". context_start = max(0, original_position - 20) context_text = task_without_emails[context_start:original_position] - if any(word.lower() in context_text.lower() for word in excluded_words): - self.logger.debug( - f'Excluding URL with word in excluded words from auto-navigation: {url} (context: "{context_text.strip()}")' - ) + if has_url_negation(context_text): + self.logger.debug(f'Excluding negated URL from auto-navigation: {url} (context: "{context_text.strip()}")') continue - # Add https:// if missing (after excluded words check to avoid position calculation issues) + # Add https:// after the negation check to preserve source positions. if not has_scheme: url = 'https://' + url diff --git a/browser_use/beta/service.py b/browser_use/beta/service.py index d8f116688..3a0bb4c45 100644 --- a/browser_use/beta/service.py +++ b/browser_use/beta/service.py @@ -76,6 +76,7 @@ from browser_use.utils import ( check_latest_browser_use_version, get_browser_use_version, get_git_info, + has_url_negation, is_placeholder_url, sanitize_url_candidate, ) @@ -1460,8 +1461,6 @@ def _extract_start_url(task: str) -> str | None: 'rpm', 'iso', } - excluded_words = {'never', 'dont', 'not', "don't"} - found_urls = [] matched_spans: list[tuple[int, int]] = [] for pattern in patterns: @@ -1483,7 +1482,7 @@ def _extract_start_url(task: str) -> str | None: continue context_start = max(0, match.start() - 20) context_text = task_without_emails[context_start : match.start()] - if any(word in context_text.lower() for word in excluded_words): + if has_url_negation(context_text): continue if not has_scheme: url = 'https://' + url diff --git a/browser_use/utils.py b/browser_use/utils.py index 03e651d38..5cbfee760 100644 --- a/browser_use/utils.py +++ b/browser_use/utils.py @@ -20,6 +20,7 @@ load_dotenv() # Pre-compiled regex for URL detection - used in URL shortening URL_PATTERN = re.compile(r'https?://[^\s<>"\']+|www\.[^\s<>"\']+|[^\s<>"\']+\.[a-z]{2,}(?:/[^\s<>"\']*)?', re.IGNORECASE) +URL_NEGATION_PATTERN = re.compile(r"\b(?:never|not|don'?t)\b", re.IGNORECASE) logger = logging.getLogger(__name__) @@ -49,6 +50,11 @@ def sanitize_url_candidate(url: str) -> str: return re.sub(r'[.,;:!?()\[\]]+$', '', candidate) +def has_url_negation(context: str) -> bool: + """Return whether nearby prose explicitly negates navigation to a URL.""" + return URL_NEGATION_PATTERN.search(context) is not None + + # Lazy import for error types # Use sentinel to avoid retrying import when package is not installed _IMPORT_NOT_FOUND: type = type('_ImportNotFound', (), {}) diff --git a/tests/ci/test_beta_agent.py b/tests/ci/test_beta_agent.py index 82aef6f9c..359f731e9 100644 --- a/tests/ci/test_beta_agent.py +++ b/tests/ci/test_beta_agent.py @@ -4297,6 +4297,10 @@ def test_beta_agent_exposes_task_helper_methods(): assert 'Expected output format: Answer' in enhanced assert '"answer"' in enhanced assert agent._extract_start_url('Open example.com and report the title.') == 'https://example.com' + assert agent._extract_start_url('Open this notable site: https://example.com') == 'https://example.com' + assert browser_use_agent._extract_start_url('Open this notable site: https://example.com') == 'https://example.com' + assert agent._extract_start_url('Do not open https://example.com.') is None + assert browser_use_agent._extract_start_url('Do not open https://example.com.') is None assert agent._extract_start_url('Email support@example.com only.') is None assert agent._extract_start_url('Open https://example.com/report.pdf and summarize it.') is None assert agent._extract_start_url('Use https://XXX.XX as a placeholder in the table.') is None From fcf0b9229ffe9ed24605f4db2af220c8554d2821 Mon Sep 17 00:00:00 2001 From: Lai Yujie <2024020752@bistu.edu.cn> Date: Thu, 20 Aug 2026 09:54:07 +0800 Subject: [PATCH 27/69] Update browser_use/utils.py Co-authored-by: cubic-dev-ai[bot] <191113872+cubic-dev-ai[bot]@users.noreply.github.com> --- browser_use/utils.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/browser_use/utils.py b/browser_use/utils.py index 5cbfee760..2dec60bd7 100644 --- a/browser_use/utils.py +++ b/browser_use/utils.py @@ -20,7 +20,7 @@ load_dotenv() # Pre-compiled regex for URL detection - used in URL shortening URL_PATTERN = re.compile(r'https?://[^\s<>"\']+|www\.[^\s<>"\']+|[^\s<>"\']+\.[a-z]{2,}(?:/[^\s<>"\']*)?', re.IGNORECASE) -URL_NEGATION_PATTERN = re.compile(r"\b(?:never|not|don'?t)\b", re.IGNORECASE) +URL_NEGATION_PATTERN = re.compile(r"\b(?:never|not|don['\u2019]?t)\b", re.IGNORECASE) logger = logging.getLogger(__name__) From a6846b698a04aa59efb879eb9563f35afa42a87e Mon Sep 17 00:00:00 2001 From: primorLee Date: Thu, 20 Aug 2026 13:03:35 +0800 Subject: [PATCH 28/69] fix(filesystem): escape plain text in generated PDFs --- browser_use/filesystem/file_system.py | 10 ++++++---- tests/ci/infrastructure/test_filesystem.py | 12 ++++++++++++ 2 files changed, 18 insertions(+), 4 deletions(-) diff --git a/browser_use/filesystem/file_system.py b/browser_use/filesystem/file_system.py index 097eca84c..1eb5b1103 100644 --- a/browser_use/filesystem/file_system.py +++ b/browser_use/filesystem/file_system.py @@ -1,6 +1,7 @@ import asyncio import base64 import csv +import html import io import os import re @@ -264,15 +265,16 @@ class PdfFile(BaseFile): for line in content_lines: if line.strip(): + escaped_line = html.escape(line) # Handle basic markdown headers if line.startswith('# '): - para = Paragraph(line[2:], styles['Title']) + para = Paragraph(escaped_line[2:], styles['Title']) elif line.startswith('## '): - para = Paragraph(line[3:], styles['Heading1']) + para = Paragraph(escaped_line[3:], styles['Heading1']) elif line.startswith('### '): - para = Paragraph(line[4:], styles['Heading2']) + para = Paragraph(escaped_line[4:], styles['Heading2']) else: - para = Paragraph(line, styles['Normal']) + para = Paragraph(escaped_line, styles['Normal']) story.append(para) else: story.append(Spacer(1, 6)) diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index d8196f379..99aa1223c 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -5,6 +5,7 @@ import tempfile from pathlib import Path import pytest +from pypdf import PdfReader from browser_use.filesystem.file_system import ( DEFAULT_FILE_SYSTEM_PATH, @@ -222,6 +223,17 @@ class TestFileSystem: except Exception: pass + async def test_write_pdf_preserves_plain_text_angle_brackets(self, empty_filesystem): + """PDF content should be rendered as plain text, not ReportLab markup.""" + content = 'Comparison: 2 4' + + result = await empty_filesystem.write_file('comparison.pdf', content) + + assert result == 'Data written to file comparison.pdf successfully.' + pdf_path = empty_filesystem.data_dir / 'comparison.pdf' + extracted_text = '\n'.join(page.extract_text() or '' for page in PdfReader(pdf_path).pages) + assert content in extracted_text + def test_filesystem_initialization(self, temp_filesystem): """Test FileSystem initialization with default files.""" fs = temp_filesystem From a9d426ab57ce0857edf69d7fe7cb5c3dd1d7d9a3 Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Mon, 24 Aug 2026 11:01:37 -0700 Subject: [PATCH 29/69] refactor(filesystem): strip header markers before escaping PDF text Escaping first only worked because '#' and ' ' are not escapable, so the prefix length happened to survive html.escape and the [2:]/[3:]/[4:] offsets still landed correctly. Slice first, escape once at the point of use, and cover the three header branches in the regression test. --- browser_use/filesystem/file_system.py | 12 ++++++------ tests/ci/infrastructure/test_filesystem.py | 11 ++++++++--- 2 files changed, 14 insertions(+), 9 deletions(-) diff --git a/browser_use/filesystem/file_system.py b/browser_use/filesystem/file_system.py index 1eb5b1103..8ddbdbeaf 100644 --- a/browser_use/filesystem/file_system.py +++ b/browser_use/filesystem/file_system.py @@ -265,17 +265,17 @@ class PdfFile(BaseFile): for line in content_lines: if line.strip(): - escaped_line = html.escape(line) # Handle basic markdown headers if line.startswith('# '): - para = Paragraph(escaped_line[2:], styles['Title']) + text, style = line[2:], styles['Title'] elif line.startswith('## '): - para = Paragraph(escaped_line[3:], styles['Heading1']) + text, style = line[3:], styles['Heading1'] elif line.startswith('### '): - para = Paragraph(escaped_line[4:], styles['Heading2']) + text, style = line[4:], styles['Heading2'] else: - para = Paragraph(escaped_line, styles['Normal']) - story.append(para) + text, style = line, styles['Normal'] + # Paragraph parses its input as ReportLab markup, but our content is plain text + story.append(Paragraph(html.escape(text), style)) else: story.append(Spacer(1, 6)) diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index 99aa1223c..c96c35294 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -225,14 +225,19 @@ class TestFileSystem: async def test_write_pdf_preserves_plain_text_angle_brackets(self, empty_filesystem): """PDF content should be rendered as plain text, not ReportLab markup.""" - content = 'Comparison: 2 4' + lines = ['# Q&A: ', 'Comparison: 2 4'] - result = await empty_filesystem.write_file('comparison.pdf', content) + result = await empty_filesystem.write_file('comparison.pdf', '\n\n'.join(lines)) assert result == 'Data written to file comparison.pdf successfully.' pdf_path = empty_filesystem.data_dir / 'comparison.pdf' extracted_text = '\n'.join(page.extract_text() or '' for page in PdfReader(pdf_path).pages) - assert content in extracted_text + # Header markers are stripped, everything else survives verbatim + assert 'Q&A: ' in extracted_text + assert 'Comparison: 2 4' in extracted_text + assert '#' not in extracted_text def test_filesystem_initialization(self, temp_filesystem): """Test FileSystem initialization with default files.""" From 0193edfb099cda87b9e644d8b4b357797e5a6448 Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Mon, 24 Aug 2026 12:13:12 -0700 Subject: [PATCH 30/69] test(filesystem): assert header markers are stripped, not that '#' is absent `'#' not in extracted_text` claimed more than the renderer promises: only a leading `# `/`## `/`### ` marker is stripped, hashes anywhere else are plain text. Assert each header's source line is gone and its rendered text survives, and cover a literal `#1` in header content. --- tests/ci/infrastructure/test_filesystem.py | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index c96c35294..a47d83a13 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -225,19 +225,24 @@ class TestFileSystem: async def test_write_pdf_preserves_plain_text_angle_brackets(self, empty_filesystem): """PDF content should be rendered as plain text, not ReportLab markup.""" - lines = ['# Q&A: ', 'Comparison: 2 4'] + # (markdown source, text expected in the PDF) + headers = [ + ('# Q&A: #1', 'Notes #1'), + ] + body = 'Comparison: 2 4' - result = await empty_filesystem.write_file('comparison.pdf', '\n\n'.join(lines)) + result = await empty_filesystem.write_file('comparison.pdf', '\n\n'.join([source for source, _ in headers] + [body])) assert result == 'Data written to file comparison.pdf successfully.' pdf_path = empty_filesystem.data_dir / 'comparison.pdf' extracted_text = '\n'.join(page.extract_text() or '' for page in PdfReader(pdf_path).pages) - # Header markers are stripped, everything else survives verbatim - assert 'Q&A: ' in extracted_text - assert 'Comparison: 2 4' in extracted_text - assert '#' not in extracted_text + assert body in extracted_text + for source, rendered in headers: + # Only the leading header marker is stripped, everything else survives verbatim + assert rendered in extracted_text + assert source not in extracted_text def test_filesystem_initialization(self, temp_filesystem): """Test FileSystem initialization with default files.""" From e62d4d2b436883aefab50e8027fc1e735ed857de Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Mon, 24 Aug 2026 12:32:08 -0700 Subject: [PATCH 31/69] test(filesystem): match whole extracted lines in the PDF regression test A substring check passes with a leftover marker: 'Notes #1' matches inside '# Notes #1', so an H3 branch using the H1 offset would slip through. Compare against whole extracted lines instead, and cover the gaps that turned up alongside it: a non-header line starting with '#' (pins the 'startswith("# ")' boundary against an lstrip-style rewrite) and quote characters (html.escape emits '/", which ReportLab must unescape back). Rename to match what the test now covers. --- tests/ci/infrastructure/test_filesystem.py | 17 +++++++++-------- 1 file changed, 9 insertions(+), 8 deletions(-) diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index a47d83a13..2d2ab7708 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -223,7 +223,7 @@ class TestFileSystem: except Exception: pass - async def test_write_pdf_preserves_plain_text_angle_brackets(self, empty_filesystem): + async def test_write_pdf_renders_content_as_plain_text(self, empty_filesystem): """PDF content should be rendered as plain text, not ReportLab markup.""" # (markdown source, text expected in the PDF) headers = [ @@ -231,18 +231,19 @@ class TestFileSystem: ('## Section 2 & 3', 'Section 2 & 3'), ('### Notes #1', 'Notes #1'), ] - body = 'Comparison: 2 4' + body = ['Comparison: 2 4', '#hashtag is not a header', 'O\'Brien said "hi" & left'] - result = await empty_filesystem.write_file('comparison.pdf', '\n\n'.join([source for source, _ in headers] + [body])) + result = await empty_filesystem.write_file('comparison.pdf', '\n\n'.join([source for source, _ in headers] + body)) assert result == 'Data written to file comparison.pdf successfully.' pdf_path = empty_filesystem.data_dir / 'comparison.pdf' extracted_text = '\n'.join(page.extract_text() or '' for page in PdfReader(pdf_path).pages) - assert body in extracted_text - for source, rendered in headers: - # Only the leading header marker is stripped, everything else survives verbatim - assert rendered in extracted_text - assert source not in extracted_text + # Match whole lines: a substring check still passes with a leftover '# ' marker + extracted_lines = [line.strip() for line in extracted_text.splitlines()] + for _, rendered in headers: + assert rendered in extracted_lines + for line in body: + assert line in extracted_lines def test_filesystem_initialization(self, temp_filesystem): """Test FileSystem initialization with default files.""" From e6b92bec5648c485e778568edb28ca28447c1f81 Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Mon, 24 Aug 2026 12:38:16 -0700 Subject: [PATCH 32/69] test(filesystem): make PDF assertions wrap-proof and cover the append path Whole-line matching broke on any content long enough to wrap, so compare against a whitespace-collapsed blob instead and check leftover header markers separately against the rendered lines. Verified against headers and body lines that wrap over 2-3 lines. Add an append regression: append_file re-renders the whole document, so put the markup that trips ReportLab in the appended half to exercise that path on its own. Both PDF tests fail without the escape fix. Also condense the stale three-line comment above the render loop. --- browser_use/filesystem/file_system.py | 4 +-- tests/ci/infrastructure/test_filesystem.py | 32 ++++++++++++++++++---- 2 files changed, 27 insertions(+), 9 deletions(-) diff --git a/browser_use/filesystem/file_system.py b/browser_use/filesystem/file_system.py index 8ddbdbeaf..292636c83 100644 --- a/browser_use/filesystem/file_system.py +++ b/browser_use/filesystem/file_system.py @@ -258,9 +258,7 @@ class PdfFile(BaseFile): styles = getSampleStyleSheet() story = [] - # Convert markdown content to simple text and add to PDF - # For basic implementation, we'll treat content as plain text - # This avoids the AGPL license issue while maintaining functionality + # Plain text plus markdown headers only, to avoid an AGPL markdown-to-PDF dependency content_lines = self.content.split('\n') for line in content_lines: diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index 2d2ab7708..15bc60bed 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -20,6 +20,15 @@ from browser_use.filesystem.file_system import ( ) +def _extract_pdf_text(path: Path) -> tuple[str, list[str]]: + """Extract a PDF as (whitespace-collapsed blob, non-empty stripped lines). + + The blob tolerates line wrapping; the lines are what header markers would survive on. + """ + text = '\n'.join(page.extract_text() or '' for page in PdfReader(path).pages) + return ' '.join(text.split()), [line.strip() for line in text.splitlines() if line.strip()] + + class TestBaseFile: """Test the BaseFile abstract base class and its implementations.""" @@ -236,14 +245,25 @@ class TestFileSystem: result = await empty_filesystem.write_file('comparison.pdf', '\n\n'.join([source for source, _ in headers] + body)) assert result == 'Data written to file comparison.pdf successfully.' - pdf_path = empty_filesystem.data_dir / 'comparison.pdf' - extracted_text = '\n'.join(page.extract_text() or '' for page in PdfReader(pdf_path).pages) - # Match whole lines: a substring check still passes with a leftover '# ' marker - extracted_lines = [line.strip() for line in extracted_text.splitlines()] + blob, lines = _extract_pdf_text(empty_filesystem.data_dir / 'comparison.pdf') for _, rendered in headers: - assert rendered in extracted_lines + assert rendered in blob for line in body: - assert line in extracted_lines + assert line in blob + # Content checks alone pass with a leftover marker, e.g. '# Notes #1' contains 'Notes #1' + assert [line for line in lines if line.startswith(('# ', '## ', '### '))] == [] + + async def test_append_pdf_escapes_both_halves(self, empty_filesystem): + """Appending re-renders the whole PDF, so old and new content must both stay plain text.""" + # The markup that breaks ReportLab goes in the appended half, so this fails on the append path alone + await empty_filesystem.write_file('report.pdf', 'First & half') + + result = await empty_filesystem.append_file('report.pdf', '\nSecond 2') + + assert result == 'Data appended to file report.pdf successfully.' + blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'report.pdf') + assert 'First & half' in blob + assert 'Second 2' in blob def test_filesystem_initialization(self, temp_filesystem): """Test FileSystem initialization with default files.""" From 96117ef8f5597c7c1f6b70005ef2fdc21348801f Mon Sep 17 00:00:00 2001 From: UniversePeak Date: Tue, 25 Aug 2026 10:02:15 +0800 Subject: [PATCH 33/69] fix(dom): expose image context for clickable elements --- browser_use/dom/serializer/serializer.py | 53 ++++++++++++ .../ci/test_image_only_dom_representation.py | 81 +++++++++++++++++++ 2 files changed, 134 insertions(+) create mode 100644 tests/ci/test_image_only_dom_representation.py diff --git a/browser_use/dom/serializer/serializer.py b/browser_use/dom/serializer/serializer.py index 9f9017874..9fcee00c6 100644 --- a/browser_use/dom/serializer/serializer.py +++ b/browser_use/dom/serializer/serializer.py @@ -918,6 +918,52 @@ class DOMTreeSerializer: return False + @staticmethod + def _get_child_image_context(node: SimplifiedNode) -> str: + """Extract compact context from image descendants of an interactive element.""" + + image_context: list[str] = [] + + def normalize_src(src: str) -> str: + clean_src = src.strip() + if clean_src.lower().startswith('data:'): + return '' + path_without_query = clean_src.split('?', 1)[0].split('#', 1)[0].rstrip('/') + return path_without_query.rsplit('/', 1)[-1] or clean_src + + def collect(current: SimplifiedNode) -> None: + if len(image_context) >= 3: + return + + original_node = current.original_node + if original_node.node_type == NodeType.ELEMENT_NODE and original_node.tag_name == 'img': + attributes = original_node.attributes or {} + parts = [] + + for attr_name, output_name in ( + ('alt', 'image_alt'), + ('title', 'image_title'), + ('aria-label', 'image_label'), + ): + attr_value = attributes.get(attr_name, '').strip() + if attr_value: + parts.append(f'{output_name}={cap_text_length(attr_value, 100)}') + + src = normalize_src(attributes.get('src', '')) + if src: + parts.append(f'image_src={cap_text_length(src, 100)}') + + if parts: + image_context.append(' '.join(parts)) + + for child in current.children: + collect(child) + + for child in node.children: + collect(child) + + return ' '.join(image_context) + @staticmethod def serialize_tree(node: SimplifiedNode | None, include_attributes: list[str], depth: int = 0) -> str: """Serialize the optimized tree to string format.""" @@ -989,6 +1035,13 @@ class DOMTreeSerializer: attributes_html_str = DOMTreeSerializer._build_attributes_string( node.original_node, include_attributes, text_content ) + if node.is_interactive: + image_context = DOMTreeSerializer._get_child_image_context(node) + if image_context: + if attributes_html_str: + attributes_html_str += f' {image_context}' + else: + attributes_html_str = image_context # Add compound component information to attributes if present if node.original_node._compound_children: diff --git a/tests/ci/test_image_only_dom_representation.py b/tests/ci/test_image_only_dom_representation.py new file mode 100644 index 000000000..7bb83fcc0 --- /dev/null +++ b/tests/ci/test_image_only_dom_representation.py @@ -0,0 +1,81 @@ +from browser_use.dom.serializer.serializer import DOMTreeSerializer +from browser_use.dom.views import ( + DEFAULT_INCLUDE_ATTRIBUTES, + DOMRect, + EnhancedDOMTreeNode, + EnhancedSnapshotNode, + NodeType, + SimplifiedNode, +) + + +def _make_element_node( + backend_node_id: int, + tag_name: str, + attributes: dict[str, str], + x: float, + y: float, + width: float = 64, + height: float = 64, + parent: EnhancedDOMTreeNode | None = None, +) -> EnhancedDOMTreeNode: + bounds = DOMRect(x=x, y=y, width=width, height=height) + return EnhancedDOMTreeNode( + node_id=backend_node_id, + backend_node_id=backend_node_id, + node_type=NodeType.ELEMENT_NODE, + node_name=tag_name.upper(), + node_value='', + attributes=attributes, + is_scrollable=None, + is_visible=True, + absolute_position=bounds, + target_id='target-1', + frame_id=None, + session_id=None, + content_document=None, + shadow_root_type=None, + shadow_roots=None, + parent_node=parent, + children_nodes=None, + ax_node=None, + snapshot_node=EnhancedSnapshotNode( + is_clickable=tag_name in {'a', 'button'}, + cursor_style='pointer' if tag_name in {'a', 'button'} else None, + bounds=bounds, + clientRects=bounds, + scrollRects=None, + computed_styles=None, + paint_order=None, + stacking_contexts=None, + ), + ) + + +def test_image_only_interactive_parent_includes_child_image_context_in_llm_dom(): + """Image-only clickable cards should expose child image context.""" + link = _make_element_node(201, 'a', {'href': '/select-payment-method'}, x=10, y=10, width=80, height=80) + image = _make_element_node( + 202, + 'img', + {'src': 'https://cdn.example.test/logos/acme-bank-primary-card.png'}, + x=18, + y=18, + width=64, + height=64, + parent=link, + ) + link.children_nodes = [image] + + llm_dom = DOMTreeSerializer.serialize_tree( + SimplifiedNode( + original_node=link, + children=[SimplifiedNode(original_node=image, children=[])], + is_interactive=True, + selector_index=201, + ), + DEFAULT_INCLUDE_ATTRIBUTES, + ) + + assert '[201] Date: Tue, 25 Aug 2026 12:47:13 +0800 Subject: [PATCH 34/69] fix(dom): address image context review feedback --- browser_use/dom/serializer/serializer.py | 27 ++++--- .../ci/test_image_only_dom_representation.py | 77 +++++++++++++++++++ 2 files changed, 92 insertions(+), 12 deletions(-) diff --git a/browser_use/dom/serializer/serializer.py b/browser_use/dom/serializer/serializer.py index 9fcee00c6..827a9a35e 100644 --- a/browser_use/dom/serializer/serializer.py +++ b/browser_use/dom/serializer/serializer.py @@ -55,6 +55,8 @@ class DOMTreeSerializer: # {'tag': 'span', 'role': 'link'}, # ] DEFAULT_CONTAINMENT_THRESHOLD = 0.99 # 99% containment by default + MAX_CHILD_IMAGE_CONTEXTS = 3 + MAX_CHILD_IMAGE_DESCENDANTS = 100 def __init__( self, @@ -929,12 +931,17 @@ class DOMTreeSerializer: if clean_src.lower().startswith('data:'): return '' path_without_query = clean_src.split('?', 1)[0].split('#', 1)[0].rstrip('/') - return path_without_query.rsplit('/', 1)[-1] or clean_src - - def collect(current: SimplifiedNode) -> None: - if len(image_context) >= 3: - return + return path_without_query.rsplit('/', 1)[-1] + pending = list(reversed(node.children)) + visited_descendants = 0 + while ( + pending + and visited_descendants < DOMTreeSerializer.MAX_CHILD_IMAGE_DESCENDANTS + and len(image_context) < DOMTreeSerializer.MAX_CHILD_IMAGE_CONTEXTS + ): + current = pending.pop() + visited_descendants += 1 original_node = current.original_node if original_node.node_type == NodeType.ELEMENT_NODE and original_node.tag_name == 'img': attributes = original_node.attributes or {} @@ -945,22 +952,18 @@ class DOMTreeSerializer: ('title', 'image_title'), ('aria-label', 'image_label'), ): - attr_value = attributes.get(attr_name, '').strip() + attr_value = str(attributes.get(attr_name) or '').strip() if attr_value: parts.append(f'{output_name}={cap_text_length(attr_value, 100)}') - src = normalize_src(attributes.get('src', '')) + src = normalize_src(str(attributes.get('src') or '')) if src: parts.append(f'image_src={cap_text_length(src, 100)}') if parts: image_context.append(' '.join(parts)) - for child in current.children: - collect(child) - - for child in node.children: - collect(child) + pending.extend(reversed(current.children)) return ' '.join(image_context) diff --git a/tests/ci/test_image_only_dom_representation.py b/tests/ci/test_image_only_dom_representation.py index 7bb83fcc0..65a6c92f5 100644 --- a/tests/ci/test_image_only_dom_representation.py +++ b/tests/ci/test_image_only_dom_representation.py @@ -79,3 +79,80 @@ def test_image_only_interactive_parent_includes_child_image_context_in_llm_dom() assert '[201] SimplifiedNode: + image = _make_element_node(backend_node_id, 'img', attributes, x=18, y=18) + return SimplifiedNode(original_node=image, children=[]) + + +def test_child_image_context_strips_query_and_fragment_without_leaking_query_only_sources(): + parent = _make_element_node(301, 'a', {'href': '/cards'}, x=10, y=10) + node = SimplifiedNode( + original_node=parent, + children=[ + _simplified_image(302, {'src': 'https://cdn.test/card.png?token=secret#preview'}), + _simplified_image(303, {'src': '?token=must-not-leak'}), + ], + ) + + context = DOMTreeSerializer._get_child_image_context(node) + + assert context == 'image_src=card.png' + assert 'secret' not in context + assert 'must-not-leak' not in context + + +def test_child_image_context_ignores_data_src_but_keeps_accessible_attributes(): + parent = _make_element_node(401, 'button', {}, x=10, y=10) + node = SimplifiedNode( + original_node=parent, + children=[ + _simplified_image( + 402, + { + 'src': 'data:image/png;base64,private-payload', + 'alt': 'Payment card', + 'title': 'Choose card', + 'aria-label': 'Primary payment method', + }, + ), + ], + ) + + context = DOMTreeSerializer._get_child_image_context(node) + + assert context == 'image_alt=Payment card image_title=Choose card image_label=Primary payment method' + assert 'data:' not in context + assert 'private-payload' not in context + + +def test_child_image_context_limits_returned_images(): + parent = _make_element_node(501, 'a', {'href': '/gallery'}, x=10, y=10) + node = SimplifiedNode( + original_node=parent, + children=[_simplified_image(502 + index, {'src': f'/image-{index}.png'}) for index in range(4)], + ) + + context = DOMTreeSerializer._get_child_image_context(node) + + assert 'image-0.png' in context + assert 'image-1.png' in context + assert 'image-2.png' in context + assert 'image-3.png' not in context + + +def test_child_image_context_caps_traversal_even_when_images_have_no_context(): + parent = _make_element_node(601, 'a', {'href': '/gallery'}, x=10, y=10) + ignored_images = [ + _simplified_image(602 + index, {'src': 'data:image/png;base64,ignored'}) + for index in range(DOMTreeSerializer.MAX_CHILD_IMAGE_DESCENDANTS) + ] + node = SimplifiedNode( + original_node=parent, + children=[*ignored_images, _simplified_image(999, {'src': '/too-deep.png'})], + ) + + context = DOMTreeSerializer._get_child_image_context(node) + + assert context == '' From 1cddc7a794163b8b20ba205a181410a80ef83819 Mon Sep 17 00:00:00 2001 From: UniversePeak Date: Tue, 25 Aug 2026 12:56:19 +0800 Subject: [PATCH 35/69] fix(dom): make image traversal allocation bounded --- browser_use/dom/serializer/serializer.py | 13 ++++--- .../ci/test_image_only_dom_representation.py | 35 +++++++++++++++++++ 2 files changed, 44 insertions(+), 4 deletions(-) diff --git a/browser_use/dom/serializer/serializer.py b/browser_use/dom/serializer/serializer.py index 827a9a35e..8b86a9c75 100644 --- a/browser_use/dom/serializer/serializer.py +++ b/browser_use/dom/serializer/serializer.py @@ -933,14 +933,18 @@ class DOMTreeSerializer: path_without_query = clean_src.split('?', 1)[0].split('#', 1)[0].rstrip('/') return path_without_query.rsplit('/', 1)[-1] - pending = list(reversed(node.children)) + child_iterators = [iter(node.children)] visited_descendants = 0 while ( - pending + child_iterators and visited_descendants < DOMTreeSerializer.MAX_CHILD_IMAGE_DESCENDANTS and len(image_context) < DOMTreeSerializer.MAX_CHILD_IMAGE_CONTEXTS ): - current = pending.pop() + try: + current = next(child_iterators[-1]) + except StopIteration: + child_iterators.pop() + continue visited_descendants += 1 original_node = current.original_node if original_node.node_type == NodeType.ELEMENT_NODE and original_node.tag_name == 'img': @@ -963,7 +967,8 @@ class DOMTreeSerializer: if parts: image_context.append(' '.join(parts)) - pending.extend(reversed(current.children)) + if current.children: + child_iterators.append(iter(current.children)) return ' '.join(image_context) diff --git a/tests/ci/test_image_only_dom_representation.py b/tests/ci/test_image_only_dom_representation.py index 65a6c92f5..1ff1d3fe9 100644 --- a/tests/ci/test_image_only_dom_representation.py +++ b/tests/ci/test_image_only_dom_representation.py @@ -86,6 +86,29 @@ def _simplified_image(backend_node_id: int, attributes: dict[str, str]) -> Simpl return SimplifiedNode(original_node=image, children=[]) +class _NoEagerReverseList(list[SimplifiedNode]): + def __reversed__(self): + raise AssertionError('child lists must be traversed lazily') + + +def test_child_image_context_finds_images_below_non_interactive_wrappers(): + parent = _make_element_node(251, 'a', {'href': '/cards'}, x=10, y=10) + wrapper = _make_element_node(252, 'div', {}, x=12, y=12) + node = SimplifiedNode( + original_node=parent, + children=[ + SimplifiedNode( + original_node=wrapper, + children=[_simplified_image(253, {'src': '/nested-card.png'})], + ), + ], + ) + + context = DOMTreeSerializer._get_child_image_context(node) + + assert context == 'image_src=nested-card.png' + + def test_child_image_context_strips_query_and_fragment_without_leaking_query_only_sources(): parent = _make_element_node(301, 'a', {'href': '/cards'}, x=10, y=10) node = SimplifiedNode( @@ -156,3 +179,15 @@ def test_child_image_context_caps_traversal_even_when_images_have_no_context(): context = DOMTreeSerializer._get_child_image_context(node) assert context == '' + + +def test_child_image_context_does_not_copy_wide_child_lists_before_traversal(): + parent = _make_element_node(701, 'a', {'href': '/gallery'}, x=10, y=10) + node = SimplifiedNode(original_node=parent, children=[]) + node.children = _NoEagerReverseList( + _simplified_image(702 + index, {'src': 'data:image/png;base64,ignored'}) for index in range(200) + ) + + context = DOMTreeSerializer._get_child_image_context(node) + + assert context == '' From 6cd3fffca3a1cbaaa55f72badba6ec7670360fb7 Mon Sep 17 00:00:00 2001 From: ECD5A Date: Tue, 25 Aug 2026 12:15:54 +0500 Subject: [PATCH 36/69] fix(llm): ignore empty reasoning model patterns --- browser_use/llm/azure/chat.py | 3 +- browser_use/llm/base.py | 15 +++++++++ browser_use/llm/openai/chat.py | 4 +-- browser_use/llm/vercel/chat.py | 8 ++--- tests/ci/models/test_llm_openai.py | 51 ++++++++++++++++++++++++++++++ 5 files changed, 73 insertions(+), 8 deletions(-) diff --git a/browser_use/llm/azure/chat.py b/browser_use/llm/azure/chat.py index 21b14aabc..ca6f510da 100644 --- a/browser_use/llm/azure/chat.py +++ b/browser_use/llm/azure/chat.py @@ -9,6 +9,7 @@ from openai.types.responses import Response from openai.types.shared import ChatModel from pydantic import BaseModel +from browser_use.llm.base import is_reasoning_model from browser_use.llm.exceptions import ModelProviderError, ModelRateLimitError from browser_use.llm.messages import BaseMessage from browser_use.llm.openai.like import ChatOpenAILike @@ -179,7 +180,7 @@ class ChatAzureOpenAI(ChatOpenAILike): model_params['service_tier'] = self.service_tier # Handle reasoning models - if self.reasoning_models and any(str(m).lower() in str(self.model).lower() for m in self.reasoning_models): + if is_reasoning_model(self.model, self.reasoning_models): # For reasoning models, use reasoning parameter instead of reasoning_effort model_params['reasoning'] = {'effort': self.reasoning_effort} model_params.pop('temperature', None) diff --git a/browser_use/llm/base.py b/browser_use/llm/base.py index 7a591c62f..0cfd2a571 100644 --- a/browser_use/llm/base.py +++ b/browser_use/llm/base.py @@ -4,6 +4,7 @@ We have switched all of our code from langchain to openai.types.chat.chat_comple For easier transition we have """ +from collections.abc import Iterable from typing import Any, Protocol, TypeVar, overload, runtime_checkable from pydantic import BaseModel @@ -14,6 +15,20 @@ from browser_use.llm.views import ChatInvokeCompletion T = TypeVar('T', bound=BaseModel) +def is_reasoning_model(model: object, reasoning_models: Iterable[object] | None) -> bool: + """Return whether a model matches a non-empty reasoning-model pattern.""" + if not reasoning_models: + return False + + model_name = str(model).lower() + for pattern in reasoning_models: + pattern_name = str(pattern).lower() + if pattern_name.strip() and pattern_name in model_name: + return True + + return False + + @runtime_checkable class BaseChatModel(Protocol): _verified_api_keys: bool = False diff --git a/browser_use/llm/openai/chat.py b/browser_use/llm/openai/chat.py index 1074d3439..51226e5db 100644 --- a/browser_use/llm/openai/chat.py +++ b/browser_use/llm/openai/chat.py @@ -11,7 +11,7 @@ from openai.types.shared_params.reasoning_effort import ReasoningEffort from openai.types.shared_params.response_format_json_schema import JSONSchema, ResponseFormatJSONSchema from pydantic import BaseModel -from browser_use.llm.base import BaseChatModel +from browser_use.llm.base import BaseChatModel, is_reasoning_model from browser_use.llm.exceptions import ModelOutputTruncatedError, ModelProviderError, ModelRateLimitError from browser_use.llm.messages import BaseMessage from browser_use.llm.openai.serializer import OpenAIMessageSerializer @@ -186,7 +186,7 @@ class ChatOpenAI(BaseChatModel): if self.service_tier is not None: model_params['service_tier'] = self.service_tier - if self.reasoning_models and any(str(m).lower() in str(self.model).lower() for m in self.reasoning_models): + if is_reasoning_model(self.model, self.reasoning_models): model_params['reasoning_effort'] = self.reasoning_effort model_params.pop('temperature', None) model_params.pop('frequency_penalty', None) diff --git a/browser_use/llm/vercel/chat.py b/browser_use/llm/vercel/chat.py index 7a6e10f1e..c10842aed 100644 --- a/browser_use/llm/vercel/chat.py +++ b/browser_use/llm/vercel/chat.py @@ -13,7 +13,7 @@ from openai.types.shared_params.response_format_json_schema import ( ) from pydantic import BaseModel -from browser_use.llm.base import BaseChatModel +from browser_use.llm.base import BaseChatModel, is_reasoning_model from browser_use.llm.exceptions import ModelProviderError, ModelRateLimitError from browser_use.llm.messages import BaseMessage, ContentPartTextParam, SystemMessage from browser_use.llm.schema import SchemaOptimizer @@ -556,11 +556,9 @@ class ChatVercel(BaseChatModel): else: is_google_model = self.model.startswith('google/') is_anthropic_model = self.model.startswith('anthropic/') - is_reasoning_model = self.reasoning_models and any( - str(pattern).lower() in str(self.model).lower() for pattern in self.reasoning_models - ) + is_reasoning = is_reasoning_model(self.model, self.reasoning_models) - if is_google_model or is_anthropic_model or is_reasoning_model: + if is_google_model or is_anthropic_model or is_reasoning: modified_messages = [m.model_copy(deep=True) for m in messages] schema = SchemaOptimizer.create_gemini_optimized_schema(output_format) diff --git a/tests/ci/models/test_llm_openai.py b/tests/ci/models/test_llm_openai.py index 5fef442a1..d29e49645 100644 --- a/tests/ci/models/test_llm_openai.py +++ b/tests/ci/models/test_llm_openai.py @@ -1,9 +1,30 @@ """Test OpenAI model button click.""" +from types import SimpleNamespace + +import pytest + +from browser_use.llm.base import is_reasoning_model +from browser_use.llm.messages import UserMessage from browser_use.llm.openai.chat import ChatOpenAI from tests.ci.models.model_test_helper import run_model_button_click_test +@pytest.mark.parametrize( + ('model', 'reasoning_models', 'expected'), + [ + ('gpt-4.1', [''], False), + ('gpt-4.1', [' ', ''], False), + ('o3-mini', ['', 'o3'], True), + ('o3-mini', [' o3'], False), + ('gpt-4.1', None, False), + ], +) +def test_reasoning_model_matching_ignores_empty_patterns(model, reasoning_models, expected): + """Empty patterns must not match every model name.""" + assert is_reasoning_model(model, reasoning_models) is expected + + async def test_openai_gpt_4_1_mini(httpserver): """Test OpenAI gpt-4.1-mini can click a button.""" await run_model_button_click_test( @@ -13,3 +34,33 @@ async def test_openai_gpt_4_1_mini(httpserver): extra_kwargs={}, httpserver=httpserver, ) + + +@pytest.mark.parametrize('reasoning_models', [[], [''], [' ', '', '']]) +async def test_openai_empty_reasoning_model_patterns_preserve_sampling_parameters(monkeypatch, reasoning_models): + """Empty reasoning patterns must not classify a regular model as reasoning.""" + captured: dict[str, object] = {} + + class FakeCompletions: + async def create(self, **kwargs): + captured.update(kwargs) + return SimpleNamespace( + choices=[SimpleNamespace(message=SimpleNamespace(content='ok'), finish_reason='stop')], + usage=None, + ) + + fake_client = SimpleNamespace(chat=SimpleNamespace(completions=FakeCompletions())) + llm = ChatOpenAI( + model='gpt-4.1', + api_key='test-key', + temperature=0.7, + frequency_penalty=0.4, + reasoning_models=reasoning_models, + ) + monkeypatch.setattr(llm, 'get_client', lambda: fake_client) + + await llm.ainvoke([UserMessage(content='hello')]) + + assert captured['temperature'] == 0.7 + assert captured['frequency_penalty'] == 0.4 + assert 'reasoning_effort' not in captured From c1b6df89589a2b8c66ce8de63c2e9b790421713b Mon Sep 17 00:00:00 2001 From: MagMueller Date: Tue, 25 Aug 2026 15:58:51 -0700 Subject: [PATCH 37/69] docs: highlight rerunnable Cloud scripts --- README.md | 1 + 1 file changed, 1 insertion(+) diff --git a/README.md b/README.md index 83e3d8c5e..8535f304b 100644 --- a/README.md +++ b/README.md @@ -142,6 +142,7 @@ Browser Use is also **#1 on the [Odysseys leaderboard](https://odysseysbench.com - Best stealth with proxy rotation and captcha solving - 1000+ integrations (Gmail, Slack, Notion, and more) - Persistent filesystem and memory +- Rerunnable scripts fetch live data, even when sites change ([guide](https://docs.browser-use.com/cloud/agent/scripts)) ```sh curl -X POST https://api.browser-use.com/api/v4/runs \ From 9a7e67fb3ece0038c1a02401a985f83f2bab4324 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Tue, 25 Aug 2026 22:33:11 -0700 Subject: [PATCH 38/69] docs(cloud): update tools guide for CLI 3.0 --- .../references/guides/tools-integration.md | 68 +++++++++---------- 1 file changed, 34 insertions(+), 34 deletions(-) diff --git a/skills/cloud/references/guides/tools-integration.md b/skills/cloud/references/guides/tools-integration.md index f44eb9301..a0dd9aa1b 100644 --- a/skills/cloud/references/guides/tools-integration.md +++ b/skills/cloud/references/guides/tools-integration.md @@ -30,7 +30,7 @@ Your agent already has tools (search, code execution, file I/O, etc.) and its ow | Your agent type | Best approach | Control level | |----------------|---------------|--------------| -| CLI coding agent in sandbox | [CLI commands](#shell-command-agents-cli) | Per-command | +| CLI coding agent in sandbox | [CLI 3.0 Python](#shell-command-agents-cli) | Per-call | | TypeScript/JS | [CDP + Playwright](#typescriptjs-cdp--playwright) | Playwright API | | MCP client (Claude Desktop, Cursor) | [Local MCP server](#mcp-native-agents) | MCP tools | | Existing Playwright/Puppeteer/Selenium | [CDP WebSocket (stealth)](#existing-playwrightpuppeteerselenium) | Your existing API | @@ -42,54 +42,54 @@ Your agent already has tools (search, code execution, file I/O, etc.) and its ow **For:** Claude Code, Codex, OpenCode, Cline, Windsurf, Cursor background agents, Hermes, OpenClaw โ€” any coding agent running in a VM/container with terminal access. -**Setup:** Install the CLI and load the browser-use SKILL.md into the agent's context. The agent calls browser commands as shell tool invocations. +**Setup:** Install the CLI and load the browser-use SKILL.md into the agent's context. CLI 3.0 runs Python from stdin. Browser helpers are already imported, and the browser stays alive between calls. ```bash uv pip install 'browser-use[cli]' ``` -**Core workflow** โ€” the agent calls these commands one at a time, reading output between each: +For Browser Use Cloud, authenticate once and start a named remote browser: ```bash -# 1. Navigate -browser-use open https://example.com +browser-use auth login -# 2. Observe โ€” ALWAYS run state first to get element indices -browser-use state -# Output: URL, title, list of clickable elements with indices -# e.g. [0] -# [1] -# [2] About +browser-use <<'PY' +start_remote_daemon("agent-1") +PY +``` -# 3. Interact โ€” use indices from state -browser-use input 0 "search query" # Type into element 0 -browser-use click 1 # Click element 1 +Use the same `BU_NAME` for every later call so the agent stays on that cloud browser: -# 4. Verify โ€” re-run state to see result -browser-use state +```bash +# 1. Navigate and observe +BU_NAME=agent-1 browser-use <<'PY' +new_tab("https://example.com") +wait_for_load() +print(page_info()) +PY -# 5. Extract data -browser-use get text 3 # Get element text -browser-use get html --selector "h1" # Get scoped HTML -browser-use eval "document.title" # Execute JavaScript -browser-use screenshot result.png # Capture visual state +# 2. Interact, then verify the result +BU_NAME=agent-1 browser-use <<'PY' +fill_input('input[name="q"]', "search query") +press_key("ENTER") +wait_for_load() +print(js("document.title")) +print(capture_screenshot()) +PY -# 6. Wait for dynamic content -browser-use wait selector ".results" # Wait for element -browser-use wait text "Success" # Wait for text - -# 7. Cleanup -browser-use close +# 3. Stop the cloud browser when the job is done +browser-use <<'PY' +stop_remote_daemon("agent-1") +PY ``` **Key details:** -- Background daemon keeps browser alive between commands (~50ms latency per call) -- Agent's reasoning loop decides which command to call next -- `state` output is the agent's "eyes" โ€” it reads element indices and decides what to click -- Commands can be chained with `&&` when intermediate output isn't needed -- `--json` flag for machine-readable output -- `--headed` for visible browser (debugging) -- `--profile "Default"` for authenticated browsing with saved Chrome logins +- CLI 3.0 removed the old `open`, `state`, `click`, `eval`, `--json`, `--headed`, and `--profile` command surface. +- The agent writes Python with helpers such as `new_tab`, `page_info`, `fill_input`, `click_at_xy`, `js`, and `cdp`. +- The background daemon keeps the browser alive between calls. Printed Python values are the tool output. +- `BU_NAME` selects the named cloud browser. Without it, the CLI uses the default local browser. +- The first navigation is `new_tab(url)`. Use `goto_url(url)` only after a real tab exists. +- Remote browsers keep billing until they stop or time out. Always call `stop_remote_daemon(name)` after the job. --- From 280cfd91415c7661d7da67658af67ff564a85f69 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Wed, 26 Aug 2026 00:00:39 -0700 Subject: [PATCH 39/69] docs(cloud): use searchable page in CLI example --- skills/cloud/references/guides/tools-integration.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/skills/cloud/references/guides/tools-integration.md b/skills/cloud/references/guides/tools-integration.md index a0dd9aa1b..04ec9019e 100644 --- a/skills/cloud/references/guides/tools-integration.md +++ b/skills/cloud/references/guides/tools-integration.md @@ -63,7 +63,7 @@ Use the same `BU_NAME` for every later call so the agent stays on that cloud bro ```bash # 1. Navigate and observe BU_NAME=agent-1 browser-use <<'PY' -new_tab("https://example.com") +new_tab("https://html.duckduckgo.com/html/") wait_for_load() print(page_info()) PY From d7b36c7dc581c247c4bdee04cf3064052ba8d771 Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Wed, 26 Aug 2026 11:27:38 -0700 Subject: [PATCH 40/69] Fix lint and skill sync checks --- browser_use/skills/browser-use/SKILL.md | 7 +++++++ examples/custom-functions/cua.py | 2 ++ skills/browser-use/SKILL.md | 7 +++++++ 3 files changed, 16 insertions(+) diff --git a/browser_use/skills/browser-use/SKILL.md b/browser_use/skills/browser-use/SKILL.md index 356e12fe9..4b359b2bf 100644 --- a/browser_use/skills/browser-use/SKILL.md +++ b/browser_use/skills/browser-use/SKILL.md @@ -184,6 +184,13 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro - Coordinate clicks default. CDP mouse events pass through iframes/shadow/cross-origin at the compositor level. - Keep the connection model simple: use the default daemon, `BU_NAME`, `BU_CDP_URL`, `BU_CDP_WS`, or `start_remote_daemon(...)`. +- Trusted orchestrators can set `BH_OPEN_LIVE_URL=0` while provisioning a Cloud + daemon to keep its interactive live-view URL from being printed or opened. + The URL is still created and returned by `start_remote_daemon()`; callers must + avoid logging or serializing that returned field. +- Trusted orchestrators that already provisioned an exact named daemon can set + `BH_REQUIRE_EXISTING_DAEMON=1`. Each CLI call then health-checks and reuses + that daemon or fails closed; it never auto-starts or discovers another Chrome. - Core helpers stay short. Put task-specific helper additions in `$BH_AGENT_WORKSPACE/agent_helpers.py`. ## Gotchas diff --git a/examples/custom-functions/cua.py b/examples/custom-functions/cua.py index 67038feca..e645ff658 100644 --- a/examples/custom-functions/cua.py +++ b/examples/custom-functions/cua.py @@ -261,6 +261,8 @@ async def openai_cua_fallback(params: OpenAICUAAction, browser_session: BrowserS raise Exception('No computer calls found in CUA response') action = computer_call.action + if action is None: + raise Exception('No action found in CUA computer call') print(f'๐ŸŽฌ Executing CUA action: {action.type} - {action}') action_result = await handle_model_action(browser_session, action) diff --git a/skills/browser-use/SKILL.md b/skills/browser-use/SKILL.md index 356e12fe9..4b359b2bf 100644 --- a/skills/browser-use/SKILL.md +++ b/skills/browser-use/SKILL.md @@ -184,6 +184,13 @@ If you get stuck on a browser mechanic, check https://github.com/browser-use/bro - Coordinate clicks default. CDP mouse events pass through iframes/shadow/cross-origin at the compositor level. - Keep the connection model simple: use the default daemon, `BU_NAME`, `BU_CDP_URL`, `BU_CDP_WS`, or `start_remote_daemon(...)`. +- Trusted orchestrators can set `BH_OPEN_LIVE_URL=0` while provisioning a Cloud + daemon to keep its interactive live-view URL from being printed or opened. + The URL is still created and returned by `start_remote_daemon()`; callers must + avoid logging or serializing that returned field. +- Trusted orchestrators that already provisioned an exact named daemon can set + `BH_REQUIRE_EXISTING_DAEMON=1`. Each CLI call then health-checks and reuses + that daemon or fails closed; it never auto-starts or discovers another Chrome. - Core helpers stay short. Put task-specific helper additions in `$BH_AGENT_WORKSPACE/agent_helpers.py`. ## Gotchas From 2f573908ca3ec171543a73a5e66caf955a7854e2 Mon Sep 17 00:00:00 2001 From: Matthew Elliot Date: Thu, 27 Aug 2026 13:06:53 +0200 Subject: [PATCH 41/69] fix(browser): clear cookies via Storage domain, not Network BrowserSession.clear_cookies() sends Network.clearBrowserCookies on the root CDP client, which always fails: RuntimeError: {'code': -32601, 'message': "'Network.clearBrowserCookies' wasn't found"} Network is a per-target domain and is not dispatchable without a session attachment; the root client has none. Storage is browser-level, and the cookies() getter two lines above already uses Storage.getCookies() there. Measured against a live CDP endpoint: Network.clearBrowserCookies() REJECTED -32601 Storage.clearCookies() OK Storage.clearCookies(session_id=...) OK Network.clearBrowserCookies(session_id=...) OK The private _cdp_clear_cookies() already uses Storage.clearCookies with a session_id, though its docstring still refers to Network.clearBrowserCookies - this looks like the public method was missed when that one was fixed. --- browser_use/browser/session.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/browser_use/browser/session.py b/browser_use/browser/session.py index baef63f48..5bbd5e4c0 100644 --- a/browser_use/browser/session.py +++ b/browser_use/browser/session.py @@ -1422,7 +1422,7 @@ class BrowserSession(BaseModel): async def clear_cookies(self) -> None: """Clear all cookies.""" - await self.cdp_client.send.Network.clearBrowserCookies() + await self.cdp_client.send.Storage.clearCookies() async def export_storage_state(self, output_path: str | Path | None = None) -> dict[str, Any]: """Export all browser cookies and storage to storage_state format. From c7214fa867edea41b8fc5fde268861975853e220 Mon Sep 17 00:00:00 2001 From: DipsyHou Date: Thu, 27 Aug 2026 21:55:20 +0800 Subject: [PATCH 42/69] fix: Update DeepSeek default model to deepseek-v4-flash after chat/reasoner retirement. --- browser_use/llm/deepseek/chat.py | 31 ++++++++++++++++--------- examples/models/deepseek-chat.py | 2 +- skills/open-source/references/models.md | 2 +- 3 files changed, 22 insertions(+), 13 deletions(-) diff --git a/browser_use/llm/deepseek/chat.py b/browser_use/llm/deepseek/chat.py index 26f656c8b..c9db7cc06 100644 --- a/browser_use/llm/deepseek/chat.py +++ b/browser_use/llm/deepseek/chat.py @@ -29,13 +29,14 @@ T = TypeVar('T', bound=BaseModel) class ChatDeepSeek(BaseChatModel): """DeepSeek /chat/completions wrapper (OpenAI-compatible).""" - model: str = 'deepseek-chat' + model: str = 'deepseek-v4-flash' # Generation parameters max_tokens: int | None = None temperature: float | None = None top_p: float | None = None seed: int | None = None + thinking: bool = False # Connection parameters api_key: str | None = None @@ -59,6 +60,23 @@ class ChatDeepSeek(BaseChatModel): def name(self) -> str: return self.model + def _request_kwargs(self) -> dict[str, Any]: + common: dict[str, Any] = {} + + if self.temperature is not None: + common['temperature'] = self.temperature + if self.max_tokens is not None: + common['max_tokens'] = self.max_tokens + if self.top_p is not None: + common['top_p'] = self.top_p + if self.seed is not None: + common['seed'] = self.seed + + common['extra_body'] = { + 'thinking': {'type': 'enabled' if self.thinking else 'disabled'}, + } + return common + @overload async def ainvoke( self, @@ -96,16 +114,7 @@ class ChatDeepSeek(BaseChatModel): """ client = self._client() ds_messages = DeepSeekMessageSerializer.serialize_messages(messages) - common: dict[str, Any] = {} - - if self.temperature is not None: - common['temperature'] = self.temperature - if self.max_tokens is not None: - common['max_tokens'] = self.max_tokens - if self.top_p is not None: - common['top_p'] = self.top_p - if self.seed is not None: - common['seed'] = self.seed + common = self._request_kwargs() # Beta conversation prefix continuation (see official documentation) if self.base_url and str(self.base_url).endswith('/beta'): diff --git a/examples/models/deepseek-chat.py b/examples/models/deepseek-chat.py index cf05ceace..1f1a9aaff 100644 --- a/examples/models/deepseek-chat.py +++ b/examples/models/deepseek-chat.py @@ -20,7 +20,7 @@ if deepseek_api_key is None: async def main(): llm = ChatDeepSeek( base_url='https://api.deepseek.com/v1', - model='deepseek-chat', + model='deepseek-v4-flash', api_key=deepseek_api_key, ) diff --git a/skills/open-source/references/models.md b/skills/open-source/references/models.md index 30438aec7..497d6fbbb 100644 --- a/skills/open-source/references/models.md +++ b/skills/open-source/references/models.md @@ -150,7 +150,7 @@ Supports profiles, IAM roles, SSO via standard AWS credential chain. Install wit ```python from browser_use import Agent, ChatDeepSeek -llm = ChatDeepSeek(model="deepseek-chat") +llm = ChatDeepSeek(model="deepseek-v4-flash") ``` **Env:** `DEEPSEEK_API_KEY` | [Available models](https://api-docs.deepseek.com/quick_start/pricing) From 0452558036f003f9a6f0f83dab17b1122b9b72d5 Mon Sep 17 00:00:00 2001 From: DipsyHou Date: Thu, 27 Aug 2026 22:29:20 +0800 Subject: [PATCH 43/69] fix(llm): gate DeepSeek thinking and preserve positional constructor --- browser_use/llm/deepseek/chat.py | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/browser_use/llm/deepseek/chat.py b/browser_use/llm/deepseek/chat.py index c9db7cc06..ba583b115 100644 --- a/browser_use/llm/deepseek/chat.py +++ b/browser_use/llm/deepseek/chat.py @@ -36,13 +36,14 @@ class ChatDeepSeek(BaseChatModel): temperature: float | None = None top_p: float | None = None seed: int | None = None - thinking: bool = False # Connection parameters api_key: str | None = None base_url: str | httpx.URL | None = 'https://api.deepseek.com/v1' timeout: float | httpx.Timeout | None = None client_params: dict[str, Any] | None = None + + thinking: bool = False @property def provider(self) -> str: @@ -60,6 +61,10 @@ class ChatDeepSeek(BaseChatModel): def name(self) -> str: return self.model + def _supports_thinking(self) -> bool: + + return 'deepseek-v4' in self.model.lower() + def _request_kwargs(self) -> dict[str, Any]: common: dict[str, Any] = {} @@ -72,9 +77,10 @@ class ChatDeepSeek(BaseChatModel): if self.seed is not None: common['seed'] = self.seed - common['extra_body'] = { - 'thinking': {'type': 'enabled' if self.thinking else 'disabled'}, - } + if self._supports_thinking(): + common['extra_body'] = { + 'thinking': {'type': 'enabled' if self.thinking else 'disabled'}, + } return common @overload From 2b66d1f1c0151d95392a3f046437a22db68f58dd Mon Sep 17 00:00:00 2001 From: DipsyHou Date: Thu, 27 Aug 2026 22:37:42 +0800 Subject: [PATCH 44/69] style(llm): apply ruff format --- browser_use/llm/deepseek/chat.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/browser_use/llm/deepseek/chat.py b/browser_use/llm/deepseek/chat.py index ba583b115..77c050e7e 100644 --- a/browser_use/llm/deepseek/chat.py +++ b/browser_use/llm/deepseek/chat.py @@ -42,7 +42,7 @@ class ChatDeepSeek(BaseChatModel): base_url: str | httpx.URL | None = 'https://api.deepseek.com/v1' timeout: float | httpx.Timeout | None = None client_params: dict[str, Any] | None = None - + thinking: bool = False @property @@ -62,7 +62,6 @@ class ChatDeepSeek(BaseChatModel): return self.model def _supports_thinking(self) -> bool: - return 'deepseek-v4' in self.model.lower() def _request_kwargs(self) -> dict[str, Any]: From 1b2f2bb4f5a4939a4bd882dc5e1db77afad416f9 Mon Sep 17 00:00:00 2001 From: Burak Keskin Date: Thu, 27 Aug 2026 13:37:01 +0300 Subject: [PATCH 45/69] fix(llm): stop putting format/stream inside Ollama options Ollama treats those as top-level chat() parameters. Forwarding them in ollama_options broke structured JSON from vision models, which is the failure in #5017. --- browser_use/llm/ollama/chat.py | 71 ++++++++++++++++++++++++------ tests/ci/models/test_llm_ollama.py | 57 ++++++++++++++++++++++++ 2 files changed, 114 insertions(+), 14 deletions(-) create mode 100644 tests/ci/models/test_llm_ollama.py diff --git a/browser_use/llm/ollama/chat.py b/browser_use/llm/ollama/chat.py index 0d592d448..bbec7030a 100644 --- a/browser_use/llm/ollama/chat.py +++ b/browser_use/llm/ollama/chat.py @@ -1,3 +1,4 @@ +import logging from collections.abc import Mapping from dataclasses import dataclass from typing import Any, TypeVar, overload @@ -5,7 +6,7 @@ from typing import Any, TypeVar, overload import httpx from ollama import AsyncClient as OllamaAsyncClient from ollama import Options -from pydantic import BaseModel +from pydantic import BaseModel, ValidationError from browser_use.llm.base import BaseChatModel from browser_use.llm.exceptions import ModelProviderError @@ -14,6 +15,20 @@ from browser_use.llm.ollama.serializer import OllamaMessageSerializer from browser_use.llm.views import ChatInvokeCompletion T = TypeVar('T', bound=BaseModel) +logger = logging.getLogger(__name__) + +# These belong on AsyncClient.chat(), not in the model `options` dict. +_TOP_LEVEL_CHAT_KEYS = frozenset({'format', 'stream'}) + + +def _unwrap_json_content(content: str) -> str: + """Strip markdown code fences that Ollama vision models often wrap around JSON.""" + text = content.strip() + if text.startswith('```json') and text.endswith('```'): + return text[7:-3].strip() + if text.startswith('```') and text.endswith('```'): + return text[3:-3].strip() + return text @dataclass @@ -57,6 +72,27 @@ class ChatOllama(BaseChatModel): def name(self) -> str: return self.model + def _chat_options(self) -> Mapping[str, Any] | Options | None: + """Return ollama_options with top-level chat keys removed. + + `format` and `stream` are parameters of `AsyncClient.chat()`, not model + runtime options. Putting them in `ollama_options` is ignored or fights + the structured-output schema we already pass as `format=`. + """ + options = self.ollama_options + if not options or not isinstance(options, Mapping): + return options + + dropped = sorted(key for key in options if key in _TOP_LEVEL_CHAT_KEYS) + if not dropped: + return options + + logger.warning( + 'Ignoring %s in ollama_options; those are top-level chat() parameters, not model options', + ', '.join(dropped), + ) + return {key: value for key, value in options.items() if key not in _TOP_LEVEL_CHAT_KEYS} + @overload async def ainvoke( self, messages: list[BaseMessage], output_format: None = None, **kwargs: Any @@ -71,29 +107,36 @@ class ChatOllama(BaseChatModel): ollama_messages = OllamaMessageSerializer.serialize_messages(messages) try: + options = self._chat_options() if output_format is None: response = await self.get_client().chat( model=self.model, messages=ollama_messages, - options=self.ollama_options, + options=options, ) return ChatInvokeCompletion(completion=response.message.content or '', usage=None) - else: - schema = output_format.model_json_schema() - response = await self.get_client().chat( - model=self.model, - messages=ollama_messages, - format=schema, - options=self.ollama_options, - ) + schema = output_format.model_json_schema() + response = await self.get_client().chat( + model=self.model, + messages=ollama_messages, + format=schema, + options=options, + ) - completion = response.message.content or '' - if output_format is not None: - completion = output_format.model_validate_json(completion) + completion = _unwrap_json_content(response.message.content or '') + try: + parsed = output_format.model_validate_json(completion) + except ValidationError as e: + raise ModelProviderError( + message=f'Ollama returned invalid JSON for structured output: {e}', + model=self.name, + ) from e - return ChatInvokeCompletion(completion=completion, usage=None) + return ChatInvokeCompletion(completion=parsed, usage=None) + except ModelProviderError: + raise except Exception as e: raise ModelProviderError(message=str(e), model=self.name) from e diff --git a/tests/ci/models/test_llm_ollama.py b/tests/ci/models/test_llm_ollama.py new file mode 100644 index 000000000..d82c3e045 --- /dev/null +++ b/tests/ci/models/test_llm_ollama.py @@ -0,0 +1,57 @@ +"""Tests for ChatOllama option handling and structured-output parsing.""" + +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest +from pydantic import BaseModel + +from browser_use.llm.exceptions import ModelProviderError +from browser_use.llm.messages import UserMessage +from browser_use.llm.ollama.chat import ChatOllama + + +class Answer(BaseModel): + answer: str + + +def _client_returning(content: str) -> MagicMock: + client = MagicMock() + client.chat = AsyncMock(return_value=MagicMock(message=MagicMock(content=content))) + return client + + +async def test_drops_format_and_stream_from_ollama_options(): + """format/stream belong on chat(), not inside options (#5017).""" + client = _client_returning('{"answer": "ok"}') + llm = ChatOllama( + model='test-model', + ollama_options={'think': False, 'format': 'json', 'stream': False, 'num_ctx': 2048}, + ) + + with patch.object(ChatOllama, 'get_client', return_value=client): + result = await llm.ainvoke([UserMessage(content='hi')], output_format=Answer) + + assert result.completion.answer == 'ok' + kwargs = client.chat.await_args.kwargs + assert kwargs['options'] == {'think': False, 'num_ctx': 2048} + assert kwargs['format'] == Answer.model_json_schema() + + +async def test_parses_json_wrapped_in_markdown_fences(): + client = _client_returning('```json\n{"answer": "ok"}\n```') + llm = ChatOllama(model='test-model') + + with patch.object(ChatOllama, 'get_client', return_value=client): + result = await llm.ainvoke([UserMessage(content='hi')], output_format=Answer) + + assert result.completion.answer == 'ok' + + +async def test_truncated_json_raises_model_provider_error(): + client = _client_returning('{\n') + llm = ChatOllama(model='test-model') + + with patch.object(ChatOllama, 'get_client', return_value=client), pytest.raises(ModelProviderError) as exc_info: + await llm.ainvoke([UserMessage(content='hi')], output_format=Answer) + + assert 'Invalid JSON' in exc_info.value.message or 'invalid JSON' in exc_info.value.message.lower() From ed5f467957893427a5aa6944bd77e336146bf452 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 27 Aug 2026 09:54:49 -0700 Subject: [PATCH 46/69] fix(llm): handle Ollama chat params and JSON fences --- browser_use/llm/ollama/chat.py | 46 +++++++++++++++++------------- tests/ci/models/test_llm_ollama.py | 26 +++++++++++++---- 2 files changed, 46 insertions(+), 26 deletions(-) diff --git a/browser_use/llm/ollama/chat.py b/browser_use/llm/ollama/chat.py index bbec7030a..c10d25618 100644 --- a/browser_use/llm/ollama/chat.py +++ b/browser_use/llm/ollama/chat.py @@ -1,4 +1,5 @@ import logging +import re from collections.abc import Mapping from dataclasses import dataclass from typing import Any, TypeVar, overload @@ -18,16 +19,17 @@ T = TypeVar('T', bound=BaseModel) logger = logging.getLogger(__name__) # These belong on AsyncClient.chat(), not in the model `options` dict. -_TOP_LEVEL_CHAT_KEYS = frozenset({'format', 'stream'}) +_PASSTHROUGH_CHAT_KEYS = frozenset({'think', 'logprobs', 'top_logprobs', 'keep_alive'}) +_IGNORED_CHAT_KEYS = frozenset({'format', 'stream'}) +_JSON_FENCE_RE = re.compile(r'\A```[ \t]*(?:json)?[ \t]*\r?\n(?P.*?)\r?\n?```[ \t]*\Z', re.IGNORECASE | re.DOTALL) def _unwrap_json_content(content: str) -> str: """Strip markdown code fences that Ollama vision models often wrap around JSON.""" text = content.strip() - if text.startswith('```json') and text.endswith('```'): - return text[7:-3].strip() - if text.startswith('```') and text.endswith('```'): - return text[3:-3].strip() + match = _JSON_FENCE_RE.fullmatch(text) + if match: + return match.group('body').strip() return text @@ -72,26 +74,28 @@ class ChatOllama(BaseChatModel): def name(self) -> str: return self.model - def _chat_options(self) -> Mapping[str, Any] | Options | None: - """Return ollama_options with top-level chat keys removed. + def _split_chat_options(self) -> tuple[Mapping[str, Any] | Options | None, dict[str, Any]]: + """Split model options from supported top-level ``chat()`` parameters. - `format` and `stream` are parameters of `AsyncClient.chat()`, not model - runtime options. Putting them in `ollama_options` is ignored or fights - the structured-output schema we already pass as `format=`. + ``format`` and ``stream`` cannot be honored here because this wrapper owns + the structured-output schema and requires a non-streaming response. """ options = self.ollama_options if not options or not isinstance(options, Mapping): - return options + return options, {} - dropped = sorted(key for key in options if key in _TOP_LEVEL_CHAT_KEYS) - if not dropped: - return options + top_level = {key: options[key] for key in _PASSTHROUGH_CHAT_KEYS if key in options} + ignored = sorted(key for key in options if key in _IGNORED_CHAT_KEYS) - logger.warning( - 'Ignoring %s in ollama_options; those are top-level chat() parameters, not model options', - ', '.join(dropped), - ) - return {key: value for key, value in options.items() if key not in _TOP_LEVEL_CHAT_KEYS} + if ignored: + logger.warning( + 'Ignoring %s in ollama_options; ChatOllama controls structured output and streaming', + ', '.join(ignored), + ) + + extracted = _PASSTHROUGH_CHAT_KEYS | _IGNORED_CHAT_KEYS + model_options = {key: value for key, value in options.items() if key not in extracted} + return model_options, top_level @overload async def ainvoke( @@ -107,12 +111,13 @@ class ChatOllama(BaseChatModel): ollama_messages = OllamaMessageSerializer.serialize_messages(messages) try: - options = self._chat_options() + options, top_level = self._split_chat_options() if output_format is None: response = await self.get_client().chat( model=self.model, messages=ollama_messages, options=options, + **top_level, ) return ChatInvokeCompletion(completion=response.message.content or '', usage=None) @@ -123,6 +128,7 @@ class ChatOllama(BaseChatModel): messages=ollama_messages, format=schema, options=options, + **top_level, ) completion = _unwrap_json_content(response.message.content or '') diff --git a/tests/ci/models/test_llm_ollama.py b/tests/ci/models/test_llm_ollama.py index d82c3e045..6c96b6f9e 100644 --- a/tests/ci/models/test_llm_ollama.py +++ b/tests/ci/models/test_llm_ollama.py @@ -20,12 +20,20 @@ def _client_returning(content: str) -> MagicMock: return client -async def test_drops_format_and_stream_from_ollama_options(): - """format/stream belong on chat(), not inside options (#5017).""" +async def test_splits_top_level_chat_parameters_from_ollama_options(): + """Top-level chat parameters must not be sent inside model options (#5017).""" client = _client_returning('{"answer": "ok"}') llm = ChatOllama( model='test-model', - ollama_options={'think': False, 'format': 'json', 'stream': False, 'num_ctx': 2048}, + ollama_options={ + 'think': False, + 'logprobs': True, + 'top_logprobs': 3, + 'keep_alive': '10m', + 'format': 'json', + 'stream': False, + 'num_ctx': 2048, + }, ) with patch.object(ChatOllama, 'get_client', return_value=client): @@ -33,12 +41,18 @@ async def test_drops_format_and_stream_from_ollama_options(): assert result.completion.answer == 'ok' kwargs = client.chat.await_args.kwargs - assert kwargs['options'] == {'think': False, 'num_ctx': 2048} + assert kwargs['options'] == {'num_ctx': 2048} + assert kwargs['think'] is False + assert kwargs['logprobs'] is True + assert kwargs['top_logprobs'] == 3 + assert kwargs['keep_alive'] == '10m' assert kwargs['format'] == Answer.model_json_schema() + assert kwargs.get('stream') is None -async def test_parses_json_wrapped_in_markdown_fences(): - client = _client_returning('```json\n{"answer": "ok"}\n```') +@pytest.mark.parametrize('fence', ['```json', '```JSON', '``` json', '```']) +async def test_parses_json_wrapped_in_markdown_fences(fence: str): + client = _client_returning(f'{fence}\n{{"answer": "ok"}}\n```') llm = ChatOllama(model='test-model') with patch.object(ChatOllama, 'get_client', return_value=client): From c04983c2b7c149e4ca61e48c3df628b980af842f Mon Sep 17 00:00:00 2001 From: green3sf <222944370+green3sf@users.noreply.github.com> Date: Mon, 24 Aug 2026 02:16:15 +0800 Subject: [PATCH 47/69] fix(deps): update vulnerable dependency pins --- pyproject.toml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index ecd445695..0f083c301 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -14,7 +14,7 @@ dependencies = [ "aiohttp==3.14.3", "anyio==4.12.1", "bubus==1.5.6", - "click==8.3.1", + "click==8.3.3", "InquirerPy==0.3.4", "rich==14.3.3", "google-api-core==2.29.0", @@ -37,7 +37,7 @@ dependencies = [ "google-auth==2.48.0", "google-auth-oauthlib==1.2.4", "mcp==1.28.1", - "pypdf==6.14.2", + "pypdf==6.15.0", "reportlab==4.4.9", "cdp-use==1.4.5", "pyotp==2.9.0", From 25b88477a7c748293f5641453bc2bf23827fd562 Mon Sep 17 00:00:00 2001 From: lei_lei <96427312+leilei3167@users.noreply.github.com> Date: Tue, 25 Aug 2026 01:48:20 +0000 Subject: [PATCH 48/69] fix(filesystem): render markdown emphasis in PDF export Escape first to keep the ReportLab parse fix, then convert bold/italic/code/bullets to RML so write_file PDFs match the documented markdown contract. --- browser_use/filesystem/file_system.py | 90 ++++++++++++++++------ tests/ci/infrastructure/test_filesystem.py | 77 ++++++++++++++++++ 2 files changed, 144 insertions(+), 23 deletions(-) diff --git a/browser_use/filesystem/file_system.py b/browser_use/filesystem/file_system.py index 292636c83..f008eee19 100644 --- a/browser_use/filesystem/file_system.py +++ b/browser_use/filesystem/file_system.py @@ -74,6 +74,43 @@ def _build_filename_error_message(file_name: str, supported_extensions: list[str ) +def _split_heading(line: str) -> tuple[str, int | None]: + """Split a markdown ATX heading into (text, level). + + Only ``# `` / ``## `` / ``### `` (note the required space) are headings. + Anything else, including ``#hashtag``, is returned unchanged with level None. + """ + if line.startswith('### '): + return line[4:], 3 + if line.startswith('## '): + return line[3:], 2 + if line.startswith('# '): + return line[2:], 1 + return line, None + + +_BULLET_RE = re.compile(r'^(\s*)[-*]\s+(.*)$') + + +def _markdown_inline_to_rml(text: str) -> str: + """Escape plain text, then convert a markdown subset to ReportLab markup. + + Order matters: after html.escape there are no user-supplied ``<`` left, so + injected ```` / ```` / ```` tags are unambiguous. + + Underscore emphasis is intentionally unsupported so ``snake_case`` identifiers + survive unchanged. + """ + text = html.escape(text) + # Bold before italic so ``**`` is not treated as two italic markers. + # Content cannot contain ``*`` โ€” otherwise globs like ``*.txt and **/*.py`` pair across tokens. + text = re.sub(r'\*\*([^\s*](?:[^*]*[^\s*])?)\*\*', r'\1', text) + # Non-space boundaries keep ``2 * 3 * 4`` literal; leading ``* `` is a bullet, not italic + text = re.sub(r'(?\1', text) + text = re.sub(r'`([^`]+)`', r'\1', text) + return text + + DEFAULT_FILE_SYSTEM_PATH = 'browseruse_agent_data' @@ -257,25 +294,36 @@ class PdfFile(BaseFile): doc = SimpleDocTemplate(str(file_path), pagesize=letter) styles = getSampleStyleSheet() story = [] + heading_styles = {1: styles['Title'], 2: styles['Heading1'], 3: styles['Heading2']} - # Plain text plus markdown headers only, to avoid an AGPL markdown-to-PDF dependency - content_lines = self.content.split('\n') + # Escape first, then markdown โ†’ RML. Avoids an AGPL markdown-to-PDF dependency. + in_fence = False + for line in self.content.split('\n'): + stripped = line.strip() + if stripped.startswith('```'): + in_fence = not in_fence + continue - for line in content_lines: - if line.strip(): - # Handle basic markdown headers - if line.startswith('# '): - text, style = line[2:], styles['Title'] - elif line.startswith('## '): - text, style = line[3:], styles['Heading1'] - elif line.startswith('### '): - text, style = line[4:], styles['Heading2'] - else: - text, style = line, styles['Normal'] - # Paragraph parses its input as ReportLab markup, but our content is plain text - story.append(Paragraph(html.escape(text), style)) - else: + if not stripped: story.append(Spacer(1, 6)) + continue + + if in_fence: + # Fenced blocks are literal: no emphasis / inline-code conversion + story.append(Paragraph(html.escape(line), styles['Code'])) + continue + + text, heading_level = _split_heading(line) + if heading_level is not None: + story.append(Paragraph(_markdown_inline_to_rml(text), heading_styles[heading_level])) + continue + + bullet = _BULLET_RE.match(line) + if bullet: + story.append(Paragraph(f'• {_markdown_inline_to_rml(bullet.group(2))}', styles['Normal'])) + continue + + story.append(Paragraph(_markdown_inline_to_rml(line), styles['Normal'])) doc.build(story) except Exception as e: @@ -305,13 +353,9 @@ class DocxFile(BaseFile): for line in content_lines: if line.strip(): - # Handle basic markdown headers - if line.startswith('# '): - doc.add_heading(line[2:], level=1) - elif line.startswith('## '): - doc.add_heading(line[3:], level=2) - elif line.startswith('### '): - doc.add_heading(line[4:], level=3) + text, heading_level = _split_heading(line) + if heading_level is not None: + doc.add_heading(text, level=heading_level) else: doc.add_paragraph(line) else: diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index 15bc60bed..f5cd81937 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -17,6 +17,7 @@ from browser_use.filesystem.file_system import ( JsonlFile, MarkdownFile, TxtFile, + _split_heading, ) @@ -43,6 +44,15 @@ class TestBaseFile: assert md_file.get_size == 13 assert md_file.get_line_count == 1 + def test_split_heading_levels(self): + """PdfFile and DocxFile share one heading parser.""" + assert _split_heading('# Title') == ('Title', 1) + assert _split_heading('## Section') == ('Section', 2) + assert _split_heading('### Notes') == ('Notes', 3) + assert _split_heading('#hashtag is not a header') == ('#hashtag is not a header', None) + assert _split_heading('plain') == ('plain', None) + assert _split_heading('#### too deep') == ('#### too deep', None) + def test_txt_file_creation(self): """Test TxtFile creation and basic properties.""" txt_file = TxtFile(name='notes', content='Hello\nWorld') @@ -265,6 +275,73 @@ class TestFileSystem: assert 'First & half' in blob assert 'Second 2' in blob + async def test_write_pdf_renders_markdown_formatting(self, empty_filesystem): + """PDF export honors bold, italic, inline code, and bullets.""" + source = '\n'.join( + [ + 'This is **bold text** in a sentence.', + 'This is *italic text* in a sentence.', + 'Use `inline_code` here.', + '- first bullet item', + '- second bullet item', + '* star bullet item', + ] + ) + result = await empty_filesystem.write_file('formatted.pdf', source) + assert result == 'Data written to file formatted.pdf successfully.' + blob, lines = _extract_pdf_text(empty_filesystem.data_dir / 'formatted.pdf') + + assert 'bold text' in blob + assert '**bold text**' not in blob + assert 'italic text' in blob + assert '*italic text*' not in blob + assert 'inline_code' in blob + assert '`inline_code`' not in blob + assert 'first bullet item' in blob + assert 'second bullet item' in blob + assert 'star bullet item' in blob + assert not any(line.startswith(('- ', '* ')) for line in lines) + + async def test_write_pdf_star_overload_is_not_emphasis(self, empty_filesystem): + """Arithmetic and glob stars must not be treated as italic or bullets.""" + source = '\n'.join( + [ + 'Arithmetic 2 * 3 * 4 stays literal', + 'Glob pattern *.txt and **/*.py stay literal', + ] + ) + result = await empty_filesystem.write_file('stars.pdf', source) + assert result == 'Data written to file stars.pdf successfully.' + blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'stars.pdf') + assert '2 * 3 * 4' in blob + assert '*.txt' in blob + assert '**/*.py' in blob + + async def test_write_pdf_underscore_is_not_italic(self, empty_filesystem): + """Underscores are identifiers, not emphasis.""" + source = 'Keep snake_case_names and file_name_here unchanged.' + result = await empty_filesystem.write_file('names.pdf', source) + assert result == 'Data written to file names.pdf successfully.' + blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'names.pdf') + assert 'snake_case_names' in blob + assert 'file_name_here' in blob + + async def test_write_pdf_fenced_backticks_stay_literal(self, empty_filesystem): + """Inline code strips backticks; fenced shell snippets keep them.""" + source = '\n'.join( + [ + 'Run `$(cmd)` inline.', + '```', + 'echo `$(cmd)`', + '```', + ] + ) + result = await empty_filesystem.write_file('backticks.pdf', source) + assert result == 'Data written to file backticks.pdf successfully.' + blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'backticks.pdf') + assert 'Run $(cmd) inline.' in blob + assert 'echo `$(cmd)`' in blob + def test_filesystem_initialization(self, temp_filesystem): """Test FileSystem initialization with default files.""" fs = temp_filesystem From beb7494caf7daa0eff69f5a6e10e11f7fee33784 Mon Sep 17 00:00:00 2001 From: lei_lei <96427312+leilei3167@users.noreply.github.com> Date: Tue, 25 Aug 2026 02:25:33 +0000 Subject: [PATCH 49/69] fix(filesystem): keep glob tokens and code-span stars literal in PDFs --- browser_use/filesystem/file_system.py | 18 +++++++++++++--- tests/ci/infrastructure/test_filesystem.py | 24 ++++++++++++++++++++++ 2 files changed, 39 insertions(+), 3 deletions(-) diff --git a/browser_use/filesystem/file_system.py b/browser_use/filesystem/file_system.py index f008eee19..5864af92f 100644 --- a/browser_use/filesystem/file_system.py +++ b/browser_use/filesystem/file_system.py @@ -99,15 +99,27 @@ def _markdown_inline_to_rml(text: str) -> str: injected ```` / ```` / ```` tags are unambiguous. Underscore emphasis is intentionally unsupported so ``snake_case`` identifiers - survive unchanged. + survive unchanged. Inline code is stashed before emphasis so markers inside + backticks stay Courier-only. Bold content cannot start with ``/``, so globs + like ``**/foo/**`` stay literal. """ text = html.escape(text) + # Stash code spans first โ€” Markdown renders their contents literally. + code_spans: list[str] = [] + + def _stash_code(match: re.Match[str]) -> str: + code_spans.append(match.group(1)) + return f'\x00C{len(code_spans) - 1}\x00' + + text = re.sub(r'`([^`]+)`', _stash_code, text) # Bold before italic so ``**`` is not treated as two italic markers. # Content cannot contain ``*`` โ€” otherwise globs like ``*.txt and **/*.py`` pair across tokens. - text = re.sub(r'\*\*([^\s*](?:[^*]*[^\s*])?)\*\*', r'\1', text) + # Content cannot start with ``/`` โ€” otherwise ``**/foo/**`` is treated as bold. + text = re.sub(r'\*\*([^\s*/](?:[^*]*[^\s*])?)\*\*', r'\1', text) # Non-space boundaries keep ``2 * 3 * 4`` literal; leading ``* `` is a bullet, not italic text = re.sub(r'(?\1', text) - text = re.sub(r'`([^`]+)`', r'\1', text) + for i, span in enumerate(code_spans): + text = text.replace(f'\x00C{i}\x00', f'{span}') return text diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index f5cd81937..0bf457482 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -17,6 +17,7 @@ from browser_use.filesystem.file_system import ( JsonlFile, MarkdownFile, TxtFile, + _markdown_inline_to_rml, _split_heading, ) @@ -53,6 +54,16 @@ class TestBaseFile: assert _split_heading('plain') == ('plain', None) assert _split_heading('#### too deep') == ('#### too deep', None) + def test_markdown_inline_glob_is_not_bold(self): + """Recursive globs must not be treated as bold delimiters.""" + assert _markdown_inline_to_rml('**/foo/**') == '**/foo/**' + assert _markdown_inline_to_rml('**bold** and **/foo/**') == 'bold and **/foo/**' + + def test_markdown_inline_code_protects_emphasis(self): + """Emphasis markers inside backticks stay Courier, not italic/bold.""" + assert _markdown_inline_to_rml('`*literal*`') == '*literal*' + assert _markdown_inline_to_rml('`**bold**`') == '**bold**' + def test_txt_file_creation(self): """Test TxtFile creation and basic properties.""" txt_file = TxtFile(name='notes', content='Hello\nWorld') @@ -308,6 +319,7 @@ class TestFileSystem: [ 'Arithmetic 2 * 3 * 4 stays literal', 'Glob pattern *.txt and **/*.py stay literal', + 'Recursive glob **/foo/** stays literal', ] ) result = await empty_filesystem.write_file('stars.pdf', source) @@ -316,6 +328,7 @@ class TestFileSystem: assert '2 * 3 * 4' in blob assert '*.txt' in blob assert '**/*.py' in blob + assert '**/foo/**' in blob async def test_write_pdf_underscore_is_not_italic(self, empty_filesystem): """Underscores are identifiers, not emphasis.""" @@ -342,6 +355,17 @@ class TestFileSystem: assert 'Run $(cmd) inline.' in blob assert 'echo `$(cmd)`' in blob + async def test_write_pdf_inline_code_protects_emphasis_markers(self, empty_filesystem): + """Stars inside inline code stay literal instead of becoming italic/bold.""" + source = 'Keep `*literal*` and `**bold**` as code.' + result = await empty_filesystem.write_file('code_stars.pdf', source) + assert result == 'Data written to file code_stars.pdf successfully.' + blob, _ = _extract_pdf_text(empty_filesystem.data_dir / 'code_stars.pdf') + assert '*literal*' in blob + assert '**bold**' in blob + assert '`*literal*`' not in blob + assert '`**bold**`' not in blob + def test_filesystem_initialization(self, temp_filesystem): """Test FileSystem initialization with default files.""" fs = temp_filesystem From f168606cb97b59f86d87e6666d2cd551af0eeb1a Mon Sep 17 00:00:00 2001 From: MagMueller Date: Wed, 26 Aug 2026 18:47:37 -0700 Subject: [PATCH 50/69] fix(filesystem): make PDF markdown restoration collision-safe --- browser_use/filesystem/file_system.py | 30 ++++++++++------------ tests/ci/infrastructure/test_filesystem.py | 29 ++++++++++++++++++++- 2 files changed, 41 insertions(+), 18 deletions(-) diff --git a/browser_use/filesystem/file_system.py b/browser_use/filesystem/file_system.py index 5864af92f..25d41cc35 100644 --- a/browser_use/filesystem/file_system.py +++ b/browser_use/filesystem/file_system.py @@ -104,23 +104,19 @@ def _markdown_inline_to_rml(text: str) -> str: like ``**/foo/**`` stay literal. """ text = html.escape(text) - # Stash code spans first โ€” Markdown renders their contents literally. - code_spans: list[str] = [] - - def _stash_code(match: re.Match[str]) -> str: - code_spans.append(match.group(1)) - return f'\x00C{len(code_spans) - 1}\x00' - - text = re.sub(r'`([^`]+)`', _stash_code, text) - # Bold before italic so ``**`` is not treated as two italic markers. - # Content cannot contain ``*`` โ€” otherwise globs like ``*.txt and **/*.py`` pair across tokens. - # Content cannot start with ``/`` โ€” otherwise ``**/foo/**`` is treated as bold. - text = re.sub(r'\*\*([^\s*/](?:[^*]*[^\s*])?)\*\*', r'\1', text) - # Non-space boundaries keep ``2 * 3 * 4`` literal; leading ``* `` is a bullet, not italic - text = re.sub(r'(?\1', text) - for i, span in enumerate(code_spans): - text = text.replace(f'\x00C{i}\x00', f'{span}') - return text + rendered: list[str] = [] + for part in re.split(r'(`[^`]+`)', text): + if part.startswith('`') and part.endswith('`'): + rendered.append(f'{part[1:-1]}') + continue + # Bold before italic so ``**`` is not treated as two italic markers. + # Content cannot contain ``*`` โ€” otherwise globs like ``*.txt and **/*.py`` pair across tokens. + # Content cannot start with ``/`` โ€” otherwise ``**/foo/**`` is treated as bold. + part = re.sub(r'\*\*([^\s*/](?:[^*]*[^\s*])?)\*\*', r'\1', part) + # Non-space boundaries keep ``2 * 3 * 4`` literal; leading ``* `` is a bullet, not italic + part = re.sub(r'(?\1', part) + rendered.append(part) + return ''.join(rendered) DEFAULT_FILE_SYSTEM_PATH = 'browseruse_agent_data' diff --git a/tests/ci/infrastructure/test_filesystem.py b/tests/ci/infrastructure/test_filesystem.py index 0bf457482..7875715d5 100644 --- a/tests/ci/infrastructure/test_filesystem.py +++ b/tests/ci/infrastructure/test_filesystem.py @@ -31,6 +31,21 @@ def _extract_pdf_text(path: Path) -> tuple[str, list[str]]: return ' '.join(text.split()), [line.strip() for line in text.splitlines() if line.strip()] +def _extract_pdf_fonts(path: Path) -> set[str]: + """Return the PDF base fonts used to render non-empty text.""" + fonts: set[str] = set() + + def _record_font(text: str, _cm, _tm, font_dictionary, _font_size) -> None: + if text.strip() and font_dictionary is not None: + base_font = font_dictionary.get('/BaseFont') + if base_font is not None: + fonts.add(str(base_font)) + + for page in PdfReader(path).pages: + page.extract_text(visitor_text=_record_font) + return fonts + + class TestBaseFile: """Test the BaseFile abstract base class and its implementations.""" @@ -64,6 +79,15 @@ class TestBaseFile: assert _markdown_inline_to_rml('`*literal*`') == '*literal*' assert _markdown_inline_to_rml('`**bold**`') == '**bold**' + def test_markdown_inline_code_token_cannot_replace_user_text(self): + """A user-provided placeholder-like sequence must survive code-span restoration.""" + literal = '\x00C0\x00' + assert _markdown_inline_to_rml(f'Keep {literal} and `code`.') == (f'Keep {literal} and code.') + prefix_literal = '\x00BROWSER_USE_INLINE_CODE_0\x00' + assert _markdown_inline_to_rml(f'Keep {prefix_literal} and `code`.') == ( + f'Keep {prefix_literal} and code.' + ) + def test_txt_file_creation(self): """Test TxtFile creation and basic properties.""" txt_file = TxtFile(name='notes', content='Hello\nWorld') @@ -300,7 +324,9 @@ class TestFileSystem: ) result = await empty_filesystem.write_file('formatted.pdf', source) assert result == 'Data written to file formatted.pdf successfully.' - blob, lines = _extract_pdf_text(empty_filesystem.data_dir / 'formatted.pdf') + pdf_path = empty_filesystem.data_dir / 'formatted.pdf' + blob, lines = _extract_pdf_text(pdf_path) + fonts = _extract_pdf_fonts(pdf_path) assert 'bold text' in blob assert '**bold text**' not in blob @@ -312,6 +338,7 @@ class TestFileSystem: assert 'second bullet item' in blob assert 'star bullet item' in blob assert not any(line.startswith(('- ', '* ')) for line in lines) + assert {'/Helvetica-Bold', '/Helvetica-Oblique', '/Courier'} <= fonts async def test_write_pdf_star_overload_is_not_emphasis(self, empty_filesystem): """Arithmetic and glob stars must not be treated as italic or bullets.""" From 9bfb1d3258ff78993d962b18580c99ccd39f7ac3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E9=83=91=E8=80=80=E7=BF=94?= <2082723353@qq.com> Date: Fri, 28 Aug 2026 09:11:41 +0800 Subject: [PATCH 51/69] fix(registry): replace sensitive placeholders in tuples --- browser_use/tools/registry/service.py | 4 ++- tests/ci/security/test_sensitive_data.py | 35 ++++++++++++++++++++++++ 2 files changed, 38 insertions(+), 1 deletion(-) diff --git a/browser_use/tools/registry/service.py b/browser_use/tools/registry/service.py index f45068b07..1f4e7838d 100644 --- a/browser_use/tools/registry/service.py +++ b/browser_use/tools/registry/service.py @@ -464,7 +464,7 @@ class Registry(Generic[Context]): # Filter out empty values applicable_secrets = {k: v for k, v in applicable_secrets.items() if v} - def recursively_replace_secrets(value: str | dict | list) -> str | dict | list: + def recursively_replace_secrets(value: str | dict | list | tuple) -> str | dict | list | tuple: if isinstance(value, str): # 1. Handle tagged secrets: label matches = secret_pattern.findall(value) @@ -499,6 +499,8 @@ class Registry(Generic[Context]): return {k: recursively_replace_secrets(v) for k, v in value.items()} elif isinstance(value, list): return [recursively_replace_secrets(v) for v in value] + elif isinstance(value, tuple): + return tuple(recursively_replace_secrets(v) for v in value) return value params_dump = params.model_dump() diff --git a/tests/ci/security/test_sensitive_data.py b/tests/ci/security/test_sensitive_data.py index 3976e31da..6160653d7 100644 --- a/tests/ci/security/test_sensitive_data.py +++ b/tests/ci/security/test_sensitive_data.py @@ -18,6 +18,18 @@ class SensitiveParams(BaseModel): text: str = Field(description='Text with sensitive data placeholders') +class TupleSensitiveParams(BaseModel): + """Test parameter model for tuple-based sensitive data placeholders.""" + + items: tuple[str, ...] = Field(description='Tuple with sensitive data placeholders') + + +class NestedTupleSensitiveParams(BaseModel): + """Test parameter model for nested tuple-based sensitive data placeholders.""" + + payload: tuple[dict[str, tuple[str, list[str]]], ...] = Field(description='Nested tuple with sensitive data placeholders') + + @pytest.fixture def registry(): return Registry() @@ -72,6 +84,29 @@ def test_replace_sensitive_data_with_missing_keys(registry, caplog): assert 'password' in result.text # Empty value's tag remains +def test_replace_sensitive_data_inside_tuple(registry): + """Test that _replace_sensitive_data replaces placeholders inside tuple fields.""" + params = TupleSensitiveParams(items=('api_key', 'username', 'unchanged')) + sensitive_data = {'api_key': 'sk-replaced', 'username': 'admin_user'} + + result = registry._replace_sensitive_data(params, sensitive_data) + + assert result.items == ('sk-replaced', 'admin_user', 'unchanged') + assert isinstance(result.items, tuple) + + +def test_replace_sensitive_data_inside_nested_tuple(registry): + """Test that _replace_sensitive_data replaces placeholders inside nested tuples.""" + params = NestedTupleSensitiveParams(payload=({'credentials': ('token', ['username'])},)) + sensitive_data = {'token': 'token-replaced', 'username': 'admin_user'} + + result = registry._replace_sensitive_data(params, sensitive_data) + + assert result.payload == ({'credentials': ('token-replaced', ['admin_user'])},) + assert isinstance(result.payload, tuple) + assert isinstance(result.payload[0]['credentials'], tuple) + + def test_simple_domain_specific_sensitive_data(registry, caplog): """Test the basic functionality of domain-specific sensitive data replacement""" # Create a simple Pydantic model with sensitive data placeholders From d33cc68f06d51169e5c481b535bcbc36fa9974b4 Mon Sep 17 00:00:00 2001 From: Aniket Wagh Date: Fri, 28 Aug 2026 08:34:44 +0530 Subject: [PATCH 52/69] fix: keep balanced trailing brackets in URLs extracted from task text sanitize_url_candidate() strips trailing prose punctuation so that "Go to https://example.com/docs." does not navigate with the sentence's period attached. It also stripped every trailing ) and ], including the ones the URL opened itself, so a task like Summarize https://en.wikipedia.org/wiki/Python_(programming_language) auto-navigated to .../Python_(programming_language and landed on the wrong page. Wikipedia disambiguation links are the common case. Whether the bracket belongs to the URL is decided by balance: a closing bracket with a matching opener inside the candidate is part of the path, while one the prose opened, as in "(see https://example.com/guide)", is not. Strip trailing punctuation as before, and only drop a closing bracket when the candidate has more of them than openers. Fixes #5575 --- browser_use/utils.py | 21 ++++++++++++++++++++- tests/ci/test_beta_agent.py | 8 ++++++++ 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/browser_use/utils.py b/browser_use/utils.py index 03e651d38..eb1814123 100644 --- a/browser_use/utils.py +++ b/browser_use/utils.py @@ -39,6 +39,10 @@ def is_placeholder_url(url: str) -> bool: return len(labels) >= 2 and all(re.fullmatch(r'x+', label) for label in labels) +_TRAILING_PROSE_PUNCTUATION = frozenset('.,;:!?([') +_CLOSING_TO_OPENING_BRACKET = {')': '(', ']': '['} + + def sanitize_url_candidate(url: str) -> str: """Normalize a URL candidate captured from prose before auto-navigation.""" candidate = url.strip() @@ -46,7 +50,22 @@ def sanitize_url_candidate(url: str) -> str: # "https://example.com/search.\\n2. Next step". Those are task text, # not part of the URL. candidate = re.split(r'\\[nrt]', candidate, maxsplit=1)[0] - return re.sub(r'[.,;:!?()\[\]]+$', '', candidate) + + # Strip trailing prose punctuation, but keep a closing bracket the URL opened + # itself, e.g. /wiki/Python_(programming_language). A closing bracket is only + # prose when it has no opener inside the candidate, as in "(see https://x.com/a)". + while candidate: + last_char = candidate[-1] + if last_char in _TRAILING_PROSE_PUNCTUATION: + candidate = candidate[:-1] + continue + opening_bracket = _CLOSING_TO_OPENING_BRACKET.get(last_char) + if opening_bracket is not None and candidate.count(last_char) > candidate.count(opening_bracket): + candidate = candidate[:-1] + continue + break + + return candidate # Lazy import for error types diff --git a/tests/ci/test_beta_agent.py b/tests/ci/test_beta_agent.py index 82aef6f9c..29f638856 100644 --- a/tests/ci/test_beta_agent.py +++ b/tests/ci/test_beta_agent.py @@ -4305,6 +4305,14 @@ def test_beta_agent_exposes_task_helper_methods(): assert agent._extract_start_url(numbered_task) == 'https://elibrary.ferc.gov/eLibrary/search' assert browser_use_agent._extract_start_url(numbered_task) == 'https://elibrary.ferc.gov/eLibrary/search' + # A closing bracket the URL opened itself is part of the path, not prose. + wikipedia_task = 'Summarize https://en.wikipedia.org/wiki/Python_(programming_language)' + assert agent._extract_start_url(wikipedia_task) == 'https://en.wikipedia.org/wiki/Python_(programming_language)' + assert browser_use_agent._extract_start_url(wikipedia_task) == 'https://en.wikipedia.org/wiki/Python_(programming_language)' + assert agent._extract_start_url('Check https://example.com/a[1] please') == 'https://example.com/a[1]' + # ...but one the prose opened is still dropped. + assert agent._extract_start_url('See the docs (https://example.com/guide) for details.') == 'https://example.com/guide' + def test_beta_agent_exposes_url_text_helper_methods(): from browser_use.beta import Agent From 3c565af3e010cd643b4683686b614b50970704df Mon Sep 17 00:00:00 2001 From: Aniket Wagh Date: Fri, 28 Aug 2026 08:59:10 +0530 Subject: [PATCH 53/69] perf: keep bracket trimming linear in sanitize_url_candidate The trimming loop counted brackets across the whole candidate on every iteration, so a candidate ending in many unmatched brackets rescanned it once per bracket. 50,000 trailing ')' took 0.9s, where the regex this replaced was linear. Count the four bracket characters once and decrement as characters are trimmed, tracking the end index instead of reslicing. Same results, and 200,000 trailing ')' now takes 0.011s. --- browser_use/utils.py | 19 +++++++++++++------ tests/ci/test_beta_agent.py | 3 ++- 2 files changed, 15 insertions(+), 7 deletions(-) diff --git a/browser_use/utils.py b/browser_use/utils.py index eb1814123..92af31202 100644 --- a/browser_use/utils.py +++ b/browser_use/utils.py @@ -54,18 +54,25 @@ def sanitize_url_candidate(url: str) -> str: # Strip trailing prose punctuation, but keep a closing bracket the URL opened # itself, e.g. /wiki/Python_(programming_language). A closing bracket is only # prose when it has no opener inside the candidate, as in "(see https://x.com/a)". - while candidate: - last_char = candidate[-1] + # Bracket totals are counted once and decremented as characters are trimmed, so + # a candidate ending in many brackets stays linear. + bracket_counts = {bracket: candidate.count(bracket) for bracket in '()[]'} + end = len(candidate) + while end: + last_char = candidate[end - 1] if last_char in _TRAILING_PROSE_PUNCTUATION: - candidate = candidate[:-1] + if last_char in bracket_counts: + bracket_counts[last_char] -= 1 + end -= 1 continue opening_bracket = _CLOSING_TO_OPENING_BRACKET.get(last_char) - if opening_bracket is not None and candidate.count(last_char) > candidate.count(opening_bracket): - candidate = candidate[:-1] + if opening_bracket is not None and bracket_counts[last_char] > bracket_counts[opening_bracket]: + bracket_counts[last_char] -= 1 + end -= 1 continue break - return candidate + return candidate[:end] # Lazy import for error types diff --git a/tests/ci/test_beta_agent.py b/tests/ci/test_beta_agent.py index 29f638856..fd831d4f9 100644 --- a/tests/ci/test_beta_agent.py +++ b/tests/ci/test_beta_agent.py @@ -4310,8 +4310,9 @@ def test_beta_agent_exposes_task_helper_methods(): assert agent._extract_start_url(wikipedia_task) == 'https://en.wikipedia.org/wiki/Python_(programming_language)' assert browser_use_agent._extract_start_url(wikipedia_task) == 'https://en.wikipedia.org/wiki/Python_(programming_language)' assert agent._extract_start_url('Check https://example.com/a[1] please') == 'https://example.com/a[1]' - # ...but one the prose opened is still dropped. + # ...but one the prose opened is still dropped, however many there are. assert agent._extract_start_url('See the docs (https://example.com/guide) for details.') == 'https://example.com/guide' + assert agent._extract_start_url('Read ((see https://example.com/a)) now') == 'https://example.com/a' def test_beta_agent_exposes_url_text_helper_methods(): From 58d88ae3094ca9ef3bf24d73e388add1b5fa39ef Mon Sep 17 00:00:00 2001 From: Saurav Panda Date: Fri, 28 Aug 2026 18:11:15 +0000 Subject: [PATCH 54/69] Bump datamodel-code-generator to 0.75.1 to clear 10 CVEs The eval extra pinned datamodel-code-generator==0.53.0, which falls inside the affected range of ten advisories (CVE-2026-54621, -54653, -54654, -54655, -54656, -54690, -54691, -55389, -55391, -55415). The widest range is CVE-2026-55415 (<= 0.63.0), so any release above 0.63.0 clears all ten; 0.75.1 is the current latest. CI installs this via `uv sync --dev --all-extras`, so the vulnerable package was landing in CI environments even though nothing in the repo imports it or invokes the datamodel-codegen CLI. Verified the full graph (399 packages, all extras + dev) still resolves, and that 0.75.1 installs and generates models correctly. Requires-python >=3.10 and pydantic <3,>=2 both stay compatible with the project's pins. --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 0f083c301..0a4c0768f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -82,7 +82,7 @@ eval = [ "lmnr[all]==0.7.42", "anyio==4.12.1", "psutil==7.2.2", - "datamodel-code-generator==0.53.0", + "datamodel-code-generator==0.75.1", ] cli-oci = ["browser-use[cli,oci]"] all = ["browser-use[cli,examples,aws,oci]"] From 38fa689a10d6b8ffcd327c2e95b038034b0d0217 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 20 Aug 2026 16:57:34 -0700 Subject: [PATCH 55/69] build: require Browser Harness 0.1.10 --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 0f083c301..82446a451 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -46,7 +46,7 @@ dependencies = [ "markdownify==1.2.2", "python-docx==1.2.0", "browser-use-sdk==3.4.2", - "browser-harness==0.1.9", + "browser-harness==0.1.10", ] # google-api-core: only used for Google LLM APIs # pyperclip: only used for examples that use copy/paste From 6de522cb9ac550eb45cb0e6202a6238d74aee078 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Fri, 28 Aug 2026 12:20:04 -0700 Subject: [PATCH 56/69] docs: sync generated Browser Harness skill --- browser_use/skills/browser-use/SKILL.md | 13 ++++++++++++- skills/browser-use/SKILL.md | 13 ++++++++++++- 2 files changed, 24 insertions(+), 2 deletions(-) diff --git a/browser_use/skills/browser-use/SKILL.md b/browser_use/skills/browser-use/SKILL.md index 4b359b2bf..a70a020a6 100644 --- a/browser_use/skills/browser-use/SKILL.md +++ b/browser_use/skills/browser-use/SKILL.md @@ -43,11 +43,22 @@ PY - Invoke as `browser-use`. Use heredocs for multi-line commands. - Helpers are pre-imported. `run.py` calls `ensure_daemon()` before `exec`. -- First navigation is `new_tab(url)`, not `goto_url(url)`. +- First navigation for a task is `new_tab(url)`, not `goto_url(url)`. The daemon + preserves the attached tab across separate CLI invocations, so do not call + `new_tab()` again in every script. +- Keep one working tab per task/site. Before opening another, inspect + `current_tab()` and `list_tabs()` and use `switch_tab()` to reuse a matching + tab. Do not leave duplicate tabs on the same URL or close tabs you did not + create. - `new_tab()` and `switch_tab()` attach and move the horse marker without changing Chrome's visible tab. Screenshots and normal CDP input work in the background; call `activate_tab(target)` only when the user explicitly asks or a page demonstrably pauses rendering while hidden. +- A timed-out `scroll(...)` on an attached background tab is evidence that the + page needs to be visible. Call `activate_tab(current_tab())`, retry the same + scroll once, then re-read the scroll position. This visibly switches tabs, + so do not use it when the user has forbidden foreground changes. Do not + invent a `Runtime.evaluate` scroll replacement or a cross-frame JS walker. - The normal local flow attaches to the running Chrome/Chromium CDP endpoint. No browser ids or local profile selection. ## Local Chrome diff --git a/skills/browser-use/SKILL.md b/skills/browser-use/SKILL.md index 4b359b2bf..a70a020a6 100644 --- a/skills/browser-use/SKILL.md +++ b/skills/browser-use/SKILL.md @@ -43,11 +43,22 @@ PY - Invoke as `browser-use`. Use heredocs for multi-line commands. - Helpers are pre-imported. `run.py` calls `ensure_daemon()` before `exec`. -- First navigation is `new_tab(url)`, not `goto_url(url)`. +- First navigation for a task is `new_tab(url)`, not `goto_url(url)`. The daemon + preserves the attached tab across separate CLI invocations, so do not call + `new_tab()` again in every script. +- Keep one working tab per task/site. Before opening another, inspect + `current_tab()` and `list_tabs()` and use `switch_tab()` to reuse a matching + tab. Do not leave duplicate tabs on the same URL or close tabs you did not + create. - `new_tab()` and `switch_tab()` attach and move the horse marker without changing Chrome's visible tab. Screenshots and normal CDP input work in the background; call `activate_tab(target)` only when the user explicitly asks or a page demonstrably pauses rendering while hidden. +- A timed-out `scroll(...)` on an attached background tab is evidence that the + page needs to be visible. Call `activate_tab(current_tab())`, retry the same + scroll once, then re-read the scroll position. This visibly switches tabs, + so do not use it when the user has forbidden foreground changes. Do not + invent a `Runtime.evaluate` scroll replacement or a cross-frame JS walker. - The normal local flow attaches to the running Chrome/Chromium CDP endpoint. No browser ids or local profile selection. ## Local Chrome From 80802eefb9cff61ad09ac0e2ef53987a8d6f6fcb Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 27 Aug 2026 21:23:18 -0700 Subject: [PATCH 57/69] docs(cloud): replace retired v1 API examples with v4 --- examples/cloud/01_basic_task.py | 193 +++---------- examples/cloud/02_fast_mode_gemini.py | 265 ------------------ examples/cloud/03_structured_output.py | 362 ------------------------- examples/cloud/04_proxy_usage.py | 331 ---------------------- examples/cloud/05_search_api.py | 355 ------------------------ examples/cloud/README.md | 137 +--------- examples/cloud/env.example | 22 +- 7 files changed, 48 insertions(+), 1617 deletions(-) delete mode 100644 examples/cloud/02_fast_mode_gemini.py delete mode 100644 examples/cloud/03_structured_output.py delete mode 100644 examples/cloud/04_proxy_usage.py delete mode 100644 examples/cloud/05_search_api.py diff --git a/examples/cloud/01_basic_task.py b/examples/cloud/01_basic_task.py index 54b8ea561..2e88ff668 100644 --- a/examples/cloud/01_basic_task.py +++ b/examples/cloud/01_basic_task.py @@ -1,187 +1,56 @@ -""" -Cloud Example 1: Your First Browser Use Cloud Task -================================================== - -This example demonstrates the most basic Browser Use Cloud functionality: -- Create a simple automation task -- Get the task ID -- Monitor completion -- Retrieve results - -Perfect for first-time cloud users to understand the API basics. - -Cost: ~$0.04 (1 task + 3 steps with GPT-4.1 mini) -""" +"""Run one Browser Use Cloud API V4 task.""" import os import time from typing import Any import requests -from requests.exceptions import RequestException +from dotenv import load_dotenv -# Configuration +load_dotenv() + +API_URL = os.getenv('BROWSER_USE_API_URL', 'https://api.browser-use.com/api/v4').rstrip('/') API_KEY = os.getenv('BROWSER_USE_API_KEY') if not API_KEY: - raise ValueError( - 'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key' - ) + raise RuntimeError('Set BROWSER_USE_API_KEY or add it to .env') -BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1') -TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30')) -HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'} +HEADERS = {'X-Browser-Use-API-Key': API_KEY} +TERMINAL_STATUSES = {'completed', 'failed', 'cancelled'} -def _request_with_retry(method: str, url: str, **kwargs) -> requests.Response: - """Make HTTP request with timeout and retry logic.""" - kwargs.setdefault('timeout', TIMEOUT) - - for attempt in range(3): - try: - response = requests.request(method, url, **kwargs) - response.raise_for_status() - return response - except RequestException as e: - if attempt == 2: # Last attempt - raise - sleep_time = 2**attempt - print(f'โš ๏ธ Request failed (attempt {attempt + 1}/3), retrying in {sleep_time}s: {e}') - time.sleep(sleep_time) - - # This line should never be reached, but satisfies type checker - raise RuntimeError('Unexpected error in retry logic') +def create_run(task: str) -> str: + response = requests.post(f'{API_URL}/runs', headers=HEADERS, json={'task': task}, timeout=30) + response.raise_for_status() + return response.json()['id'] -def create_task(instructions: str) -> str: - """ - Create a new browser automation task. - - Args: - instructions: Natural language description of what the agent should do - - Returns: - task_id: Unique identifier for the created task - """ - print(f'๐Ÿ“ Creating task: {instructions}') - - payload = { - 'task': instructions, - 'llm_model': 'gpt-4.1-mini', # Cost-effective model - 'max_agent_steps': 10, # Prevent runaway costs - 'enable_public_share': True, # Enable shareable execution URLs - } - - response = _request_with_retry('post', f'{BASE_URL}/run-task', headers=HEADERS, json=payload) - - task_id = response.json()['id'] - print(f'โœ… Task created with ID: {task_id}') - return task_id - - -def get_task_status(task_id: str) -> dict[str, Any]: - """Get the current status of a task.""" - response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}/status', headers=HEADERS) - return response.json() - - -def get_task_details(task_id: str) -> dict[str, Any]: - """Get full task details including steps and output.""" - response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS) - return response.json() - - -def wait_for_completion(task_id: str, poll_interval: int = 3) -> dict[str, Any]: - """ - Wait for task completion and show progress. - - Args: - task_id: The task to monitor - poll_interval: How often to check status (seconds) - - Returns: - Complete task details with output - """ - print(f'โณ Monitoring task {task_id}...') - - step_count = 0 - start_time = time.time() - +def wait_for_run(run_id: str, poll_seconds: float = 2) -> dict[str, Any]: while True: - details = get_task_details(task_id) - status = details['status'] - current_steps = len(details.get('steps', [])) - elapsed = time.time() - start_time + response = requests.get(f'{API_URL}/runs/{run_id}/status', headers=HEADERS, timeout=30) + response.raise_for_status() + status = response.json()['status'] + print(f'Status: {status}') - # Clear line and show current progress - if current_steps > step_count: - step_count = current_steps + if status in TERMINAL_STATUSES: + break - # Build status message - if status == 'running': - if current_steps > 0: - status_msg = f'๐Ÿ”„ Step {current_steps} | โฑ๏ธ {elapsed:.0f}s | ๐Ÿค– Agent working...' - else: - status_msg = f'๐Ÿค– Agent starting... | โฑ๏ธ {elapsed:.0f}s' - else: - status_msg = f'๐Ÿ”„ Step {current_steps} | โฑ๏ธ {elapsed:.0f}s | Status: {status}' + time.sleep(poll_seconds) - # Clear line and print status - print(f'\r{status_msg:<80}', end='', flush=True) - - # Check if finished - if status == 'finished': - print(f'\rโœ… Task completed successfully! ({current_steps} steps in {elapsed:.1f}s)' + ' ' * 20) - return details - elif status in ['failed', 'stopped']: - print(f'\rโŒ Task {status} after {current_steps} steps' + ' ' * 30) - return details - - time.sleep(poll_interval) + response = requests.get(f'{API_URL}/runs/{run_id}', headers=HEADERS, timeout=30) + response.raise_for_status() + return response.json() -def main(): - """Run a basic cloud automation task.""" - print('๐Ÿš€ Browser Use Cloud - Basic Task Example') - print('=' * 50) +def main() -> None: + run_id = create_run('Find the top story on Hacker News and summarize it in one sentence.') + print(f'Run: {run_id}') - # Define a simple search task (using DuckDuckGo to avoid captchas) - task_description = ( - "Go to DuckDuckGo and search for 'browser automation tools'. Tell me the top 3 results with their titles and URLs." - ) + run = wait_for_run(run_id) + if run['status'] != 'completed': + raise RuntimeError(run.get('error') or f'Run {run["status"]}') - try: - # Step 1: Create the task - task_id = create_task(task_description) - - # Step 2: Wait for completion - result = wait_for_completion(task_id) - - # Step 3: Display results - print('\n๐Ÿ“Š Results:') - print('-' * 30) - print(f'Status: {result["status"]}') - print(f'Steps taken: {len(result.get("steps", []))}') - - if result.get('output'): - print(f'Output: {result["output"]}') - else: - print('No output available') - - # Show share URLs for viewing execution - if result.get('live_url'): - print(f'\n๐Ÿ”— Live Preview: {result["live_url"]}') - if result.get('public_share_url'): - print(f'๐ŸŒ Share URL: {result["public_share_url"]}') - elif result.get('share_url'): - print(f'๐ŸŒ Share URL: {result["share_url"]}') - - if not result.get('live_url') and not result.get('public_share_url') and not result.get('share_url'): - print("\n๐Ÿ’ก Tip: Add 'enable_public_share': True to task payload to get shareable URLs") - - except requests.exceptions.RequestException as e: - print(f'โŒ API Error: {e}') - except Exception as e: - print(f'โŒ Error: {e}') + print(f'Result: {run["result"]}') + print(f'Cost: ${run["totalCostUsd"]}') if __name__ == '__main__': diff --git a/examples/cloud/02_fast_mode_gemini.py b/examples/cloud/02_fast_mode_gemini.py deleted file mode 100644 index 245da5c20..000000000 --- a/examples/cloud/02_fast_mode_gemini.py +++ /dev/null @@ -1,265 +0,0 @@ -""" -Cloud Example 2: Ultra-Fast Mode with Gemini Flash โšก -==================================================== - -This example demonstrates the fastest and most cost-effective configuration: -- Gemini 2.5 Flash model ($0.01 per step) -- No proxy (faster execution, but no captcha solving) -- No element highlighting (better performance) -- Optimized viewport size -- Maximum speed configuration - -Perfect for: Quick content generation, humor tasks, fast web scraping - -Cost: ~$0.03 (1 task + 2-3 steps with Gemini Flash) -Speed: 2-3x faster than default configuration -Fun Factor: ๐Ÿ’ฏ (Creates hilarious tech commentary) -""" - -import argparse -import os -import time -from typing import Any - -import requests -from requests.exceptions import RequestException - -# Configuration -API_KEY = os.getenv('BROWSER_USE_API_KEY') -if not API_KEY: - raise ValueError( - 'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key' - ) - -BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1') -TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30')) -HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'} - - -def _request_with_retry(method: str, url: str, **kwargs) -> requests.Response: - """Make HTTP request with timeout and retry logic.""" - kwargs.setdefault('timeout', TIMEOUT) - - for attempt in range(3): - try: - response = requests.request(method, url, **kwargs) - response.raise_for_status() - return response - except RequestException as e: - if attempt == 2: # Last attempt - raise - sleep_time = 2**attempt - print(f'โš ๏ธ Request failed (attempt {attempt + 1}/3), retrying in {sleep_time}s: {e}') - time.sleep(sleep_time) - - raise RuntimeError('Unexpected error in retry logic') - - -def create_fast_task(instructions: str) -> str: - """ - Create a browser automation task optimized for speed and cost. - - Args: - instructions: Natural language description of what the agent should do - - Returns: - task_id: Unique identifier for the created task - """ - print(f'โšก Creating FAST task: {instructions}') - - # Ultra-fast configuration - payload = { - 'task': instructions, - # Model: Fastest and cheapest - 'llm_model': 'gemini-2.5-flash', - # Performance optimizations - 'use_proxy': False, # No proxy = faster execution - 'highlight_elements': False, # No highlighting = better performance - 'use_adblock': True, # Block ads for faster loading - # Viewport optimization (smaller = faster) - 'browser_viewport_width': 1024, - 'browser_viewport_height': 768, - # Cost control - 'max_agent_steps': 25, # Reasonable limit for fast tasks - # Enable sharing for viewing execution - 'enable_public_share': True, # Get shareable URLs - # Optional: Speed up with domain restrictions - # "allowed_domains": ["google.com", "*.google.com"] - } - - response = _request_with_retry('post', f'{BASE_URL}/run-task', headers=HEADERS, json=payload) - - task_id = response.json()['id'] - print(f'โœ… Fast task created with ID: {task_id}') - print('โšก Configuration: Gemini Flash + No Proxy + No Highlighting') - return task_id - - -def monitor_fast_task(task_id: str) -> dict[str, Any]: - """ - Monitor task with optimized polling for fast execution. - - Args: - task_id: The task to monitor - - Returns: - Complete task details with output - """ - print(f'๐Ÿš€ Fast monitoring task {task_id}...') - - start_time = time.time() - step_count = 0 - last_step_time = start_time - - # Faster polling for quick tasks - poll_interval = 1 # Check every second for fast tasks - - while True: - response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS) - details = response.json() - status = details['status'] - - # Show progress with timing - current_steps = len(details.get('steps', [])) - elapsed = time.time() - start_time - - # Build status message - if current_steps > step_count: - step_time = time.time() - last_step_time - last_step_time = time.time() - step_count = current_steps - step_msg = f'๐Ÿ”ฅ Step {current_steps} | โšก {step_time:.1f}s | Total: {elapsed:.1f}s' - else: - if status == 'running': - step_msg = f'๐Ÿš€ Step {current_steps} | โฑ๏ธ {elapsed:.1f}s | Fast processing...' - else: - step_msg = f'๐Ÿš€ Step {current_steps} | โฑ๏ธ {elapsed:.1f}s | Status: {status}' - - # Clear line and show progress - print(f'\r{step_msg:<80}', end='', flush=True) - - # Check completion - if status == 'finished': - total_time = time.time() - start_time - if current_steps > 0: - avg_msg = f'โšก Average: {total_time / current_steps:.1f}s per step' - else: - avg_msg = 'โšก No steps recorded' - print(f'\r๐Ÿ Task completed in {total_time:.1f}s! {avg_msg}' + ' ' * 20) - return details - - elif status in ['failed', 'stopped']: - print(f'\rโŒ Task {status} after {elapsed:.1f}s' + ' ' * 30) - return details - - time.sleep(poll_interval) - - -def run_speed_comparison(): - """Run multiple tasks to compare speed vs accuracy.""" - print('\n๐Ÿƒโ€โ™‚๏ธ Speed Comparison Demo') - print('=' * 40) - - tasks = [ - 'Go to ProductHunt and roast the top product like a sarcastic tech reviewer', - 'Visit Reddit r/ProgrammerHumor and summarize the top post as a dramatic news story', - "Check GitHub trending and write a conspiracy theory about why everyone's switching to Rust", - ] - - results = [] - - for i, task in enumerate(tasks, 1): - print(f'\n๐Ÿ“ Fast Task {i}/{len(tasks)}') - print(f'Task: {task}') - - start = time.time() - task_id = create_fast_task(task) - result = monitor_fast_task(task_id) - end = time.time() - - results.append( - { - 'task': task, - 'duration': end - start, - 'steps': len(result.get('steps', [])), - 'status': result['status'], - 'output': result.get('output', '')[:100] + '...' if result.get('output') else 'No output', - } - ) - - # Summary - print('\n๐Ÿ“Š Speed Summary') - print('=' * 50) - total_time = sum(r['duration'] for r in results) - total_steps = sum(r['steps'] for r in results) - - for i, result in enumerate(results, 1): - print(f'Task {i}: {result["duration"]:.1f}s ({result["steps"]} steps) - {result["status"]}') - - print(f'\nโšก Total time: {total_time:.1f}s') - print(f'๐Ÿ”ฅ Average per task: {total_time / len(results):.1f}s') - if total_steps > 0: - print(f'๐Ÿ’จ Average per step: {total_time / total_steps:.1f}s') - else: - print('๐Ÿ’จ Average per step: N/A (no steps recorded)') - - -def main(): - """Demonstrate ultra-fast cloud automation.""" - print('โšก Browser Use Cloud - Ultra-Fast Mode with Gemini Flash') - print('=' * 60) - - print('๐ŸŽฏ Configuration Benefits:') - print('โ€ข Gemini Flash: $0.01 per step (cheapest)') - print('โ€ข No proxy: 30% faster execution') - print('โ€ข No highlighting: Better performance') - print('โ€ข Optimized viewport: Faster rendering') - - try: - # Single fast task - print('\n๐Ÿš€ Single Fast Task Demo') - print('-' * 30) - - task = """ - Go to Hacker News (news.ycombinator.com) and get the top 3 articles from the front page. - - Then, write a funny tech news segment in the style of Fireship YouTube channel: - - Be sarcastic and witty about tech trends - - Use developer humor and memes - - Make fun of common programming struggles - - Include phrases like "And yes, it runs on JavaScript" or "Plot twist: it's written in Rust" - - Keep it under 250 words but make it entertaining - - Structure it like a news anchor delivering breaking tech news - - Make each story sound dramatic but also hilarious, like you're reporting on the most important events in human history. - """ - task_id = create_fast_task(task) - result = monitor_fast_task(task_id) - - print(f'\n๐Ÿ“Š Result: {result.get("output", "No output")}') - - # Show execution URLs - if result.get('live_url'): - print(f'\n๐Ÿ”— Live Preview: {result["live_url"]}') - if result.get('public_share_url'): - print(f'๐ŸŒ Share URL: {result["public_share_url"]}') - elif result.get('share_url'): - print(f'๐ŸŒ Share URL: {result["share_url"]}') - - # Optional: Run speed comparison with --compare flag - parser = argparse.ArgumentParser(description='Fast mode demo with Gemini Flash') - parser.add_argument('--compare', action='store_true', help='Run speed comparison with 3 tasks') - args = parser.parse_args() - - if args.compare: - print('\n๐Ÿƒโ€โ™‚๏ธ Running speed comparison...') - run_speed_comparison() - - except requests.exceptions.RequestException as e: - print(f'โŒ API Error: {e}') - except Exception as e: - print(f'โŒ Error: {e}') - - -if __name__ == '__main__': - main() diff --git a/examples/cloud/03_structured_output.py b/examples/cloud/03_structured_output.py deleted file mode 100644 index cbe73ac11..000000000 --- a/examples/cloud/03_structured_output.py +++ /dev/null @@ -1,362 +0,0 @@ -""" -Cloud Example 3: Structured JSON Output ๐Ÿ“‹ -========================================== - -This example demonstrates how to get structured, validated JSON output: -- Define Pydantic schemas for type safety -- Extract structured data from websites -- Validate and parse JSON responses -- Handle different data types and nested structures - -Perfect for: Data extraction, API integration, structured analysis - -Cost: ~$0.06 (1 task + 5-6 steps with GPT-4.1 mini) -""" - -import argparse -import json -import os -import time -from typing import Any - -import requests -from pydantic import BaseModel, Field, ValidationError -from requests.exceptions import RequestException - -# Configuration -API_KEY = os.getenv('BROWSER_USE_API_KEY') -if not API_KEY: - raise ValueError( - 'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key' - ) - -BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1') -TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30')) -HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'} - - -def _request_with_retry(method: str, url: str, **kwargs) -> requests.Response: - """Make HTTP request with timeout and retry logic.""" - kwargs.setdefault('timeout', TIMEOUT) - - for attempt in range(3): - try: - response = requests.request(method, url, **kwargs) - response.raise_for_status() - return response - except RequestException as e: - if attempt == 2: # Last attempt - raise - sleep_time = 2**attempt - print(f'โš ๏ธ Request failed (attempt {attempt + 1}/3), retrying in {sleep_time}s: {e}') - time.sleep(sleep_time) - - raise RuntimeError('Unexpected error in retry logic') - - -# Define structured output schemas using Pydantic -class NewsArticle(BaseModel): - """Schema for a news article.""" - - title: str = Field(description='The headline of the article') - summary: str = Field(description='Brief summary of the article') - url: str = Field(description='Direct link to the article') - published_date: str | None = Field(description='Publication date if available') - category: str | None = Field(description='Article category/section') - - -class NewsResponse(BaseModel): - """Schema for multiple news articles.""" - - articles: list[NewsArticle] = Field(description='List of news articles') - source_website: str = Field(description='The website where articles were found') - extracted_at: str = Field(description='When the data was extracted') - - -class ProductInfo(BaseModel): - """Schema for product information.""" - - name: str = Field(description='Product name') - price: float = Field(description='Product price in USD') - rating: float | None = Field(description='Average rating (0-5 scale)') - availability: str = Field(description='Stock status (in stock, out of stock, etc.)') - description: str = Field(description='Product description') - - -class CompanyInfo(BaseModel): - """Schema for company information.""" - - name: str = Field(description='Company name') - stock_symbol: str | None = Field(description='Stock ticker symbol') - market_cap: str | None = Field(description='Market capitalization') - industry: str = Field(description='Primary industry') - headquarters: str = Field(description='Headquarters location') - founded_year: int | None = Field(description='Year founded') - - -def create_structured_task(instructions: str, schema_model: type[BaseModel], **kwargs) -> str: - """ - Create a task that returns structured JSON output. - - Args: - instructions: Task description - schema_model: Pydantic model defining the expected output structure - **kwargs: Additional task parameters - - Returns: - task_id: Unique identifier for the created task - """ - print(f'๐Ÿ“ Creating structured task: {instructions}') - print(f'๐Ÿ—๏ธ Expected schema: {schema_model.__name__}') - - # Generate JSON schema from Pydantic model - json_schema = schema_model.model_json_schema() - - payload = { - 'task': instructions, - 'structured_output_json': json.dumps(json_schema), - 'llm_model': 'gpt-4.1-mini', - 'max_agent_steps': 15, - 'enable_public_share': True, # Enable shareable execution URLs - **kwargs, - } - - response = _request_with_retry('post', f'{BASE_URL}/run-task', headers=HEADERS, json=payload) - - task_id = response.json()['id'] - print(f'โœ… Structured task created: {task_id}') - return task_id - - -def wait_for_structured_completion(task_id: str, max_wait_time: int = 300) -> dict[str, Any]: - """Wait for task completion and return the result.""" - print(f'โณ Waiting for structured output (max {max_wait_time}s)...') - - start_time = time.time() - - while True: - response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}/status', headers=HEADERS) - status = response.json() - elapsed = time.time() - start_time - - # Check for timeout - if elapsed > max_wait_time: - print(f'\rโฐ Task timeout after {max_wait_time}s - stopping wait' + ' ' * 30) - # Get final details before timeout - details_response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS) - details = details_response.json() - return details - - # Get step count from full details for better progress tracking - details_response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS) - details = details_response.json() - steps = len(details.get('steps', [])) - - # Build status message - if status == 'running': - status_msg = f'๐Ÿ“‹ Structured task | Step {steps} | โฑ๏ธ {elapsed:.0f}s | ๐Ÿ”„ Extracting...' - else: - status_msg = f'๐Ÿ“‹ Structured task | Step {steps} | โฑ๏ธ {elapsed:.0f}s | Status: {status}' - - # Clear line and show status - print(f'\r{status_msg:<80}', end='', flush=True) - - if status == 'finished': - print(f'\rโœ… Structured data extracted! ({steps} steps in {elapsed:.1f}s)' + ' ' * 20) - return details - - elif status in ['failed', 'stopped']: - print(f'\rโŒ Task {status} after {steps} steps' + ' ' * 30) - return details - - time.sleep(3) - - -def validate_and_display_output(output: str, schema_model: type[BaseModel]): - """ - Validate the JSON output against the schema and display results. - - Args: - output: Raw JSON string from the task - schema_model: Pydantic model for validation - """ - print('\n๐Ÿ“Š Structured Output Analysis') - print('=' * 40) - - try: - # Parse and validate the JSON - parsed_data = schema_model.model_validate_json(output) - print('โœ… JSON validation successful!') - - # Pretty print the structured data - print('\n๐Ÿ“‹ Parsed Data:') - print('-' * 20) - print(parsed_data.model_dump_json(indent=2)) - - # Display specific fields based on model type - if isinstance(parsed_data, NewsResponse): - print(f'\n๐Ÿ“ฐ Found {len(parsed_data.articles)} articles from {parsed_data.source_website}') - for i, article in enumerate(parsed_data.articles[:3], 1): - print(f'\n{i}. {article.title}') - print(f' Summary: {article.summary[:100]}...') - print(f' URL: {article.url}') - - elif isinstance(parsed_data, ProductInfo): - print(f'\n๐Ÿ›๏ธ Product: {parsed_data.name}') - print(f' Price: ${parsed_data.price}') - print(f' Rating: {parsed_data.rating}/5' if parsed_data.rating else ' Rating: N/A') - print(f' Status: {parsed_data.availability}') - - elif isinstance(parsed_data, CompanyInfo): - print(f'\n๐Ÿข Company: {parsed_data.name}') - print(f' Industry: {parsed_data.industry}') - print(f' Headquarters: {parsed_data.headquarters}') - if parsed_data.founded_year: - print(f' Founded: {parsed_data.founded_year}') - - return parsed_data - - except ValidationError as e: - print('โŒ JSON validation failed!') - print(f'Errors: {e}') - print(f'\nRaw output: {output[:500]}...') - return None - - except json.JSONDecodeError as e: - print('โŒ Invalid JSON format!') - print(f'Error: {e}') - print(f'\nRaw output: {output[:500]}...') - return None - - -def demo_news_extraction(): - """Demo: Extract structured news data.""" - print('\n๐Ÿ“ฐ Demo 1: News Article Extraction') - print('-' * 40) - - task = """ - Go to a major news website (like BBC, CNN, or Reuters) and extract information - about the top 3 news articles. For each article, get the title, summary, URL, - and any other available metadata. - """ - - task_id = create_structured_task(task, NewsResponse) - result = wait_for_structured_completion(task_id) - - if result.get('output'): - parsed_result = validate_and_display_output(result['output'], NewsResponse) - - # Show execution URLs - if result.get('live_url'): - print(f'\n๐Ÿ”— Live Preview: {result["live_url"]}') - if result.get('public_share_url'): - print(f'๐ŸŒ Share URL: {result["public_share_url"]}') - elif result.get('share_url'): - print(f'๐ŸŒ Share URL: {result["share_url"]}') - - return parsed_result - else: - print('โŒ No structured output received') - return None - - -def demo_product_extraction(): - """Demo: Extract structured product data.""" - print('\n๐Ÿ›๏ธ Demo 2: Product Information Extraction') - print('-' * 40) - - task = """ - Go to Amazon and search for 'wireless headphones'. Find the first product result - and extract detailed information including name, price, rating, availability, - and description. - """ - - task_id = create_structured_task(task, ProductInfo) - result = wait_for_structured_completion(task_id) - - if result.get('output'): - parsed_result = validate_and_display_output(result['output'], ProductInfo) - - # Show execution URLs - if result.get('live_url'): - print(f'\n๐Ÿ”— Live Preview: {result["live_url"]}') - if result.get('public_share_url'): - print(f'๐ŸŒ Share URL: {result["public_share_url"]}') - elif result.get('share_url'): - print(f'๐ŸŒ Share URL: {result["share_url"]}') - - return parsed_result - else: - print('โŒ No structured output received') - return None - - -def demo_company_extraction(): - """Demo: Extract structured company data.""" - print('\n๐Ÿข Demo 3: Company Information Extraction') - print('-' * 40) - - task = """ - Go to a financial website and look up information about Apple Inc. - Extract company details including name, stock symbol, market cap, - industry, headquarters, and founding year. - """ - - task_id = create_structured_task(task, CompanyInfo) - result = wait_for_structured_completion(task_id) - - if result.get('output'): - parsed_result = validate_and_display_output(result['output'], CompanyInfo) - - # Show execution URLs - if result.get('live_url'): - print(f'\n๐Ÿ”— Live Preview: {result["live_url"]}') - if result.get('public_share_url'): - print(f'๐ŸŒ Share URL: {result["public_share_url"]}') - elif result.get('share_url'): - print(f'๐ŸŒ Share URL: {result["share_url"]}') - - return parsed_result - else: - print('โŒ No structured output received') - return None - - -def main(): - """Demonstrate structured output extraction.""" - print('๐Ÿ“‹ Browser Use Cloud - Structured JSON Output') - print('=' * 50) - - print('๐ŸŽฏ Features:') - print('โ€ข Type-safe Pydantic schemas') - print('โ€ข Automatic JSON validation') - print('โ€ข Structured data extraction') - print('โ€ข Multiple output formats') - - try: - # Parse command line arguments - parser = argparse.ArgumentParser(description='Structured output extraction demo') - parser.add_argument('--demo', choices=['news', 'product', 'company', 'all'], default='news', help='Which demo to run') - args = parser.parse_args() - - print(f'\n๐Ÿ” Running {args.demo} demo(s)...') - - if args.demo == 'news': - demo_news_extraction() - elif args.demo == 'product': - demo_product_extraction() - elif args.demo == 'company': - demo_company_extraction() - elif args.demo == 'all': - demo_news_extraction() - demo_product_extraction() - demo_company_extraction() - - except requests.exceptions.RequestException as e: - print(f'โŒ API Error: {e}') - except Exception as e: - print(f'โŒ Error: {e}') - - -if __name__ == '__main__': - main() diff --git a/examples/cloud/04_proxy_usage.py b/examples/cloud/04_proxy_usage.py deleted file mode 100644 index 71dca7975..000000000 --- a/examples/cloud/04_proxy_usage.py +++ /dev/null @@ -1,331 +0,0 @@ -""" -Cloud Example 4: Proxy Usage ๐ŸŒ -=============================== - -This example demonstrates reliable proxy usage scenarios: -- Different country proxies for geo-restrictions -- IP address and location verification -- Region-specific content access (streaming, news) -- Search result localization by country -- Mobile/residential proxy benefits - -Perfect for: Geo-restricted content, location testing, regional analysis - -Cost: ~$0.08 (1 task + 6-8 steps with proxy enabled) -""" - -import argparse -import os -import time -from typing import Any - -import requests -from requests.exceptions import RequestException - -# Configuration -API_KEY = os.getenv('BROWSER_USE_API_KEY') -if not API_KEY: - raise ValueError( - 'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key' - ) - -BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1') -TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30')) -HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'} - - -def _request_with_retry(method: str, url: str, **kwargs) -> requests.Response: - """Make HTTP request with timeout and retry logic.""" - kwargs.setdefault('timeout', TIMEOUT) - - for attempt in range(3): - try: - response = requests.request(method, url, **kwargs) - response.raise_for_status() - return response - except RequestException as e: - if attempt == 2: # Last attempt - raise - sleep_time = 2**attempt - print(f'โš ๏ธ Request failed (attempt {attempt + 1}/3), retrying in {sleep_time}s: {e}') - time.sleep(sleep_time) - - raise RuntimeError('Unexpected error in retry logic') - - -def create_task_with_proxy(instructions: str, country_code: str = 'us') -> str: - """ - Create a task with proxy enabled from a specific country. - - Args: - instructions: Task description - country_code: Proxy country ('us', 'fr', 'it', 'jp', 'au', 'de', 'fi', 'ca') - - Returns: - task_id: Unique identifier for the created task - """ - print(f'๐ŸŒ Creating task with {country_code.upper()} proxy') - print(f'๐Ÿ“ Task: {instructions}') - - payload = { - 'task': instructions, - 'llm_model': 'gpt-4.1-mini', - # Proxy configuration - 'use_proxy': True, # Required for captcha solving - 'proxy_country_code': country_code, # Choose proxy location - # Standard settings - 'use_adblock': True, # Block ads for faster loading - 'highlight_elements': True, # Keep highlighting for visibility - 'max_agent_steps': 15, - # Enable sharing for viewing execution - 'enable_public_share': True, # Get shareable URLs - } - - response = _request_with_retry('post', f'{BASE_URL}/run-task', headers=HEADERS, json=payload) - - task_id = response.json()['id'] - print(f'โœ… Task created with {country_code.upper()} proxy: {task_id}') - return task_id - - -def test_ip_location(country_code: str) -> dict[str, Any]: - """Test IP address and location detection with proxy.""" - task = """ - Go to whatismyipaddress.com and tell me: - 1. The detected IP address - 2. The detected country/location - 3. The ISP/organization - 4. Any other location details shown - - Please be specific about what you see on the page. - """ - - task_id = create_task_with_proxy(task, country_code) - return wait_for_completion(task_id) - - -def test_geo_restricted_content(country_code: str) -> dict[str, Any]: - """Test access to geo-restricted content.""" - task = """ - Go to a major news website (like BBC, CNN, or local news) and check: - 1. What content is available - 2. Any geo-restriction messages - 3. Local/regional content differences - 4. Language or currency preferences shown - - Note any differences from what you might expect. - """ - - task_id = create_task_with_proxy(task, country_code) - return wait_for_completion(task_id) - - -def test_streaming_service_access(country_code: str) -> dict[str, Any]: - """Test access to region-specific streaming content.""" - task = """ - Go to a major streaming service website (like Netflix, YouTube, or BBC iPlayer) - and check what content or messaging appears. - - Report: - 1. What homepage content is shown - 2. Any geo-restriction messages or content differences - 3. Available content regions or language options - 4. Any pricing or availability differences - - Note: Don't try to log in, just observe the publicly available content. - """ - - task_id = create_task_with_proxy(task, country_code) - return wait_for_completion(task_id) - - -def test_search_results_by_location(country_code: str) -> dict[str, Any]: - """Test how search results vary by location.""" - task = """ - Go to Google and search for "best restaurants near me" or "local news". - - Report: - 1. What local results appear - 2. The detected location in search results - 3. Any location-specific content or ads - 4. Language preferences - - This will show how search results change based on proxy location. - """ - - task_id = create_task_with_proxy(task, country_code) - return wait_for_completion(task_id) - - -def wait_for_completion(task_id: str) -> dict[str, Any]: - """Wait for task completion and return results.""" - print(f'โณ Waiting for task {task_id} to complete...') - - start_time = time.time() - - while True: - response = _request_with_retry('get', f'{BASE_URL}/task/{task_id}', headers=HEADERS) - details = response.json() - - status = details['status'] - steps = len(details.get('steps', [])) - elapsed = time.time() - start_time - - # Build status message - if status == 'running': - status_msg = f'๐ŸŒ Proxy task | Step {steps} | โฑ๏ธ {elapsed:.0f}s | ๐Ÿค– Processing...' - else: - status_msg = f'๐ŸŒ Proxy task | Step {steps} | โฑ๏ธ {elapsed:.0f}s | Status: {status}' - - # Clear line and show status - print(f'\r{status_msg:<80}', end='', flush=True) - - if status == 'finished': - print(f'\rโœ… Task completed in {steps} steps! ({elapsed:.1f}s total)' + ' ' * 20) - return details - - elif status in ['failed', 'stopped']: - print(f'\rโŒ Task {status} after {steps} steps' + ' ' * 30) - return details - - time.sleep(3) - - -def demo_proxy_countries(): - """Demonstrate proxy usage across different countries.""" - print('\n๐ŸŒ Demo 1: Proxy Countries Comparison') - print('-' * 45) - - countries = [('us', 'United States'), ('de', 'Germany'), ('jp', 'Japan'), ('au', 'Australia')] - - results = {} - - for code, name in countries: - print(f'\n๐ŸŒ Testing {name} ({code.upper()}) proxy:') - print('=' * 40) - - result = test_ip_location(code) - results[code] = result - - if result.get('output'): - print(f'๐Ÿ“ Location Result: {result["output"][:200]}...') - - # Show execution URLs - if result.get('live_url'): - print(f'๐Ÿ”— Live Preview: {result["live_url"]}') - if result.get('public_share_url'): - print(f'๐ŸŒ Share URL: {result["public_share_url"]}') - elif result.get('share_url'): - print(f'๐ŸŒ Share URL: {result["share_url"]}') - - print('-' * 40) - time.sleep(2) # Brief pause between tests - - # Summary comparison - print('\n๐Ÿ“Š Proxy Location Summary:') - print('=' * 30) - for code, result in results.items(): - status = result.get('status', 'unknown') - print(f'{code.upper()}: {status}') - - -def demo_geo_restrictions(): - """Demonstrate geo-restriction bypass.""" - print('\n๐Ÿšซ Demo 2: Geo-Restriction Testing') - print('-' * 40) - - # Test from different locations - locations = [('us', 'US content'), ('de', 'European content')] - - for code, description in locations: - print(f'\n๐ŸŒ Testing {description} with {code.upper()} proxy:') - result = test_geo_restricted_content(code) - - if result.get('output'): - print(f'๐Ÿ“ฐ Content Access: {result["output"][:200]}...') - - time.sleep(2) - - -def demo_streaming_access(): - """Demonstrate streaming service access with different proxies.""" - print('\n๐Ÿ“บ Demo 3: Streaming Service Access') - print('-' * 40) - - locations = [('us', 'US'), ('de', 'Germany')] - - for code, name in locations: - print(f'\n๐ŸŒ Testing streaming access from {name}:') - result = test_streaming_service_access(code) - - if result.get('output'): - print(f'๐Ÿ“บ Access Result: {result["output"][:200]}...') - - time.sleep(2) - - -def demo_search_localization(): - """Demonstrate search result localization.""" - print('\n๐Ÿ” Demo 4: Search Localization') - print('-' * 35) - - locations = [('us', 'US'), ('de', 'Germany')] - - for code, name in locations: - print(f'\n๐ŸŒ Testing search results from {name}:') - result = test_search_results_by_location(code) - - if result.get('output'): - print(f'๐Ÿ” Search Results: {result["output"][:200]}...') - - time.sleep(2) - - -def main(): - """Demonstrate comprehensive proxy usage.""" - print('๐ŸŒ Browser Use Cloud - Proxy Usage Examples') - print('=' * 50) - - print('๐ŸŽฏ Proxy Benefits:') - print('โ€ข Bypass geo-restrictions') - print('โ€ข Test location-specific content') - print('โ€ข Access region-locked websites') - print('โ€ข Mobile/residential IP addresses') - print('โ€ข Verify IP geolocation') - - print('\n๐ŸŒ Available Countries:') - countries = ['๐Ÿ‡บ๐Ÿ‡ธ US', '๐Ÿ‡ซ๐Ÿ‡ท France', '๐Ÿ‡ฎ๐Ÿ‡น Italy', '๐Ÿ‡ฏ๐Ÿ‡ต Japan', '๐Ÿ‡ฆ๐Ÿ‡บ Australia', '๐Ÿ‡ฉ๐Ÿ‡ช Germany', '๐Ÿ‡ซ๐Ÿ‡ฎ Finland', '๐Ÿ‡จ๐Ÿ‡ฆ Canada'] - print(' โ€ข '.join(countries)) - - try: - # Parse command line arguments - parser = argparse.ArgumentParser(description='Proxy usage examples') - parser.add_argument( - '--demo', choices=['countries', 'geo', 'streaming', 'search', 'all'], default='countries', help='Which demo to run' - ) - args = parser.parse_args() - - print(f'\n๐Ÿ” Running {args.demo} demo(s)...') - - if args.demo == 'countries': - demo_proxy_countries() - elif args.demo == 'geo': - demo_geo_restrictions() - elif args.demo == 'streaming': - demo_streaming_access() - elif args.demo == 'search': - demo_search_localization() - elif args.demo == 'all': - demo_proxy_countries() - demo_geo_restrictions() - demo_streaming_access() - demo_search_localization() - - except requests.exceptions.RequestException as e: - print(f'โŒ API Error: {e}') - except Exception as e: - print(f'โŒ Error: {e}') - - -if __name__ == '__main__': - main() diff --git a/examples/cloud/05_search_api.py b/examples/cloud/05_search_api.py deleted file mode 100644 index 64c01dc04..000000000 --- a/examples/cloud/05_search_api.py +++ /dev/null @@ -1,355 +0,0 @@ -""" -Cloud Example 5: Search API (Beta) ๐Ÿ” -===================================== - -This example demonstrates the Browser Use Search API (BETA): -- Simple search: Search Google and extract from multiple results -- URL search: Extract specific content from a target URL -- Deep navigation through websites (depth parameter) -- Real-time content extraction vs cached results - -Perfect for: Content extraction, research, competitive analysis -""" - -import argparse -import asyncio -import json -import os -import time -from typing import Any - -import aiohttp - -# Configuration -API_KEY = os.getenv('BROWSER_USE_API_KEY') -if not API_KEY: - raise ValueError( - 'Please set BROWSER_USE_API_KEY environment variable. You can also create an API key at https://cloud.browser-use.com/new-api-key' - ) - -BASE_URL = os.getenv('BROWSER_USE_BASE_URL', 'https://api.browser-use.com/api/v1') -TIMEOUT = int(os.getenv('BROWSER_USE_TIMEOUT', '30')) -HEADERS = {'Authorization': f'Bearer {API_KEY}', 'Content-Type': 'application/json'} - - -async def simple_search(query: str, max_websites: int = 5, depth: int = 2) -> dict[str, Any]: - """ - Search Google and extract content from multiple top results. - - Args: - query: Search query to process - max_websites: Number of websites to process (1-10) - depth: How deep to navigate (2-5) - - Returns: - Dictionary with results from multiple websites - """ - # Validate input parameters - max_websites = max(1, min(max_websites, 10)) # Clamp to 1-10 - depth = max(2, min(depth, 5)) # Clamp to 2-5 - - start_time = time.time() - - print(f"๐Ÿ” Simple Search: '{query}'") - print(f'๐Ÿ“Š Processing {max_websites} websites at depth {depth}') - print(f'๐Ÿ’ฐ Estimated cost: {depth * max_websites}ยข') - - payload = {'query': query, 'max_websites': max_websites, 'depth': depth} - - timeout = aiohttp.ClientTimeout(total=TIMEOUT) - connector = aiohttp.TCPConnector(limit=10) # Limit concurrent connections - - async with aiohttp.ClientSession(timeout=timeout, connector=connector) as session: - async with session.post(f'{BASE_URL}/simple-search', json=payload, headers=HEADERS) as response: - elapsed = time.time() - start_time - if response.status == 200: - try: - result = await response.json() - print(f'โœ… Found results from {len(result.get("results", []))} websites in {elapsed:.1f}s') - return result - except (aiohttp.ContentTypeError, json.JSONDecodeError) as e: - error_text = await response.text() - print(f'โŒ Invalid JSON response: {e} (after {elapsed:.1f}s)') - return {'error': 'Invalid JSON', 'details': error_text} - else: - error_text = await response.text() - print(f'โŒ Search failed: {response.status} - {error_text} (after {elapsed:.1f}s)') - return {'error': f'HTTP {response.status}', 'details': error_text} - - -async def search_url(url: str, query: str, depth: int = 2) -> dict[str, Any]: - """ - Extract specific content from a target URL. - - Args: - url: Target URL to extract from - query: What specific content to look for - depth: How deep to navigate (2-5) - - Returns: - Dictionary with extracted content - """ - # Validate input parameters - depth = max(2, min(depth, 5)) # Clamp to 2-5 - - start_time = time.time() - - print(f'๐ŸŽฏ URL Search: {url}') - print(f"๐Ÿ” Looking for: '{query}'") - print(f'๐Ÿ“Š Navigation depth: {depth}') - print(f'๐Ÿ’ฐ Estimated cost: {depth}ยข') - - payload = {'url': url, 'query': query, 'depth': depth} - - timeout = aiohttp.ClientTimeout(total=TIMEOUT) - connector = aiohttp.TCPConnector(limit=10) # Limit concurrent connections - - async with aiohttp.ClientSession(timeout=timeout, connector=connector) as session: - async with session.post(f'{BASE_URL}/search-url', json=payload, headers=HEADERS) as response: - elapsed = time.time() - start_time - if response.status == 200: - try: - result = await response.json() - print(f'โœ… Extracted content from {result.get("url", "website")} in {elapsed:.1f}s') - return result - except (aiohttp.ContentTypeError, json.JSONDecodeError) as e: - error_text = await response.text() - print(f'โŒ Invalid JSON response: {e} (after {elapsed:.1f}s)') - return {'error': 'Invalid JSON', 'details': error_text} - else: - error_text = await response.text() - print(f'โŒ URL search failed: {response.status} - {error_text} (after {elapsed:.1f}s)') - return {'error': f'HTTP {response.status}', 'details': error_text} - - -def display_simple_search_results(results: dict[str, Any]): - """Display simple search results in a readable format.""" - if 'error' in results: - print(f'โŒ Error: {results["error"]}') - return - - websites = results.get('results', []) - - print(f'\n๐Ÿ“‹ Search Results ({len(websites)} websites)') - print('=' * 50) - - for i, site in enumerate(websites, 1): - url = site.get('url', 'Unknown URL') - content = site.get('content', 'No content') - - print(f'\n{i}. ๐ŸŒ {url}') - print('-' * 40) - - # Show first 300 chars of content - if len(content) > 300: - print(f'{content[:300]}...') - print(f'[Content truncated - {len(content)} total characters]') - else: - print(content) - - # Show execution URLs if available - if results.get('live_url'): - print(f'\n๐Ÿ”— Live Preview: {results["live_url"]}') - if results.get('public_share_url'): - print(f'๐ŸŒ Share URL: {results["public_share_url"]}') - elif results.get('share_url'): - print(f'๐ŸŒ Share URL: {results["share_url"]}') - - -def display_url_search_results(results: dict[str, Any]): - """Display URL search results in a readable format.""" - if 'error' in results: - print(f'โŒ Error: {results["error"]}') - return - - url = results.get('url', 'Unknown URL') - content = results.get('content', 'No content') - - print(f'\n๐Ÿ“„ Extracted Content from: {url}') - print('=' * 60) - print(content) - - # Show execution URLs if available - if results.get('live_url'): - print(f'\n๐Ÿ”— Live Preview: {results["live_url"]}') - if results.get('public_share_url'): - print(f'๐ŸŒ Share URL: {results["public_share_url"]}') - elif results.get('share_url'): - print(f'๐ŸŒ Share URL: {results["share_url"]}') - - -async def demo_news_search(): - """Demo: Search for latest news across multiple sources.""" - print('\n๐Ÿ“ฐ Demo 1: Latest News Search') - print('-' * 35) - - demo_start = time.time() - query = 'latest developments in artificial intelligence 2024' - results = await simple_search(query, max_websites=4, depth=2) - demo_elapsed = time.time() - demo_start - - display_simple_search_results(results) - print(f'\nโฑ๏ธ Total demo time: {demo_elapsed:.1f}s') - - return results - - -async def demo_competitive_analysis(): - """Demo: Analyze competitor websites.""" - print('\n๐Ÿข Demo 2: Competitive Analysis') - print('-' * 35) - - query = 'browser automation tools comparison features pricing' - results = await simple_search(query, max_websites=3, depth=3) - display_simple_search_results(results) - - return results - - -async def demo_deep_website_analysis(): - """Demo: Deep analysis of a specific website.""" - print('\n๐ŸŽฏ Demo 3: Deep Website Analysis') - print('-' * 35) - - demo_start = time.time() - url = 'https://docs.browser-use.com' - query = 'Browser Use features, pricing, and API capabilities' - results = await search_url(url, query, depth=3) - demo_elapsed = time.time() - demo_start - - display_url_search_results(results) - print(f'\nโฑ๏ธ Total demo time: {demo_elapsed:.1f}s') - - return results - - -async def demo_product_research(): - """Demo: Product research and comparison.""" - print('\n๐Ÿ›๏ธ Demo 4: Product Research') - print('-' * 30) - - query = 'best wireless headphones 2024 reviews comparison' - results = await simple_search(query, max_websites=5, depth=2) - display_simple_search_results(results) - - return results - - -async def demo_real_time_vs_cached(): - """Demo: Show difference between real-time and cached results.""" - print('\nโšก Demo 5: Real-time vs Cached Data') - print('-' * 40) - - print('๐Ÿ”„ Browser Use Search API benefits:') - print('โ€ข Actually browses websites like a human') - print('โ€ข Gets live, current data (not cached)') - print('โ€ข Navigates deep into sites via clicks') - print('โ€ข Handles JavaScript and dynamic content') - print('โ€ข Accesses pages requiring navigation') - - # Example with live data - query = 'current Bitcoin price USD live' - results = await simple_search(query, max_websites=3, depth=2) - - print('\n๐Ÿ’ฐ Live Bitcoin Price Search Results:') - display_simple_search_results(results) - - return results - - -async def demo_search_depth_comparison(): - """Demo: Compare different search depths.""" - print('\n๐Ÿ“Š Demo 6: Search Depth Comparison') - print('-' * 40) - - url = 'https://news.ycombinator.com' - query = 'trending technology discussions' - - depths = [2, 3, 4] - results = {} - - for depth in depths: - print(f'\n๐Ÿ” Testing depth {depth}:') - result = await search_url(url, query, depth) - results[depth] = result - - if 'content' in result: - content_length = len(result['content']) - print(f'๐Ÿ“ Content length: {content_length} characters') - - # Brief pause between requests - await asyncio.sleep(1) - - # Summary - print('\n๐Ÿ“Š Depth Comparison Summary:') - print('-' * 30) - for depth, result in results.items(): - if 'content' in result: - length = len(result['content']) - print(f'Depth {depth}: {length} characters') - else: - print(f'Depth {depth}: Error or no content') - - return results - - -async def main(): - """Demonstrate comprehensive Search API usage.""" - print('๐Ÿ” Browser Use Cloud - Search API (BETA)') - print('=' * 45) - - print('โš ๏ธ Note: This API is in BETA and may change') - print() - print('๐ŸŽฏ Search API Features:') - print('โ€ข Real-time website browsing (not cached)') - print('โ€ข Deep navigation through multiple pages') - print('โ€ข Dynamic content and JavaScript handling') - print('โ€ข Multiple result aggregation') - print('โ€ข Cost-effective content extraction') - - print('\n๐Ÿ’ฐ Pricing:') - print('โ€ข Simple Search: 1ยข ร— depth ร— websites') - print('โ€ข URL Search: 1ยข ร— depth') - print('โ€ข Example: depth=2, 5 websites = 10ยข') - - try: - # Parse command line arguments - parser = argparse.ArgumentParser(description='Search API (BETA) examples') - parser.add_argument( - '--demo', - choices=['news', 'competitive', 'deep', 'product', 'realtime', 'depth', 'all'], - default='news', - help='Which demo to run', - ) - args = parser.parse_args() - - print(f'\n๐Ÿ” Running {args.demo} demo(s)...') - - if args.demo == 'news': - await demo_news_search() - elif args.demo == 'competitive': - await demo_competitive_analysis() - elif args.demo == 'deep': - await demo_deep_website_analysis() - elif args.demo == 'product': - await demo_product_research() - elif args.demo == 'realtime': - await demo_real_time_vs_cached() - elif args.demo == 'depth': - await demo_search_depth_comparison() - elif args.demo == 'all': - await demo_news_search() - await demo_competitive_analysis() - await demo_deep_website_analysis() - await demo_product_research() - await demo_real_time_vs_cached() - await demo_search_depth_comparison() - - except aiohttp.ClientError as e: - print(f'โŒ Network Error: {e}') - except Exception as e: - print(f'โŒ Error: {e}') - - -if __name__ == '__main__': - asyncio.run(main()) diff --git a/examples/cloud/README.md b/examples/cloud/README.md index e2aba21d0..6c2024ac1 100644 --- a/examples/cloud/README.md +++ b/examples/cloud/README.md @@ -1,137 +1,28 @@ -# Browser Use Cloud Examples ๐Ÿš€ +# Browser Use Cloud API V4 -Welcome to the Browser Use Cloud examples! This folder contains progressively complex examples to help you get started with the Browser Use Cloud API quickly and efficiently. +Run one browser task through the current Cloud API. The example creates a run, polls the lightweight status endpoint, and fetches the result once the run is terminal. -## ๐Ÿ“‹ Prerequisites +## Setup -1. **API Key**: Get your API key from [cloud.browser-use.com](https://cloud.browser-use.com/new-api-key) -2. **Python Environment**: Python 3.11+ with dependencies -3. **Environment Variables**: Configure your API settings - -### Quick Setup +From the repository root: ```bash -# Create virtual environment and install dependencies (from project root) -uv venv --python 3.11 -source .venv/bin/activate # On Windows: .venv\Scripts\activate uv sync - -# Set environment variables -export BROWSER_USE_API_KEY="your_api_key_here" -export BROWSER_USE_BASE_URL="https://api.browser-use.com/api/v1" # Optional -export BROWSER_USE_TIMEOUT="30" # Optional: request timeout in seconds - -# Or use .env file (recommended) cp examples/cloud/env.example .env -# Edit .env with your values - -# Run examples from project root -python examples/cloud/01_basic_task.py +# Add your API key to .env +uv run python examples/cloud/01_basic_task.py ``` -## ๐ŸŽฏ Examples Overview +Create an API key at [cloud.browser-use.com/new-api-key](https://cloud.browser-use.com/new-api-key). -### ๐Ÿš€ Easy Cloud Setup Examples +## V4 request flow -- **[01_basic_task.py](./01_basic_task.py)** - Your first cloud task (start here!) -- **[02_fast_mode_gemini.py](./02_fast_mode_gemini.py)** - โšก Ultra-fast mode with Gemini Flash & Fireship humor -- **[03_structured_output.py](./03_structured_output.py)** - Get structured JSON responses -- **[04_proxy_usage.py](./04_proxy_usage.py)** - ๐ŸŒ Proxy for geo-restrictions & captcha solving -- **[05_search_api.py](./05_search_api.py)** - ๐Ÿ” Search API for content extraction (BETA) +The example uses the three endpoints needed for a basic run: -## ๐Ÿ’ฐ Cost Optimization Tips +1. `POST /api/v4/runs` +2. `GET /api/v4/runs/{run_id}/status` until the run is `completed`, `failed`, or `cancelled` +3. `GET /api/v4/runs/{run_id}` for the result, error, and cost -1. **Use Gemini Flash** for fastest/cheapest execution ($0.01/step) -2. **Disable proxy** when not needed for captcha solving -3. **Disable element highlighting** for better performance -4. **Set max_agent_steps** to prevent runaway costs -5. **Use structured output** to reduce parsing overhead -6. **Add timeouts and retries** for reliability in production -7. **Use domain restrictions** when working with secrets +Authentication uses the `X-Browser-Use-API-Key` header. See the live [V4 OpenAPI specification](https://api.browser-use.com/api/v4/openapi.json) for optional models, browser settings, sessions, files, secrets, and judge settings. -## ๐ŸŽจ Fast Mode Configuration - -For maximum speed and cost efficiency: - -```python -{ - "llm_model": "gemini-2.5-flash", - "use_proxy": False, - "highlight_elements": False, - "use_adblock": True, - "max_agent_steps": 50 -} -``` - -## ๐Ÿ” Security & Advanced Features - -### Using Proxy -```python -{ - "use_proxy": True, - "proxy_country_code": "us", # 'us', 'fr', 'it', 'jp', 'au', 'de', 'fi', 'ca' -} -``` - -### Passing Secrets Securely -```python -{ - "secrets": { - "username": "your_username", - "password": "your_password", - "api_key": "your_api_key" - }, - "allowed_domains": ["*.yoursite.com"] # Recommended with secrets -} -``` - -## ๐Ÿ” Search API (BETA) - -The Search API extracts content by actually browsing websites (not cached results): - -### Simple Search (Multi-site) -```python -# Cost: 1ยข ร— depth ร— websites -{ - "query": "latest AI news", - "max_websites": 5, - "depth": 2 -} -``` - -### URL Search (Single site) -```python -# Cost: 1ยข ร— depth -{ - "url": "https://example.com", - "query": "pricing information", - "depth": 3 -} -``` - -## ๐Ÿ”— Quick Links - -- [Cloud API Documentation](https://docs.browser-use.com/cloud) -- [API Reference](https://docs.browser-use.com/api-reference) -- [Pricing](https://cloud.browser-use.com/billing) -- [Discord Community](https://link.browser-use.com/discord) - -## ๐Ÿ”ง Production Best Practices - -- **Timeouts**: All examples include 30-second timeouts with retry logic -- **Error Handling**: Comprehensive error catching and status code validation -- **Security**: Use environment variables, domain restrictions with secrets -- **Reliability**: Built-in retries for network issues and rate limits -- **Automation**: CLI arguments instead of interactive prompts for CI/CD - -## ๐Ÿ†˜ Support - -Need help? - -- ๐Ÿ“ง Email: support@browser-use.com -- ๐Ÿ’ฌ Discord: [Join our community](https://link.browser-use.com/discord) -- ๐Ÿ“– Docs: - ---- - -**๐Ÿ’ก Pro Tip**: Start with `01_basic_task.py` and work your way up. Each example builds on the previous ones! +Review usage and credits in [Cloud billing](https://cloud.browser-use.com/billing). diff --git a/examples/cloud/env.example b/examples/cloud/env.example index 3fbbcda19..73f1e2ebf 100644 --- a/examples/cloud/env.example +++ b/examples/cloud/env.example @@ -1,21 +1,5 @@ -# Browser Use Cloud API Configuration -# Copy this file to .env and fill in your values - -# Required: Your Browser Use Cloud API key -# Get it from: https://cloud.browser-use.com/new-api-key +# Create a key at https://cloud.browser-use.com/new-api-key BROWSER_USE_API_KEY=your_api_key_here -# Optional: Custom API base URL (for enterprise installations) -# BROWSER_USE_BASE_URL=https://api.browser-use.com/api/v1 - -# Optional: Default model preference -# BROWSER_USE_DEFAULT_MODEL=gemini-2.5-flash - -# Optional: Cost limits -# BROWSER_USE_MAX_COST_PER_TASK=5.0 - -# Optional: Request timeout (seconds) -# BROWSER_USE_TIMEOUT=30 - -# Optional: Logging configuration -# LOG_LEVEL=INFO +# Optional: override the API root for an enterprise installation. +# BROWSER_USE_API_URL=https://api.browser-use.com/api/v4 From 2391a244b3bb0007bd6d1b8b9fe6d638ba27b680 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 27 Aug 2026 21:36:44 -0700 Subject: [PATCH 58/69] docs(cloud): cancel runs after example timeout --- examples/cloud/01_basic_task.py | 10 +++++++++- examples/cloud/README.md | 2 +- examples/cloud/env.example | 3 +++ 3 files changed, 13 insertions(+), 2 deletions(-) diff --git a/examples/cloud/01_basic_task.py b/examples/cloud/01_basic_task.py index 2e88ff668..1be9d67f7 100644 --- a/examples/cloud/01_basic_task.py +++ b/examples/cloud/01_basic_task.py @@ -14,6 +14,7 @@ API_KEY = os.getenv('BROWSER_USE_API_KEY') if not API_KEY: raise RuntimeError('Set BROWSER_USE_API_KEY or add it to .env') +RUN_TIMEOUT_SECONDS = float(os.getenv('BROWSER_USE_RUN_TIMEOUT', '900')) HEADERS = {'X-Browser-Use-API-Key': API_KEY} TERMINAL_STATUSES = {'completed', 'failed', 'cancelled'} @@ -24,7 +25,9 @@ def create_run(task: str) -> str: return response.json()['id'] -def wait_for_run(run_id: str, poll_seconds: float = 2) -> dict[str, Any]: +def wait_for_run(run_id: str, poll_seconds: float = 2, timeout_seconds: float = RUN_TIMEOUT_SECONDS) -> dict[str, Any]: + deadline = time.monotonic() + timeout_seconds + while True: response = requests.get(f'{API_URL}/runs/{run_id}/status', headers=HEADERS, timeout=30) response.raise_for_status() @@ -34,6 +37,11 @@ def wait_for_run(run_id: str, poll_seconds: float = 2) -> dict[str, Any]: if status in TERMINAL_STATUSES: break + if time.monotonic() >= deadline: + response = requests.post(f'{API_URL}/runs/{run_id}/cancel', headers=HEADERS, timeout=30) + response.raise_for_status() + raise TimeoutError(f'Cancelled run {run_id} after {timeout_seconds:g} seconds') + time.sleep(poll_seconds) response = requests.get(f'{API_URL}/runs/{run_id}', headers=HEADERS, timeout=30) diff --git a/examples/cloud/README.md b/examples/cloud/README.md index 6c2024ac1..adea218ed 100644 --- a/examples/cloud/README.md +++ b/examples/cloud/README.md @@ -1,6 +1,6 @@ # Browser Use Cloud API V4 -Run one browser task through the current Cloud API. The example creates a run, polls the lightweight status endpoint, and fetches the result once the run is terminal. +Run one browser task through the current Cloud API. The example creates a run, polls the lightweight status endpoint, and fetches the result once the run is terminal. It cancels a run that exceeds the configurable 15-minute wait limit. ## Setup diff --git a/examples/cloud/env.example b/examples/cloud/env.example index 73f1e2ebf..32f021455 100644 --- a/examples/cloud/env.example +++ b/examples/cloud/env.example @@ -3,3 +3,6 @@ BROWSER_USE_API_KEY=your_api_key_here # Optional: override the API root for an enterprise installation. # BROWSER_USE_API_URL=https://api.browser-use.com/api/v4 + +# Optional: cancel a run if it has not finished after this many seconds. +# BROWSER_USE_RUN_TIMEOUT=900 From 1359fb22cbce66d702f744a916d62146fc030358 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 27 Aug 2026 21:40:42 -0700 Subject: [PATCH 59/69] docs(cloud): validate example timeout setting --- examples/cloud/01_basic_task.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/examples/cloud/01_basic_task.py b/examples/cloud/01_basic_task.py index 1be9d67f7..ce2054245 100644 --- a/examples/cloud/01_basic_task.py +++ b/examples/cloud/01_basic_task.py @@ -2,6 +2,7 @@ import os import time +from math import isfinite from typing import Any import requests @@ -14,7 +15,12 @@ API_KEY = os.getenv('BROWSER_USE_API_KEY') if not API_KEY: raise RuntimeError('Set BROWSER_USE_API_KEY or add it to .env') -RUN_TIMEOUT_SECONDS = float(os.getenv('BROWSER_USE_RUN_TIMEOUT', '900')) +try: + RUN_TIMEOUT_SECONDS = float(os.getenv('BROWSER_USE_RUN_TIMEOUT', '900')) +except ValueError as error: + raise RuntimeError('BROWSER_USE_RUN_TIMEOUT must be a positive number of seconds') from error +if not isfinite(RUN_TIMEOUT_SECONDS) or RUN_TIMEOUT_SECONDS <= 0: + raise RuntimeError('BROWSER_USE_RUN_TIMEOUT must be a positive number of seconds') HEADERS = {'X-Browser-Use-API-Key': API_KEY} TERMINAL_STATUSES = {'completed', 'failed', 'cancelled'} From 295a57d9579bf61c295cf336e55aaf21a0095c30 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 27 Aug 2026 10:19:43 -0700 Subject: [PATCH 60/69] docs(cloud): add API v4 skill reference Signed-off-by: MagMueller --- skills/cloud/SKILL.md | 11 +- skills/cloud/references/api-v4.md | 134 ++++++++++++++++++ skills/cloud/references/quickstart.md | 4 + .../ci/test_browser_use_skill_install_docs.py | 11 ++ 4 files changed, 157 insertions(+), 3 deletions(-) create mode 100644 skills/cloud/references/api-v4.md diff --git a/skills/cloud/SKILL.md b/skills/cloud/SKILL.md index d96b53777..01dfea136 100644 --- a/skills/cloud/SKILL.md +++ b/skills/cloud/SKILL.md @@ -3,7 +3,7 @@ name: cloud description: > Documentation reference for using Browser Use Cloud โ€” the hosted API and SDK for browser automation. Use this skill whenever the user needs - help with the Cloud REST API (v2 or v3), browser-use-sdk (Python or + help with the Cloud REST API (v2, v3, or v4), browser-use-sdk (Python or TypeScript), X-Browser-Use-API-Key authentication, cloud sessions, browser profiles, profile sync, CDP WebSocket connections, stealth browsers, residential proxies, CAPTCHA handling, webhooks, workspaces, @@ -25,7 +25,8 @@ Read the relevant file based on what the user needs. | Topic | Read | |-------|------| -| Setup, first task, pricing, FAQ | `references/quickstart.md` | +| Current v4 setup, first run, sessions, workspaces, browsers | `references/api-v4.md` | +| Legacy v2 setup, pricing, FAQ | `references/quickstart.md` | | v2 REST API: all 30 endpoints, cURL examples, schemas | `references/api-v2.md` | | v3 BU Agent API: sessions, messages, files, workspaces | `references/api-v3.md` | | Sessions, profiles, auth strategies, 1Password | `references/sessions.md` | @@ -43,13 +44,17 @@ Read the relevant file based on what the user needs. ## Critical Notes -- Cloud API base URL: `https://api.browser-use.com/api/v2/` (v2) or `https://api.browser-use.com/api/v3` (v3) +- Use v4 for new hosted-agent integrations. Keep v2 or v3 only when maintaining an existing integration or using a resource not yet wrapped by the v4 SDK. +- Cloud API base URL: `https://api.browser-use.com/api/v2/` (v2), `https://api.browser-use.com/api/v3` (v3), or `https://api.browser-use.com/api/v4` (v4) - Auth header: `X-Browser-Use-API-Key: ` - Get API key: https://cloud.browser-use.com/new-api-key - Set env var: `BROWSER_USE_API_KEY=` - Cloud SDK: `uv pip install browser-use-sdk` (Python) or `npm install browser-use-sdk` (TypeScript) - Python v2: `from browser_use_sdk import AsyncBrowserUse` - Python v3: `from browser_use_sdk.v3 import AsyncBrowserUse` +- Python v4: `from browser_use_sdk.v4 import BrowserUse` or `AsyncBrowserUse` - TypeScript v2: `import { BrowserUse } from "browser-use-sdk"` - TypeScript v3: `import { BrowserUse } from "browser-use-sdk/v3"` +- TypeScript v4: `import { BrowserUse } from "browser-use-sdk/v4"` +- Browser management is available at the v4 REST `/browsers` resource, but its SDK wrapper still uses the explicit v3 namespace. Always stop a browser explicitly; closing CDP does not stop billing. - CDP WebSocket: `wss://connect.browser-use.com?apiKey=KEY&proxyCountryCode=us` diff --git a/skills/cloud/references/api-v4.md b/skills/cloud/references/api-v4.md new file mode 100644 index 000000000..64180974f --- /dev/null +++ b/skills/cloud/references/api-v4.md @@ -0,0 +1,134 @@ +# API v4: Hosted Agent Runs + +Use v4 for new hosted-agent integrations. A **run** is one agent turn, a +**session** is the conversation shared by follow-up runs, and a **workspace** +is the persistent filesystem that can be reused across sessions. + +- REST base: `https://api.browser-use.com/api/v4` +- Auth header: `X-Browser-Use-API-Key: ` +- Python: `from browser_use_sdk.v4 import BrowserUse` +- TypeScript: `import { BrowserUse } from "browser-use-sdk/v4"` + +## First Run + +### Python + +```python +from browser_use_sdk.v4 import BrowserUse + +with BrowserUse() as client: + created = client.runs.create("Find the top Hacker News story") + run = client.runs.wait_for_completion(created.id) + print(run.result) +``` + +### TypeScript + +```typescript +import { BrowserUse } from "browser-use-sdk/v4"; + +const client = new BrowserUse(); +const created = await client.runs.create({ + task: "Find the top Hacker News story", +}); +const run = await client.runs.waitForCompletion(created.id); +console.log(run.result); +``` + +### REST + +Create the run, poll the lightweight status route, then fetch the full result +only after the status is `completed`, `failed`, or `cancelled`: + +```bash +curl -X POST https://api.browser-use.com/api/v4/runs \ + -H "X-Browser-Use-API-Key: $BROWSER_USE_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"task":"Find the top Hacker News story"}' + +curl https://api.browser-use.com/api/v4/runs/RUN_ID/status \ + -H "X-Browser-Use-API-Key: $BROWSER_USE_API_KEY" + +curl https://api.browser-use.com/api/v4/runs/RUN_ID \ + -H "X-Browser-Use-API-Key: $BROWSER_USE_API_KEY" +``` + +Do not repeatedly poll the full run resource. The SDK wait helpers use the +status route and fetch the full run once at the end. + +## Sessions and Follow-ups + +Every new run implicitly creates a session. Reuse its session ID to continue +the same conversation: + +```python +from browser_use_sdk.v4 import BrowserUse + +with BrowserUse() as client: + first = client.runs.create("Open Hacker News") + client.runs.wait_for_completion(first.id) + + follow_up = client.runs.create( + "Now summarize the top story", + session_id=first.session_id, + ) + result = client.runs.wait_for_completion(follow_up.id) +``` + +For a busy session, queue a next turn with +`client.sessions.send_message(session_id, text)`. Pass `interrupt=True` only +when the active run should be cancelled so the new message can start. The REST +equivalent is `POST /sessions/{session_id}/queue` with `text` and optional +`interrupt`. + +## Workspaces and Files + +A workspace persists files independently of a session. Upload a local file, +then attach its returned file ID to a run: + +```python +from browser_use_sdk.v4 import BrowserUse + +with BrowserUse() as client: + workspace = client.workspaces.create(name="research") + uploaded = client.workspaces.upload(workspace.id, "people.csv") + + run = client.runs.create( + "Read the CSV and save a report", + workspace_id=workspace.id, + attached_file_ids=[uploaded[0].id], + ) +``` + +Attachments are run-scoped. Reusing a workspace does not automatically attach +every file in it. List generated files with `client.workspaces.files(workspace.id)`; +presigned download URLs expire after 60 seconds, so request them immediately +before downloading. + +## Direct Browser Control + +The v4 REST API can create a browser for direct CDP control: + +1. `POST /browsers` returns the browser `id` (its session ID) and `cdpUrl`. +2. Connect Browser Use, Playwright, Puppeteer, or Selenium to `cdpUrl`. +3. `PATCH /browsers/{session_id}` with `{"action":"stop"}` stops the browser; + replace `session_id` with the returned `id`. + +Closing a CDP client does not stop the cloud browser or its billing. The +browser-management SDK wrapper currently uses the explicit v3 namespace; use +`browser_use_sdk.v3` or `browser-use-sdk/v3` for that resource, or call the v4 +REST endpoint directly. + +## Resource Map + +| Resource | Common operations | +|----------|-------------------| +| Runs | create, list, get, status, events, cancel, attachments | +| Sessions | list, get, queue messages, inspect/remove queued messages, purge | +| Workspaces | create, get, update, archive, size, upload/list/delete files | +| Browsers (REST) | create, inspect, stop | + +For the complete current contract, use: + +- Docs: https://docs.browser-use.com/cloud/api-v4 +- OpenAPI: https://docs.browser-use.com/cloud/openapi/v4.json diff --git a/skills/cloud/references/quickstart.md b/skills/cloud/references/quickstart.md index 69a322af4..2f6c0c6a3 100644 --- a/skills/cloud/references/quickstart.md +++ b/skills/cloud/references/quickstart.md @@ -1,5 +1,9 @@ # Cloud Quickstart, Pricing & FAQ +> This page keeps the v2 quickstart for existing integrations. For a new +> hosted-agent integration, use [API v4](api-v4.md). V4 is the current API and +> SDK path. + ## Table of Contents - [Overview](#overview) - [Setup](#setup) diff --git a/tests/ci/test_browser_use_skill_install_docs.py b/tests/ci/test_browser_use_skill_install_docs.py index 41da21a81..754e991ef 100644 --- a/tests/ci/test_browser_use_skill_install_docs.py +++ b/tests/ci/test_browser_use_skill_install_docs.py @@ -1,4 +1,5 @@ import os +import re import subprocess import sys from pathlib import Path @@ -57,6 +58,16 @@ def test_docs_install_browser_use_skill_from_package_alias(): assert 'raw.githubusercontent.com/browser-use/browser-harness/main/SKILL.md' not in readme +def test_cloud_v4_reference_scopes_workspace_file_listing(): + api_v4 = (ROOT / 'skills' / 'cloud' / 'references' / 'api-v4.md').read_text(encoding='utf-8') + + assert 'client.workspaces.files(workspace.id)' in api_v4 + assert 'client.workspaces.files()' not in api_v4 + python_examples = re.findall(r'```python\n(.*?)```', api_v4, flags=re.DOTALL) + assert python_examples + assert all('BrowserUse' in example for example in python_examples if 'client.' in example) + + def test_browser_use_cli_installs_browser_harness_package_skill(tmp_path): bin_dir = _fake_browser_harness_tools(tmp_path, '---\nname: browser-harness\n---\n\n# Browser Harness\n') From 5525c08c54bbdcb7cfacde1474ba8c566297012c Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 27 Aug 2026 23:23:42 -0700 Subject: [PATCH 61/69] docs: migrate remote browser skill to CLI 3.0 --- skills/remote-browser/SKILL.md | 198 +++++------------- .../ci/test_browser_use_skill_install_docs.py | 26 +++ 2 files changed, 81 insertions(+), 143 deletions(-) diff --git a/skills/remote-browser/SKILL.md b/skills/remote-browser/SKILL.md index 9c8b0efe9..7c0e52d6b 100644 --- a/skills/remote-browser/SKILL.md +++ b/skills/remote-browser/SKILL.md @@ -1,179 +1,91 @@ --- name: remote-browser -description: Controls a local browser from a sandboxed remote machine. Use when the agent is running in a sandbox (no GUI) and needs to navigate websites, interact with web pages, fill forms, take screenshots, or expose local dev servers via tunnels. +description: Controls an isolated Browser Use Cloud browser from a sandboxed machine with the current Browser Use CLI. allowed-tools: Bash(browser-use:*) --- -# Browser Automation for Sandboxed Agents +# Remote Browser -This skill is for agents running on **sandboxed remote machines** (cloud VMs, CI, coding agents) that need to control a headless browser. +Use this skill when an agent runs on a machine without a usable local Chrome and needs an isolated browser. The current Browser Use CLI runs Python from stdin. Do not use the removed `open`, `state`, `click`, `input`, `tab`, `cloud connect`, or `--connect` commands. -## Prerequisites +## Check the CLI ```bash -browser-use doctor # Verify installation +browser-use --doctor +browser-use skill show ``` -For setup details, see https://github.com/browser-use/browser-use/blob/main/browser_use/skill_cli/README.md +If setup fails, follow the current [Browser Use skill](../browser-use/SKILL.md). -## Core Workflow +## Start an isolated browser -1. **Navigate**: `browser-use open ` โ€” starts headless browser if needed -2. **Inspect**: `browser-use state` โ€” returns clickable elements with indices -3. **Interact**: use indices from state (`browser-use click 5`, `browser-use input 3 "text"`) -4. **Verify**: `browser-use state` or `browser-use screenshot` to confirm -5. **Repeat**: browser stays open between commands -6. **Cleanup**: `browser-use close` when done - -## Browser Modes +Authenticate once: ```bash -browser-use open # Default: headless Chromium -browser-use cloud connect # Provision cloud browser and connect -browser-use --connect open # Auto-discover running Chrome via CDP -browser-use --cdp-url ws://localhost:9222/... open # Connect via CDP URL +browser-use auth login ``` -## Commands +Pick a short unique name. `r7k2` below is only an example. ```bash -# Navigation -browser-use open # Navigate to URL -browser-use back # Go back in history -browser-use scroll down # Scroll down (--amount N for pixels) -browser-use scroll up # Scroll up -browser-use tab list # List all tabs with lock status -browser-use tab new [url] # Open a new tab (blank or with URL) -browser-use tab switch # Switch to tab by index -browser-use tab close [index...] # Close one or more tabs - -# Page State โ€” always run state first to get element indices -browser-use state # URL, title, clickable elements with indices -browser-use screenshot [path.png] # Screenshot (base64 if no path, --full for full page) - -# Interactions โ€” use indices from state -browser-use click # Click element by index -browser-use click # Click at pixel coordinates -browser-use type "text" # Type into focused element -browser-use input "text" # Click element, then type -browser-use keys "Enter" # Send keyboard keys (also "Control+a", etc.) -browser-use select "option" # Select dropdown option -browser-use upload # Upload file to file input -browser-use hover # Hover over element -browser-use dblclick # Double-click element -browser-use rightclick # Right-click element - -# Data Extraction -browser-use eval "js code" # Execute JavaScript, return result -browser-use get title # Page title -browser-use get html [--selector "h1"] # Page HTML (or scoped to selector) -browser-use get text # Element text content -browser-use get value # Input/textarea value -browser-use get attributes # Element attributes -browser-use get bbox # Bounding box (x, y, width, height) - -# Wait -browser-use wait selector "css" # Wait for element (--state visible|hidden|attached|detached, --timeout ms) -browser-use wait text "text" # Wait for text to appear - -# Cookies -browser-use cookies get [--url ] # Get cookies (optionally filtered) -browser-use cookies set # Set cookie (--domain, --secure, --http-only, --same-site, --expires) -browser-use cookies clear [--url ] # Clear cookies -browser-use cookies export # Export to JSON -browser-use cookies import # Import from JSON - -# Python โ€” persistent session with browser access -browser-use python "code" # Execute Python (variables persist across calls) -browser-use python --file script.py # Run file -browser-use python --vars # Show defined variables -browser-use python --reset # Clear namespace - -# Session -browser-use close # Close browser and stop daemon -browser-use sessions # List active sessions -browser-use close --all # Close all sessions +browser-use <<'PY' +start_remote_daemon("r7k2") +PY ``` -The Python `browser` object provides: `browser.url`, `browser.title`, `browser.html`, `browser.goto(url)`, `browser.back()`, `browser.click(index)`, `browser.type(text)`, `browser.input(index, text)`, `browser.keys(keys)`, `browser.upload(index, path)`, `browser.screenshot(path)`, `browser.scroll(direction, amount)`, `browser.wait(seconds)`. - -## Tunnels - -Expose local dev servers to the browser via Cloudflare tunnels. +Use the same name for every command in this browser: ```bash -browser-use tunnel # Start tunnel (idempotent) -browser-use tunnel list # Show active tunnels -browser-use tunnel stop # Stop tunnel -browser-use tunnel stop --all # Stop all tunnels +BU_NAME=r7k2 browser-use <<'PY' +new_tab("https://example.com") +wait_for_load() +print(page_info()) +PY ``` -## Command Chaining +Each remote daemon is a separate Browser Use Cloud browser. Use a different name for each parallel task. Remote browsers can bill until they stop or time out. -Commands can be chained with `&&`. The browser persists via the daemon, so chaining is safe and efficient. +## Inspect and interact + +Helpers are pre-imported. Keep multi-step work in one heredoc when practical. ```bash -browser-use open https://example.com && browser-use state -browser-use input 5 "user@example.com" && browser-use input 6 "password" && browser-use click 7 +BU_NAME=r7k2 browser-use <<'PY' +print(page_info()) +print(js("document.title")) + +fill_input('input[name="q"]', "browser automation") +press_key("Enter") +wait_for_load() + +print(page_info()) +PY ``` -Chain when you don't need intermediate output. Run separately when you need to parse `state` to discover indices first. +Useful helpers: -## Common Workflows +- Navigate: `new_tab(url)`, `goto_url(url)`, `wait_for_load()` +- Inspect: `page_info()`, `js(code)`, `cdp(method, ...)` +- Interact: `click_at_xy(x, y)`, `type_text(text)`, `fill_input(selector, text)`, `press_key(key)`, `scroll(x, y)` +- Tabs: `list_tabs()`, `switch_tab(target)`, `close_tab(target)` +- Files and proof: `capture_screenshot()`, `wait_for_element(selector)` -### Exposing Local Dev Servers +Prefer the accessibility tree for element discovery: + +```python +nodes = cdp("Accessibility.getFullAXTree")["nodes"] +``` + +Use a targeted `js(...)` query when the accessibility tree lacks the element. Verify each action with `page_info()`, a focused DOM check, or a screenshot. + +## Stop the browser + +When the work is done, stop the exact named browser: ```bash -python -m http.server 3000 & # Start dev server -browser-use tunnel 3000 # โ†’ https://abc.trycloudflare.com -browser-use open https://abc.trycloudflare.com # Browse the tunnel +browser-use <<'PY' +stop_remote_daemon("r7k2") +PY ``` -Tunnels are independent of browser sessions and persist across `browser-use close`. - -## Multi-Agent (--connect mode) - -Multiple agents can share one browser via `--connect`. Each agent gets its own tab โ€” other agents can't interfere. - -**Setup**: Register once, then pass the index with every `--connect` command: - -```bash -INDEX=$(browser-use register) # โ†’ prints "1" -browser-use --connect $INDEX open # Navigate in agent's own tab -browser-use --connect $INDEX state # Get state from agent's tab -browser-use --connect $INDEX click # Click in agent's tab -``` - -- **Tab locking**: When an agent mutates a tab (click, type, navigate), that tab is locked to it. Other agents get an error if they try to mutate the same tab. -- **Read-only access**: `state`, `screenshot`, `get`, and `wait` commands work on any tab regardless of locks. -- **Agent sessions expire** after 5 minutes of inactivity. Run `browser-use register` again to get a new index. - -## Global Options - -| Option | Description | -|--------|-------------| -| `--headed` | Show browser window | -| `--connect` | Auto-discover running Chrome via CDP | -| `--cdp-url ` | Connect via CDP URL (`http://` or `ws://`) | -| `--session NAME` | Target a named session (default: "default") | -| `--json` | Output as JSON | - -## Tips - -1. **Always run `state` first** to see available elements and their indices -2. **Sessions persist** โ€” browser stays open between commands until you close it -3. **Tunnels are independent** โ€” they persist across `browser-use close` -4. **`tunnel` is idempotent** โ€” calling again for the same port returns the existing URL - -## Troubleshooting - -- **Browser won't start?** `browser-use close` then retry. Run `browser-use doctor` to check. -- **Element not found?** `browser-use scroll down` then `browser-use state` -- **Tunnel not working?** `which cloudflared` to check, `browser-use tunnel list` to see active tunnels - -## Cleanup - -```bash -browser-use close # Close browser session -browser-use tunnel stop --all # Stop tunnels (if any) -``` +Do not leave an unused remote browser running. diff --git a/tests/ci/test_browser_use_skill_install_docs.py b/tests/ci/test_browser_use_skill_install_docs.py index 41da21a81..878e570a6 100644 --- a/tests/ci/test_browser_use_skill_install_docs.py +++ b/tests/ci/test_browser_use_skill_install_docs.py @@ -57,6 +57,32 @@ def test_docs_install_browser_use_skill_from_package_alias(): assert 'raw.githubusercontent.com/browser-use/browser-harness/main/SKILL.md' not in readme +def test_remote_browser_skill_uses_current_cli(): + remote_skill = (ROOT / 'skills' / 'remote-browser' / 'SKILL.md').read_text(encoding='utf-8') + + for removed_command in ( + 'browser-use open', + 'browser-use state', + 'browser-use click', + 'browser-use input', + 'browser-use tab', + 'browser-use cloud connect', + 'browser-use --connect', + 'browser_use/skill_cli/README.md', + ): + assert removed_command not in remote_skill + + for current_command in ( + "browser-use <<'PY'", + 'start_remote_daemon("r7k2")', + 'BU_NAME=r7k2 browser-use', + 'new_tab("https://example.com")', + 'print(page_info())', + 'stop_remote_daemon("r7k2")', + ): + assert current_command in remote_skill + + def test_browser_use_cli_installs_browser_harness_package_skill(tmp_path): bin_dir = _fake_browser_harness_tools(tmp_path, '---\nname: browser-harness\n---\n\n# Browser Harness\n') From 7ed622f0de3ec8733b40ab7c7a6a94e6f9e5bcba Mon Sep 17 00:00:00 2001 From: MagMueller Date: Thu, 27 Aug 2026 23:28:44 -0700 Subject: [PATCH 62/69] test: cover all retired browser commands --- tests/ci/test_browser_use_skill_install_docs.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/tests/ci/test_browser_use_skill_install_docs.py b/tests/ci/test_browser_use_skill_install_docs.py index 878e570a6..933614c1c 100644 --- a/tests/ci/test_browser_use_skill_install_docs.py +++ b/tests/ci/test_browser_use_skill_install_docs.py @@ -66,6 +66,14 @@ def test_remote_browser_skill_uses_current_cli(): 'browser-use click', 'browser-use input', 'browser-use tab', + 'browser-use screenshot', + 'browser-use eval', + 'browser-use cookies', + 'browser-use close', + 'browser-use sessions', + 'browser-use tunnel', + 'browser-use wait', + 'browser-use register', 'browser-use cloud connect', 'browser-use --connect', 'browser_use/skill_cli/README.md', From a2663996457211a8f995b4d4102296d10956ebf1 Mon Sep 17 00:00:00 2001 From: Aniket Wagh Date: Sat, 29 Aug 2026 20:27:38 +0530 Subject: [PATCH 63/69] fix(llm/google): keep the first user message when include_system_in_user is set --- browser_use/llm/google/serializer.py | 21 ++--- tests/ci/models/test_llm_google.py | 116 +++++++++++++++++++++++++++ 2 files changed, 127 insertions(+), 10 deletions(-) diff --git a/browser_use/llm/google/serializer.py b/browser_use/llm/google/serializer.py index 56622dc91..09eef9bdb 100644 --- a/browser_use/llm/google/serializer.py +++ b/browser_use/llm/google/serializer.py @@ -77,20 +77,21 @@ class GoogleMessageSerializer: message_parts: list[Part] = [] # If this is the first user message and we have system parts, prepend them + system_text = None if include_system_in_user and system_parts and role == 'user' and not formatted_messages: system_text = '\n\n'.join(system_parts) - if isinstance(message.content, str): - message_parts.append(Part.from_text(text=f'{system_text}\n\n{message.content}')) - else: - # Add system text as the first part - message_parts.append(Part.from_text(text=system_text)) system_parts = [] # Clear after using + + # Extract content and create parts + if isinstance(message.content, str): + # Regular text content + text = f'{system_text}\n\n{message.content}' if system_text is not None else message.content + message_parts.append(Part.from_text(text=text)) else: - # Extract content and create parts normally - if isinstance(message.content, str): - # Regular text content - message_parts = [Part.from_text(text=message.content)] - elif message.content is not None: + if system_text is not None: + # Add system text as the first part, the message's own parts still follow + message_parts.append(Part.from_text(text=system_text)) + if message.content is not None: # Handle Iterable of content parts for part in message.content: if part.type == 'text': diff --git a/tests/ci/models/test_llm_google.py b/tests/ci/models/test_llm_google.py index 5389dddd3..6edfd5edb 100644 --- a/tests/ci/models/test_llm_google.py +++ b/tests/ci/models/test_llm_google.py @@ -127,3 +127,119 @@ async def test_chat_google_temperature_fallback(): mock_models.generate_content.assert_called_once() args, kwargs = mock_models.generate_content.call_args assert kwargs['config']['temperature'] == 1.0 + + +# A 1x1 PNG, small enough to inline and still be a real decodable image. +_PNG_1PX = 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg==' + + +def _flatten(contents) -> list: + """Flatten serialized Google contents into a single list of parts.""" + return [part for content in contents for part in (content.parts or [])] + + +def _describe(contents) -> tuple[str, int]: + """Summarise serialized Google contents as (all text, number of inline images).""" + parts = _flatten(contents) + return ''.join(part.text or '' for part in parts), sum(1 for part in parts if part.inline_data is not None) + + +def test_include_system_in_user_keeps_list_content_parts(): + """include_system_in_user must prepend the system text, not replace the user content. + + With vision on, the agent's user message is a list of parts, and every part of it was + being dropped from the first user message. + """ + from browser_use.llm.google.serializer import GoogleMessageSerializer + from browser_use.llm.messages import ( + ContentPartImageParam, + ContentPartTextParam, + ImageURL, + SystemMessage, + UserMessage, + ) + + messages = [ + SystemMessage(content='You are a browser agent.'), + UserMessage( + content=[ + ContentPartTextParam(text='the page and the task live here'), + ContentPartImageParam(image_url=ImageURL(url=f'data:image/png;base64,{_PNG_1PX}', media_type='image/png')), + ] + ), + ] + + contents, system_instruction = GoogleMessageSerializer.serialize_messages(messages, include_system_in_user=True) + + assert system_instruction is None, 'system message should have moved into the user turn' + text, images = _describe(contents) + assert 'You are a browser agent.' in text + assert 'the page and the task live here' in text, 'user text was dropped' + assert images == 1, 'screenshot was dropped' + + +def test_include_system_in_user_string_content_is_merged_into_one_part(): + """String content keeps the existing behaviour: system text and user text in a single part.""" + from browser_use.llm.google.serializer import GoogleMessageSerializer + from browser_use.llm.messages import SystemMessage, UserMessage + + messages = [ + SystemMessage(content='You are a browser agent.'), + UserMessage(content='Buy a red stapler'), + ] + + contents, system_instruction = GoogleMessageSerializer.serialize_messages(messages, include_system_in_user=True) + + assert system_instruction is None + parts = _flatten(contents) + assert len(parts) == 1 + assert parts[0].text == 'You are a browser agent.\n\nBuy a red stapler' + + +def test_include_system_in_user_keeps_the_agent_state_message(tmp_path): + """End-to-end shape: what the agent actually builds with vision on must survive serialization.""" + from browser_use.agent.prompts import AgentMessagePrompt, SystemPrompt + from browser_use.agent.views import AgentStepInfo + from browser_use.browser.views import BrowserStateSummary, PageInfo, TabInfo + from browser_use.dom.views import SerializedDOMState + from browser_use.filesystem.file_system import FileSystem + from browser_use.llm.google.serializer import GoogleMessageSerializer + + browser_state = BrowserStateSummary( + url='https://example.test/foo', + title='Test', + tabs=[TabInfo(target_id='abcd1234', url='https://example.test/foo', title='Test')], + page_info=PageInfo( + viewport_width=1280, + viewport_height=720, + page_width=1280, + page_height=1440, + scroll_x=0, + scroll_y=0, + pixels_above=0, + pixels_below=720, + pixels_left=0, + pixels_right=0, + ), + dom_state=SerializedDOMState(_root=None, selector_map={}), + is_pdf_viewer=False, + recent_events=None, + closed_popup_messages=[], + screenshot=_PNG_1PX, + ) + user_message = AgentMessagePrompt( + browser_state_summary=browser_state, + file_system=FileSystem(base_dir=str(tmp_path), create_default_files=False), + agent_history_description='existing history', + task='Buy a red stapler', + step_info=AgentStepInfo(step_number=1, max_steps=50), + screenshots=[_PNG_1PX], + ).get_user_message(use_vision=True) + assert isinstance(user_message.content, list), 'vision messages are lists of parts' + + messages = [SystemPrompt(max_actions_per_step=5).get_system_message(), user_message] + contents, _ = GoogleMessageSerializer.serialize_messages(messages, include_system_in_user=True) + + text, images = _describe(contents) + assert 'Buy a red stapler' in text, 'the task never reached the model' + assert images == 1, 'the screenshot never reached the model' From b4bad86d21bf693b9ae5748b667fc32fc8b8e7d5 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Wed, 26 Aug 2026 09:27:15 -0700 Subject: [PATCH 64/69] fix: allow pydantic 2.13 --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index dfad337c0..81a6d224b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -21,7 +21,7 @@ dependencies = [ "httpx==0.28.1", "posthog==7.7.0", "psutil==7.2.2", - "pydantic==2.12.5", + "pydantic>=2.12.5,<2.14", "pyobjc==12.1; platform_system == 'darwin'", "python-dotenv==1.2.2", "requests==2.33.0", From b4e68f19b2fd88e9df38847b641447e8aedaffba Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sat, 29 Aug 2026 21:41:22 -0700 Subject: [PATCH 65/69] docs: correct signup credit in Cloud guide Signed-off-by: MagMueller --- CLOUD.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CLOUD.md b/CLOUD.md index 4c301df7b..e0b4c3c31 100644 --- a/CLOUD.md +++ b/CLOUD.md @@ -19,7 +19,7 @@ The key product of Browser Use Cloud is the completion of user tasks. - Profile Sync is the best way to handle authentication for tasks. This feature allows users to upload their local browser cookies (where the user is already logged into the services they need authentication for) to a Browser Profile that can be used for tasks on the cloud. To initiate a Profile Sync, a user must run `export BROWSER_USE_API_KEY= && curl -fsSL https://browser-use.com/profile.sh | sh` and follow the steps in the interactive terminal. ## Quickstart -To get started, direct the user to first must create an account, purchase credits (or simply claim the five free tasks given on account creation), and generate an API key on the Browser Use online platform: https://cloud.browser-use.com/. These are the only steps that can only be done on the platform. +To get started, direct the user to first create an account, claim the $15 one-time signup credit if eligible (or purchase credits), and generate an API key on the Browser Use online platform: https://cloud.browser-use.com/. These are the only steps that can only be done on the platform. Avoid giving the user all of the following steps at once as it may seem overwheling. Instead present one step at a time and only continue when asked. Do as much for the user as you are able to. From 3fdb4ab69a250347b0af7d09678449ebfdb54e3f Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sat, 29 Aug 2026 20:57:36 -0700 Subject: [PATCH 66/69] docs: correct cloud browser session price --- skills/cloud/references/quickstart.md | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/skills/cloud/references/quickstart.md b/skills/cloud/references/quickstart.md index 69a322af4..5277cb326 100644 --- a/skills/cloud/references/quickstart.md +++ b/skills/cloud/references/quickstart.md @@ -122,8 +122,7 @@ Typical task: 10 steps = ~$0.03 (with Browser Use LLM) | BU Max (Claude Sonnet 4.6) | ~$3.60 | ~$18.00 | ### Browser Sessions -- PAYG: $0.06/hour -- Business: $0.03/hour +- All plans: $0.02/hour - Billed upfront, proportional refund on stop. Min 1 minute. ### Skills From cf0b917fba64163c7bba857e87bd56a04cd7579f Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sun, 30 Aug 2026 09:08:29 -0700 Subject: [PATCH 67/69] docs: remove stale session discount --- skills/cloud/references/quickstart.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/skills/cloud/references/quickstart.md b/skills/cloud/references/quickstart.md index 5277cb326..054b6671b 100644 --- a/skills/cloud/references/quickstart.md +++ b/skills/cloud/references/quickstart.md @@ -133,7 +133,7 @@ Typical task: 10 steps = ~$0.03 (with Browser Use LLM) - PAYG: $10/GB, Business: $5/GB, Scaleup: $4/GB ### Tiers -- **Business**: 25% off per-step, 50% off sessions/skills/proxy +- **Business**: 25% off per-step, 50% off skills/proxy - **Scaleup**: 50% off per-step, 60% off proxy - **Enterprise**: Contact for ZDR, compliance, on-prem From c48b1c99289dd3ad88c4af6367756baacccfa636 Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sun, 30 Aug 2026 04:01:46 -0700 Subject: [PATCH 68/69] fix(llm): preserve schema-keyword field names --- browser_use/llm/schema.py | 9 +++++++-- tests/ci/models/test_llm_schema_optimizer.py | 19 +++++++++++++++++++ 2 files changed, 26 insertions(+), 2 deletions(-) diff --git a/browser_use/llm/schema.py b/browser_use/llm/schema.py index d4e01be42..75c33d07a 100644 --- a/browser_use/llm/schema.py +++ b/browser_use/llm/schema.py @@ -45,11 +45,16 @@ class SchemaOptimizer: skip_fields = ['additionalProperties', '$defs'] for key, value in obj.items(): + # Keys inside `properties` are user field names, not schema keywords. + if in_properties: + optimized[key] = optimize_schema(value, defs_lookup) + continue + if key in skip_fields: continue - # Skip metadata "title" unless we're iterating inside an actual `properties` map - if key == 'title' and not in_properties: + # Skip metadata "title" + if key == 'title': continue # Preserve FULL descriptions without truncation, skip empty ones diff --git a/tests/ci/models/test_llm_schema_optimizer.py b/tests/ci/models/test_llm_schema_optimizer.py index 5aa452b05..e3e472732 100644 --- a/tests/ci/models/test_llm_schema_optimizer.py +++ b/tests/ci/models/test_llm_schema_optimizer.py @@ -74,3 +74,22 @@ def test_gemini_schema_retains_required_fields(): required_fields = set(schema['required']) assert {'price', 'title'}.issubset(required_fields), 'Mandatory fields must stay required for Gemini.' + + +def test_optimizer_treats_property_names_as_data_not_schema_keywords(): + """Nested fields named after schema keywords must still have their refs flattened.""" + + class Details(BaseModel): + summary: str + + class Article(BaseModel): + description: Details + properties: Details + + schema = SchemaOptimizer.create_optimized_json_schema(Article) + + assert '$defs' not in schema + for field_name in ('description', 'properties'): + field_schema = schema['properties'][field_name] + assert '$ref' not in field_schema + assert field_schema['properties']['summary']['type'] == 'string' From b2507091aa2234bdbfebc39300a7174b04505c3f Mon Sep 17 00:00:00 2001 From: MagMueller Date: Sun, 30 Aug 2026 14:56:47 -0700 Subject: [PATCH 69/69] test: cover aliased schema keyword fields --- tests/ci/models/test_llm_schema_optimizer.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/tests/ci/models/test_llm_schema_optimizer.py b/tests/ci/models/test_llm_schema_optimizer.py index e3e472732..49142c84d 100644 --- a/tests/ci/models/test_llm_schema_optimizer.py +++ b/tests/ci/models/test_llm_schema_optimizer.py @@ -3,7 +3,7 @@ Tests for the SchemaOptimizer to ensure it correctly processes and optimizes the schemas for agent actions without losing information. """ -from pydantic import BaseModel +from pydantic import BaseModel, Field from browser_use.agent.views import AgentOutput from browser_use.llm.schema import SchemaOptimizer @@ -85,11 +85,13 @@ def test_optimizer_treats_property_names_as_data_not_schema_keywords(): class Article(BaseModel): description: Details properties: Details + additional_properties: Details = Field(alias='additionalProperties') + defs: Details = Field(alias='$defs') schema = SchemaOptimizer.create_optimized_json_schema(Article) assert '$defs' not in schema - for field_name in ('description', 'properties'): + for field_name in ('description', 'properties', 'additionalProperties', '$defs'): field_schema = schema['properties'][field_name] assert '$ref' not in field_schema assert field_schema['properties']['summary']['type'] == 'string'