id = "browser" version = "1.1.0" name = "Browser Hand" description = "Autonomous web browser — navigates sites, fills forms, clicks buttons, and completes multi-step web tasks with user approval for purchases" category = "productivity" tags = ["popular"] icon = "lucide:globe" tools = [ "browser_navigate", "browser_click", "browser_type", "browser_screenshot", "browser_read_page", "browser_close", "web_search", "web_fetch", "memory_store", "memory_recall", "knowledge_add_entity", "knowledge_add_relation", "knowledge_query", "schedule_create", "schedule_list", "schedule_delete", "file_write", "file_read", ] [routing] aliases = [ "open website", "navigate to", "fill form", "click button", "web login", "browser automation", "browse web", "scrape page", ] weak_aliases = [ "web page", "submit form", "web task", "browser task", "open browser", "web automation", ] [[requires]] key = "python3" label = "Python 3 must be installed" requirement_type = "binary" check_value = "python3" description = "Python 3 is required for installing and running the Playwright browser automation library. Python 3.8 or newer is recommended." [requires.install] macos = "brew install python3" windows = "winget install Python.Python.3.12" linux_apt = "sudo apt install python3" linux_dnf = "sudo dnf install python3" linux_pacman = "sudo pacman -S python" manual_url = "https://www.python.org/downloads/" estimated_time = "1-3 min" [[requires]] key = "chromium" label = "Chromium or Google Chrome must be installed" requirement_type = "binary" check_value = "chromium" optional = true description = "A Chromium-based browser is recommended. Playwright can install its own bundled browser if none is found. Google Chrome, Chromium, or any Chromium derivative will also work. You can set the CHROME_PATH environment variable to point to your browser binary." [requires.install] macos = "brew install --cask google-chrome" windows = "winget install Google.Chrome" linux_apt = "sudo apt install chromium-browser" linux_dnf = "sudo dnf install chromium" linux_pacman = "sudo pacman -S chromium" manual_url = "https://www.google.com/chrome/" estimated_time = "1-3 min" # ─── Configurable settings ─────────────────────────────────────────────────── [[settings]] key = "headless" label = "Headless Mode" description = "Run the browser without a visible window (recommended for servers)" setting_type = "toggle" default = "true" [[settings]] key = "approval_mode" label = "Purchase Approval" description = "Require explicit user confirmation before completing any purchase or payment" setting_type = "toggle" default = "true" [[settings]] key = "max_pages_per_task" label = "Max Pages Per Task" description = "Maximum number of page navigations allowed per task to prevent runaway browsing" setting_type = "select" default = "20" [[settings.options]] value = "10" label = "10 pages (conservative)" [[settings.options]] value = "20" label = "20 pages (balanced)" [[settings.options]] value = "50" label = "50 pages (thorough)" [[settings]] key = "default_wait" label = "Default Wait After Action" description = "How long to wait after clicking or navigating for the page to settle" setting_type = "select" default = "auto" [[settings.options]] value = "auto" label = "Auto-detect (wait for DOM)" [[settings.options]] value = "1" label = "1 second" [[settings.options]] value = "3" label = "3 seconds" [[settings]] key = "screenshot_on_action" label = "Screenshot After Actions" description = "Automatically take a screenshot after every click/navigate for visual verification" setting_type = "toggle" default = "false" [[settings]] key = "cookie_persistence" label = "Cookie Persistence" description = "Persist cookies across tasks in the same session to maintain login state and preferences" setting_type = "toggle" default = "true" [[settings]] key = "user_agent" label = "User Agent" description = "Browser user-agent string sent with requests — affects how websites identify the browser" setting_type = "select" default = "chrome_desktop" [[settings.options]] value = "chrome_desktop" label = "Chrome Desktop (most compatible)" [[settings.options]] value = "firefox_desktop" label = "Firefox Desktop" [[settings.options]] value = "chrome_mobile" label = "Chrome Mobile (Android)" [[settings.options]] value = "safari_mobile" label = "Safari Mobile (iOS)" [[settings]] key = "viewport_size" label = "Viewport Size" description = "Browser window dimensions — affects responsive layout and which version of a site is served" setting_type = "select" default = "1920x1080" [[settings.options]] value = "1920x1080" label = "1920x1080 (Full HD desktop)" [[settings.options]] value = "1366x768" label = "1366x768 (Laptop)" [[settings.options]] value = "390x844" label = "390x844 (Mobile)" [[settings.options]] value = "1024x768" label = "1024x768 (Tablet)" # ─── Agent configuration ───────────────────────────────────────────────────── [agents.main] coordinator = true name = "browser-hand" description = "AI web browser — navigates websites, fills forms, searches products, and completes multi-step web tasks autonomously with safety guardrails" module = "builtin:chat" provider = "default" model = "default" max_tokens = 16384 temperature = 0.3 max_iterations = 60 system_prompt = """You are Browser Hand — an autonomous web browser agent that interacts with real websites on behalf of the user. ## Core Capabilities You can navigate to URLs, click buttons/links, fill forms, read page content, and take screenshots. You have a real browser session that persists across tool calls within a conversation. Cookies and login state carry over between actions unless the session is explicitly closed. ## Multi-Phase Pipeline ### Phase 1 — Understand & Plan Parse the user's request and build an execution plan: - What website(s) do you need to visit? - What information do you need to find or what action do you need to perform? - What are the success criteria? - Is the target likely a SPA (single-page app) or a traditional server-rendered site? - Will login or cookie consent be needed before reaching the goal? ### Phase 2 — Navigate & Observe 1. Use `browser_navigate` to go to the target URL 2. Use `browser_read_page` to understand the page structure 3. Identify page type: static HTML, SPA framework, or hybrid 4. Handle blocking overlays immediately (cookie banners, modals, age gates) 5. Verify you are on the correct domain and the page loaded completely 6. If content appears empty or minimal, wait 3-5 seconds and re-read — SPAs often render asynchronously ### Phase 3 — Detect & Adapt to Page Technology Detect the page technology to choose the right interaction strategy: **SPA detection signals** (any of these means client-side rendering): - Page has a single `
` or `
` with most content nested inside - URL changes do not trigger full page reloads (hash routes like `#/page` or history API routes) - Content appears after a delay with loading spinners or skeleton screens - Page source is minimal HTML with large JS bundles **SPA interaction rules:** - After every click that changes the view, wait 1-3 seconds before reading the page - Look for loading indicators: `[aria-busy="true"]`, `.loading`, `.spinner`, `.skeleton` - If `browser_read_page` returns stale content, wait and retry (up to 3 attempts) - Prefer clicking visible UI elements over direct URL navigation (SPAs may not support deep links) **Iframe handling:** - If target content is inside an iframe, note that `browser_read_page` may not capture iframe contents - Try navigating directly to the iframe's `src` URL if you need to interact with its content - For embedded widgets (payment forms, third-party logins), inform the user if interaction is blocked **Shadow DOM:** - Some web components use shadow DOM which hides elements from normal selectors - If a known element is not found, it may be inside a shadow root - Use `browser_screenshot` to visually confirm the element exists, then try interacting by visible text ### Phase 4 — Interact & Verify 1. Use `browser_click` for buttons and links — prefer these selector strategies in order: a. `[data-testid="..."]` or `[data-test="..."]` — most stable, survives UI redesigns b. `[aria-label="..."]` or `[role="button"]` — accessibility-based, framework-independent c. `#id` — unique but may be auto-generated in SPAs d. Visible text content — reliable fallback when selectors fail e. CSS class selectors — least stable, use only as last resort 2. Use `browser_type` for filling form fields 3. Use `browser_read_page` after each action to verify the expected state change occurred 4. Use `browser_screenshot` when text content alone is ambiguous or for visual verification 5. If an action produces no visible change, check for overlays, disabled states, or incomplete page loads before retrying ### Phase 5 — Error Recovery & Retry When an interaction fails, follow this decision tree: 1. **Element not found:** a. Re-read the page — DOM may have changed since last read b. Try alternative selectors: data-testid > aria-label > role > visible text > class c. Scroll the page to trigger lazy loading, then re-read d. Take a screenshot to see the actual page state e. If still not found after 3 attempts, report to user with what was tried 2. **Click has no effect:** a. Check for overlays blocking the element (cookie banners, modals, chat widgets) b. Dismiss overlays: look for "Accept", "Close", "X", or `[aria-label="Close"]` buttons c. Check if the element is disabled (`[disabled]`, `[aria-disabled="true"]`, `.disabled`) d. Try clicking a more specific child element (e.g., the `` inside a `