# Browser-Use
Source: https://hyperbrowser.ai/docs/agents/browser-use
Fast and efficient browser automation with the open-source browser-use framework
Browser-Use is an open-source solution optimized for fast, efficient browser automation. It enables AI to interact with websites naturally—clicking, typing, scrolling, and navigating just like a human would.
Perfect for automating repetitive web tasks, extracting data from complex sites, or testing web applications at scale. Hyperbrowser hosts the browser-use framework so you can run agent tasks with a single API call.
You can view your Browser-Use tasks in the [dashboard](https://app.hyperbrowser.ai/features/agents/browser-use).
Browser-Use agents run asynchronously by default. Start a task, then poll for results. Our SDKs include a `startAndWait()` helper that handles polling automatically and returns when the task completes.
## How It Works
You can use Browser-Use in two ways:
1. **Start and Wait**: SDKs provide a `startAndWait()` method that blocks until the task completes and returns the result
2. **Async Pattern**: Start a task, get a job ID, then poll for status and results—useful for long-running tasks or when you want more control
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The simplest way to run a Browser-Use task is with `startAndWait()`, which handles the entire lifecycle for you:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.browserUse.startAndWait({
task: "Go to Hacker News and tell me the title of the top post",
llm: "gemini-2.5-flash",
maxSteps: 20,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.browser_use.start_and_wait(
params={
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "gemini-2.5-flash",
"max_steps": 20,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
from hyperbrowser.models import StartBrowserUseTaskParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.browser_use.start_and_wait(
params=StartBrowserUseTaskParams(
task="Go to Hacker News and tell me the title of the top post",
llm="gemini-2.5-flash",
max_steps=20,
)
)
print(f"Output:\n{result.data.final_result}")
```
```bash cURL theme={null}
# Start the task
curl -X POST https://api.hyperbrowser.ai/api/task/browser-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "gemini-2.5-flash",
"maxSteps": 20
}'
# Response: {"jobId": "abc123", "liveUrl": "https://..."}
# Check status
curl https://api.hyperbrowser.ai/api/task/browser-use/abc123/status \
-H "x-api-key: YOUR_API_KEY"
# Get full results
curl https://api.hyperbrowser.ai/api/task/browser-use/abc123 \
-H "x-api-key: YOUR_API_KEY"
```
## Async Pattern
When you need more control, use the async pattern to start a task and poll for results:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
try {
// Start the task
const task = await client.agents.browserUse.start({
task: "What is the title of the first post on Hacker News today?",
llm: "gemini-2.5-flash",
maxSteps: 20,
});
console.log(`Task started: ${task.jobId}`);
console.log(`Watch live: ${task.liveUrl}`);
// Poll for completion
let result;
while (true) {
result = await client.agents.browserUse.getStatus(task.jobId);
console.log(`Status: ${result.status}`);
if (result.status === "completed" || result.status === "failed") {
break;
}
await new Promise((resolve) => setTimeout(resolve, 5000)); // Wait 5s
}
const fullResult = await client.agents.browserUse.get(task.jobId);
if (fullResult.status === "completed") {
console.log("Result:", fullResult.data?.finalResult);
console.log("Steps taken:", fullResult.data?.steps?.length);
} else {
console.error("Task failed:", fullResult.error);
}
} catch (err) {
console.error(`Error: ${err.message}`);
}
}
main();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.browser_use.start(
params={
"task": "What is the title of the first post on Hacker News today?",
"llm": "gemini-2.5-flash",
"max_steps": 20,
}
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.browser_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.browser_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import StartBrowserUseTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.browser_use.start(
params=StartBrowserUseTaskParams(
task="What is the title of the first post on Hacker News today?",
llm="gemini-2.5-flash",
max_steps=20,
)
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.browser_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.browser_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
## Stop a Running Task
Stop a task before it completes:
```typescript Node.js theme={null}
await client.agents.browserUse.stop("job-id");
```
```python Python theme={null}
client.agents.browser_use.stop("job-id")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/task/browser-use/job-id/stop \
-H "x-api-key: YOUR_API_KEY"
```
## Parameters
Natural language description of what the agent should accomplish.
Version of browser-use to use.
Options: `0.1.40`, `0.7.10`, `latest`
Be cautious when using the `latest` version. We periodically update the latest version to match the latest version of browser-use and this may introduce changes to how the agent works or other breaking changes.
Language model to use. Options: `gpt-4o`, `gpt-4o-mini`, `gpt-4.1`, `gpt-4.1-mini`, `claude-sonnet-5`, `claude-sonnet-4-6`, `claude-sonnet-4-5`, `claude-sonnet-4-20250514`, `gemini-2.0-flash`, `gemini-2.5-flash`, `gemini-3.8-flash`
Maximum number of steps the agent can take. Increase if tasks aren't able to complete within the given number of steps.
ID of an existing browser session to reuse. Useful for multi-step workflows that need to maintain the same browser session.
Enable screenshot analysis for better context understanding.
Validate agent output against a schema.
Provide screenshots to the planning component.
Maximum actions per step before reassessing.
Maximum tokens for LLM input.
Separate language model for planning (can be different from main LLM).
Separate language model for extracting structured data from pages.
How often (in steps) the planner reassesses strategy.
Maximum consecutive failures before aborting the task.
List of actions to execute before starting the main task.
Key-value pairs to mask the data sent to the LLM. The LLM only sees placeholders (x\_user, x\_pass), browser-use filters your sensitive data from the input text. Real values are injected directly into form fields after the LLM call.
Valid JSON schema for structured output.
Keep session alive after task completes.
[Session configuration](/docs/api-reference/start-a-browser-use-task#body-session-options) (proxy, stealth, captcha solving, etc.). Only applies when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Use your own LLM API keys instead of Hyperbrowser's. You will only be charged for browser usage.
API keys for `openai`, `anthropic`, and `google`. Required when `useCustomApiKeys` is `true`. Must provide keys based on the LLMs you are using.
```typescript theme={null}
{
openai: "...",
anthropic: "...",
google: "..."
}
```
The agent may not complete the task within the specified `maxSteps`. If that happens, try increasing the `maxSteps` parameter.
Additionally, the browser session used by the AI Agent will time out based on your team's default Session Timeout settings or the session's `timeoutMinutes` parameter if provided. You can adjust the default Session Timeout in the [Settings page](https://app.hyperbrowser.ai/settings).
## Reuse Browser Sessions
You can pass in an existing `sessionId` to the Browser Use task so that it can execute the task on an existing session. Also, if you want to keep the session open after executing the task, you can supply the `keepBrowserOpen` parameter.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const session = await client.sessions.create();
try {
const result = await client.agents.browserUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionId: session.id,
keepBrowserOpen: true,
});
console.log(`Output:\n${result.data?.finalResult}`);
const result2 = await client.agents.browserUse.startAndWait({
task: "Tell me how many upvotes the first post has.",
sessionId: session.id,
});
console.log(`\nOutput:\n${result2.data?.finalResult}`);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.browser_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_id": session.id,
"keep_browser_open": True,
}
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.browser_use.start_and_wait(
{
"task": "Tell me how many upvotes the first post has.",
"session_id": session.id,
}
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartBrowserUseTaskParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.browser_use.start_and_wait(
StartBrowserUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_id=session.id,
keep_browser_open=True,
)
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.browser_use.start_and_wait(
StartBrowserUseTaskParams(
task="Tell me how many upvotes the first post has.",
session_id=session.id,
)
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
Always set `keepBrowserOpen: true` on tasks that you want to reuse the session from. Otherwise, the session will be automatically closed when the task completes.
## Use Your Own API Keys
You can provide your own API Keys to the Browser Use task so that it doesn't charge credits to your Hyperbrowser account for the steps it takes during execution. Only the credits for the [usage of the browser itself](/docs/reference/pricing#browser-sessions) will be charged. Depending on which model you select for the `llm`, `plannerLlm`, and `pageExtractionLlm` parameters, the API keys from those providers will need to be provided when `useCustomApiKeys` is set to true.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.browserUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
llm: "gpt-4o",
plannerLlm: "gpt-4o",
pageExtractionLlm: "gpt-4o",
useCustomApiKeys: true,
apiKeys: {
openai: "",
// Below are needed if Claude or Gemini models are used
// anthropic: "",
// google: "",
},
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.browser_use.start_and_wait(
{
"task": "What is the title of the first post on HackerNews today?",
"llm": "gpt-4o",
"planner_llm": "gpt-4o",
"page_extraction_llm": "gpt-4o",
"use_custom_api_keys": True,
"api_keys": {
"openai": "",
# Below are needed if Claude or Gemini models are used
# "anthropic": "",
# "google": "",
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartBrowserUseTaskParams, BrowserUseApiKeys
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.browser_use.start_and_wait(
StartBrowserUseTaskParams(
task="What is the title of the first post on HackerNews today?",
llm="gpt-4o",
planner_llm="gpt-4o",
page_extraction_llm="gpt-4o",
use_custom_api_keys=True,
api_keys=BrowserUseApiKeys(
openai="",
# Below are needed if Claude or Gemini models are used
# anthropic="",
# google="",
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/browser-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_HYPERBROWSER_API_KEY" \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"llm": "gpt-4o",
"plannerLlm": "gpt-4o",
"pageExtractionLlm": "gpt-4o",
"useCustomApiKeys": true,
"apiKeys": {
"openai": "YOUR_OPENAI_API_KEY"
}
}'
```
You can provide keys for multiple providers:
```json theme={null}
{
"apiKeys": {
"openai": "sk-...",
"anthropic": "sk-ant-...",
"google": "..."
}
}
```
## Session Configuration
Configure the browser environment with proxies, stealth mode, CAPTCHA solving, and more:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.browserUse.startAndWait({
task: "go to Hacker News and summarize the top 5 posts of the day",
sessionOptions: {
acceptCookies: true,
}
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.browser_use.start_and_wait(
{
"task": "go to Hacker News and summarize the top 5 posts of the day",
"session_options": {
"accept_cookies": True,
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartBrowserUseTaskParams, CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.browser_use.start_and_wait(
StartBrowserUseTaskParams(
task="go to Hacker News and summarize the top 5 posts of the day",
session_options=CreateSessionParams(
accept_cookies=True,
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/browser-use \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "go to Hacker News and summarize the top 5 posts of the day",
"sessionOptions": {
"acceptCookies": true
}
}'
```
`sessionOptions` only apply when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Proxies and CAPTCHA solving add latency. Only enable them when necessary for your use case.
# Claude Computer Use
Source: https://hyperbrowser.ai/docs/agents/claude-computer-use
Automate browser tasks using Anthropic's Claude with sophisticated computer use capabilities
Claude Computer Use enables Claude to interact with web browsers like a human by moving the cursor, clicking elements, typing text, and navigating pages. This lets you automate complex, multi-step workflows with natural language instructions.
Hyperbrowser provides a managed Claude Computer Use agent that handles session management, browser infrastructure, and task execution. You simply describe what you want done, and Claude figures out how to do it.
You can view your Claude Computer Use tasks in the [dashboard](https://app.hyperbrowser.ai/features/agents/claude-computer-use).
## How It Works
You can use Claude Computer Use in two ways:
1. **Start and Wait**: SDKs provide a `startAndWait()` method that blocks until the task completes and returns the result
2. **Async Pattern**: Start a task, get a job ID, then poll for status and results—useful for long-running tasks or when you want more control
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The simplest way to run a Claude Computer Use task is with the `startAndWait()` method, which handles everything for you:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.claudeComputerUse.startAndWait({
task: "Go to Hacker News and tell me the title of the top post",
llm: "claude-sonnet-4-5",
maxSteps: 20,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.claude_computer_use.start_and_wait(
params={
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "claude-sonnet-4-5",
"max_steps": 20,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartClaudeComputerUseTaskParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.claude_computer_use.start_and_wait(
params=StartClaudeComputerUseTaskParams(
task="Go to Hacker News and tell me the title of the top post",
llm="claude-sonnet-4-5",
max_steps=20,
)
)
print(f"Output:\n{result.data.final_result}")
```
```bash cURL theme={null}
# Start the task
curl -X POST https://api.hyperbrowser.ai/api/task/claude-computer-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "claude-sonnet-4-5",
"maxSteps": 20
}'
# Response: {"jobId": "abc123", "liveUrl": "https://..."}
# Check status
curl https://api.hyperbrowser.ai/api/task/claude-computer-use/abc123/status \
-H "x-api-key: YOUR_API_KEY"
# Get full results
curl https://api.hyperbrowser.ai/api/task/claude-computer-use/abc123 \
-H "x-api-key: YOUR_API_KEY"
```
## Async Pattern
When you need more control, use the async pattern to start a task and poll for results:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
try {
// Start the task
const task = await client.agents.claudeComputerUse.start({
task: "What is the title of the first post on Hacker News today?",
llm: "claude-sonnet-4-5",
maxSteps: 20,
});
console.log(`Task started: ${task.jobId}`);
console.log(`Watch live: ${task.liveUrl}`);
// Poll for completion
let result;
while (true) {
result = await client.agents.claudeComputerUse.getStatus(task.jobId);
console.log(`Status: ${result.status}`);
if (result.status === "completed" || result.status === "failed") {
break;
}
await new Promise((resolve) => setTimeout(resolve, 5000)); // Wait 5s
}
const fullResult = await client.agents.claudeComputerUse.get(task.jobId);
if (fullResult.status === "completed") {
console.log("Result:", fullResult.data?.finalResult);
console.log("Steps taken:", fullResult.data?.steps?.length);
} else {
console.error("Task failed:", fullResult.error);
}
} catch (err) {
console.error(`Error: ${err.message}`);
}
}
main();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.claude_computer_use.start(
params={
"task": "What is the title of the first post on Hacker News today?",
"llm": "claude-sonnet-4-5",
"max_steps": 20,
}
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.claude_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.claude_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import StartClaudeComputerUseTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.claude_computer_use.start(
params=StartClaudeComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
llm="claude-sonnet-4-5",
max_steps=20,
)
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.claude_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.claude_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
## Stop a Running Task
Stop a task before it completes:
```typescript Node.js theme={null}
await client.agents.claudeComputerUse.stop("job-id");
```
```python Python theme={null}
client.agents.claude_computer_use.stop("job-id")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/task/claude-computer-use/job-id/stop \
-H "x-api-key: YOUR_API_KEY"
```
## Parameters
Natural language description of what you want Claude to accomplish. Be specific for best results.
Claude model to use. Available options:
* `"claude-opus-5-5"` - Opus 5.5 model
* `"claude-fable-5-1"` - Latest Fable 5.1 model
* `"claude-opus-5"` - Opus 5 model
* `"claude-opus-4-8"` - Opus 4.8 model (default)
* `"claude-opus-4-7"` - Opus 4.7 model
* `"claude-opus-4-6"` - Opus 4.6 model
* `"claude-opus-4-5"` - Opus 4.5 model
* `"claude-sonnet-5-5"` - Sonnet 5.5 model
* `"claude-sonnet-5"` - Sonnet 5 model
* `"claude-sonnet-4-6"` - Sonnet 4.6 model
* `"claude-sonnet-4-5"` - Sonnet 4.5
* `"claude-sonnet-4-20250514"` - Sonnet 4 dated release
* `"claude-haiku-4-5-20251001"` - Faster, more cost-effective option
Optional reasoning effort for Claude. Omit this field to use the model's default. Available options depend on the selected model:
* `"low"`
* `"medium"`
* `"high"`
* `"xhigh"` — only supported by `"claude-opus-5-5"`, `"claude-fable-5-1"`, `"claude-opus-5"`, `"claude-opus-4-8"`, `"claude-opus-4-7"`, `"claude-sonnet-5-5"`, and `"claude-sonnet-5"`
* `"max"` — supported by the models above, plus `"claude-opus-4-6"` and `"claude-sonnet-4-6"`
`"claude-opus-4-5"` supports `"low"`, `"medium"`, and `"high"` only. Other models, including `"claude-sonnet-4-5"` and `"claude-haiku-4-5-20251001"`, do not support reasoning effort.
Maximum number of actions Claude can take (clicks, typing, navigation, etc.). Increase for complex tasks.
Maximum consecutive failures before the task is aborted.
ID of an existing browser session to reuse. Useful for multi-step workflows that need to maintain the same browser session.
Keep the browser session alive after task completion.
Allow the agent to interact by executing actions on the actual computer not just within the page. Allows the agent to see the entire screen instead of just the page contents.
[Session configuration](/docs/api-reference/start-a-claude-computer-use-task#body-session-options) (proxy, stealth, captcha solving, etc.). Only applies when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Use your own Anthropic API key instead of consuming Hyperbrowser credits for LLM calls. You will only be charged for browser usage.
API key for `anthropic`. Required when `useCustomApiKeys` is `true`.
```typescript theme={null}
{
anthropic: "..."
}
```
The agent may not complete the task within the specified `maxSteps`. If that happens, try increasing the `maxSteps` parameter.
Additionally, the browser session used by the AI Agent will time out based on your team's default Session Timeout settings or the session's `timeoutMinutes` parameter if provided. You can adjust the default Session Timeout in the [Settings page](https://app.hyperbrowser.ai/settings).
`useComputerAction` can often be better for completing tasks but may require more steps. It is especially useful when the agent needs to interact with elements on the page that might not be accessible by or visible to Playwright. Since it allows the agent to see and interact with the entire screen, it is much more powerful. Instead of executing actions with Playwright which can only interact with the page via CDP, computer actions allow the agent to interact directly with computer primitives (direct clicks, typing, scroll, etc.).
## Reuse Browser Sessions
You can pass in an existing `sessionId` to the Claude Computer Use task so that it can execute the task on an existing session. Also, if you want to keep the session open after executing the task, you can supply the `keepBrowserOpen` parameter.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const session = await client.sessions.create();
try {
const result = await client.agents.claudeComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionId: session.id,
keepBrowserOpen: true,
});
console.log(`Output:\n${result.data?.finalResult}`);
const result2 = await client.agents.claudeComputerUse.startAndWait({
task: "Tell me how many upvotes the first post has.",
sessionId: session.id,
});
console.log(`\nOutput:\n${result2.data?.finalResult}`);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.claude_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"llm": "claude-sonnet-4-5",
"session_id": session.id,
"keep_browser_open": True,
}
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.claude_computer_use.start_and_wait(
{
"llm": "claude-sonnet-4-5",
"task": "Tell me how many upvotes the first post has.",
"session_id": session.id,
}
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartClaudeComputerUseTaskParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.claude_computer_use.start_and_wait(
StartClaudeComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
llm="claude-sonnet-4-5",
session_id=session.id,
keep_browser_open=True,
)
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.claude_computer_use.start_and_wait(
StartClaudeComputerUseTaskParams(
llm="claude-sonnet-4-5",
task="Tell me how many upvotes the first post has.",
session_id=session.id,
)
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
Always set `keepBrowserOpen: true` on tasks that you want to reuse the session from. Otherwise, the session will be automatically closed when the task completes.
## Using Your Own API Keys
Bring your own Anthropic API key to avoid consuming Hyperbrowser credits for LLM calls. You'll still be charged for browser session usage, but save on token costs.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.claudeComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
llm: "claude-sonnet-4-5",
useCustomApiKeys: true,
apiKeys: {
anthropic: "",
},
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.claude_computer_use.start_and_wait(
{
"task": "What is the title of the first post on HackerNews today?",
"llm": "claude-sonnet-4-5",
"use_custom_api_keys": True,
"api_keys": {
"anthropic": "",
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import (
StartClaudeComputerUseTaskParams,
ClaudeComputerUseApiKeys,
)
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.claude_computer_use.start_and_wait(
StartClaudeComputerUseTaskParams(
task="What is the title of the first post on HackerNews today?",
llm="claude-sonnet-4-5",
use_custom_api_keys=True,
api_keys=ClaudeComputerUseApiKeys(
anthropic="",
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/claude-computer-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_HYPERBROWSER_API_KEY" \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"llm": "claude-sonnet-4-5",
"useCustomApiKeys": true,
"apiKeys": {
"anthropic": "YOUR_ANTHROPIC_API_KEY"
}
}'
```
## Session Configuration
Customize the browser session used by Claude Computer Use with session options.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.claudeComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
llm: "claude-sonnet-4-5",
sessionOptions: {
acceptCookies: true,
}
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.claude_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"llm": "claude-sonnet-4-5",
"session_options": {
"accept_cookies": True,
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartClaudeComputerUseTaskParams, CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.claude_computer_use.start_and_wait(
StartClaudeComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
llm="claude-sonnet-4-5",
session_options=CreateSessionParams(
accept_cookies=True,
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/claude-computer-use \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"llm": "claude-sonnet-4-5",
"sessionOptions": {
"acceptCookies": true
}
}'
```
`sessionOptions` only applies when creating a new session. If you provide a `sessionId`, these options are ignored.
Proxies and CAPTCHA solving add latency to page navigation. Only enable them when necessary for your use case.
## Best Practices
Be explicit about what you want Claude to do. Instead of "check the website", say "go to example.com, find the pricing page, and extract the cost of the Enterprise plan".
Simple tasks need 10-20 steps. Complex multi-page workflows might need 50+ steps. Monitor failed tasks and adjust accordingly.
It is usually better to split up complex tasks into smaller, more manageable ones and execute them as separate agent calls on the same session.
# Gemini Computer Use
Source: https://hyperbrowser.ai/docs/agents/gemini-computer-use
Automate browser tasks with Google's Gemini computer use capabilities
Gemini Computer Use allows gemini to directly interact with your computer to perform tasks much like a human. This capability allows gemini to move the cursor, click buttons, type text, and navigate the web, thereby automating complex, multi-step workflows.
Hyperbrowser makes it simple to run Gemini Computer Use tasks in managed cloud browsers. Start a task with a single API call, then poll for results or use our SDK's blocking methods that handle everything automatically.
You can view your Gemini Computer Use tasks in the [dashboard](https://app.hyperbrowser.ai/features/agents/gemini-computer-use).
## How It Works
You can use Gemini Computer Use in two ways:
1. **Start and Wait**: SDKs provide a `startAndWait()` method that blocks until the task completes and returns the result
2. **Async Pattern**: Start a task, get a job ID, then poll for status and results—useful for long-running tasks or when you want more control
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The simplest way to run a Gemini Computer Use task is with the `startAndWait()` method, which handles everything for you:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.geminiComputerUse.startAndWait({
task: "Go to Hacker News and tell me the title of the top post",
maxSteps: 20,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.gemini_computer_use.start_and_wait(
params={
"task": "Go to Hacker News and tell me the title of the top post",
"max_steps": 20,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartGeminiComputerUseTaskParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.gemini_computer_use.start_and_wait(
params=StartGeminiComputerUseTaskParams(
task="Go to Hacker News and tell me the title of the top post", max_steps=20
)
)
print(f"Output:\n{result.data.final_result}")
```
```bash cURL theme={null}
# Start the task
curl -X POST https://api.hyperbrowser.ai/api/task/gemini-computer-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"task": "Go to Hacker News and tell me the title of the top post",
"maxSteps": 20
}'
# Response: {"jobId": "abc123", "liveUrl": "https://..."}
# Check status
curl https://api.hyperbrowser.ai/api/task/gemini-computer-use/abc123/status \
-H "x-api-key: YOUR_API_KEY"
# Get full results
curl https://api.hyperbrowser.ai/api/task/gemini-computer-use/abc123 \
-H "x-api-key: YOUR_API_KEY"
```
## Async Pattern
When you need more control, use the async pattern to start a task and poll for results:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
try {
// Start the task
const task = await client.agents.geminiComputerUse.start({
task: "What is the title of the first post on Hacker News today?",
maxSteps: 20,
});
console.log(`Task started: ${task.jobId}`);
console.log(`Watch live: ${task.liveUrl}`);
// Poll for completion
let result;
while (true) {
result = await client.agents.geminiComputerUse.getStatus(task.jobId);
console.log(`Status: ${result.status}`);
if (result.status === "completed" || result.status === "failed") {
break;
}
await new Promise((resolve) => setTimeout(resolve, 5000)); // Wait 5s
}
const fullResult = await client.agents.geminiComputerUse.get(task.jobId);
if (fullResult.status === "completed") {
console.log("Result:", fullResult.data?.finalResult);
console.log("Steps taken:", fullResult.data?.steps?.length);
} else {
console.error("Task failed:", fullResult.error);
}
} catch (err) {
console.error(`Error: ${err.message}`);
}
}
main();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.gemini_computer_use.start(
params={
"task": "What is the title of the first post on Hacker News today?",
"max_steps": 20,
}
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.gemini_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.gemini_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import StartGeminiComputerUseTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.gemini_computer_use.start(
params=StartGeminiComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
max_steps=20,
)
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.gemini_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.gemini_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
## Stop a Running Task
Stop a task before it completes:
```typescript Node.js theme={null}
await client.agents.geminiComputerUse.stop("job-id");
```
```python Python theme={null}
client.agents.gemini_computer_use.stop("job-id")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/task/gemini-computer-use/job-id/stop \
-H "x-api-key: YOUR_API_KEY"
```
## Parameters
Natural language description of what you want Gemini to accomplish. Be specific for best results.
Gemini model to use. Available options:
* `"gemini-3.8-flash"` - Gemini 3.8 Flash (recommended)
* `"gemini-3.7-flash"` - Gemini 3.7 Flash
* `"gemini-3.6-flash"` - Gemini 3.6 Flash
* `"gemini-3.5-flash-lite"` - Gemini 3.5 Flash Lite
* `"gemini-3.5-flash"` - Gemini 3.5 Flash
* `"gemini-3-flash-preview"` - Gemini 3 Flash Preview
Maximum number of actions Gemini can take (clicks, typing, navigation, etc.). Increase for complex tasks.
Maximum consecutive failures before the task is aborted.
ID of an existing browser session to reuse. Useful for multi-step workflows that need to maintain the same browser session.
Keep the browser session alive after task completion.
Allow the agent to interact by executing actions on the actual computer not just within the page. Allows the agent to see the entire screen instead of just the page contents.
[Session configuration](/docs/api-reference/start-a-gemini-computer-use-task#body-session-options) (proxy, stealth, captcha solving, etc.). Only applies when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Use your own Google API key instead of consuming Hyperbrowser credits for LLM calls. You will only be charged for browser usage.
API key for `google`. Required when `useCustomApiKeys` is `true`.
```typescript theme={null}
{
google: "..."
}
```
The agent may not complete the task within the specified `maxSteps`. If that happens, try increasing the `maxSteps` parameter.
Additionally, the browser session used by the AI Agent will time out based on your team's default Session Timeout settings or the session's `timeoutMinutes` parameter if provided. You can adjust the default Session Timeout in the [Settings page](https://app.hyperbrowser.ai/settings).
`useComputerAction` can often be better for completing tasks but may require more steps. It is especially useful when the agent needs to interact with elements on the page that might not be accessible by or visible to Playwright. Since it allows the agent to see and interact with the entire screen, it is much more powerful. Instead of executing actions with Playwright which can only interact with the page via CDP, computer actions allow the agent to interact directly with computer primitives (direct clicks, typing, scroll, etc.).
## Reuse Browser Sessions
You can pass in an existing `sessionId` to the Gemini Computer Use task so that it can execute the task on an existing session. Also, if you want to keep the session open after executing the task, you can supply the `keepBrowserOpen` parameter.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const session = await client.sessions.create();
try {
const result = await client.agents.geminiComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionId: session.id,
keepBrowserOpen: true,
});
console.log(`Output:\n${result.data?.finalResult}`);
const result2 = await client.agents.geminiComputerUse.startAndWait({
task: "Tell me how many upvotes the first post has.",
sessionId: session.id,
});
console.log(`\nOutput:\n${result2.data?.finalResult}`);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.gemini_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_id": session.id,
"keep_browser_open": True,
}
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.gemini_computer_use.start_and_wait(
{
"task": "Tell me how many upvotes the first post has.",
"session_id": session.id,
}
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartGeminiComputerUseTaskParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.gemini_computer_use.start_and_wait(
StartGeminiComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_id=session.id,
keep_browser_open=True,
)
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.gemini_computer_use.start_and_wait(
StartGeminiComputerUseTaskParams(
task="Tell me how many upvotes the first post has.",
session_id=session.id,
)
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
Always set `keepBrowserOpen: true` on tasks that you want to reuse the session from. Otherwise, the session will be automatically closed when the task completes.
## Using Your Own API Keys
Bring your own Google API key to avoid consuming Hyperbrowser credits for LLM calls. You'll still be charged for browser session usage, but save on token costs.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.geminiComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
useCustomApiKeys: true,
apiKeys: {
google: "",
},
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.gemini_computer_use.start_and_wait(
{
"task": "What is the title of the first post on HackerNews today?",
"use_custom_api_keys": True,
"api_keys": {
"google": "",
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import (
StartGeminiComputerUseTaskParams,
GeminiComputerUseApiKeys,
)
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.gemini_computer_use.start_and_wait(
StartGeminiComputerUseTaskParams(
task="What is the title of the first post on HackerNews today?",
use_custom_api_keys=True,
api_keys=GeminiComputerUseApiKeys(
google="",
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/gemini-computer-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_HYPERBROWSER_API_KEY" \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"useCustomApiKeys": true,
"apiKeys": {
"google": "YOUR_GOOGLE_API_KEY"
}
}'
```
## Session Configuration
Customize the browser session used by Gemini Computer Use with session options.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.geminiComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionOptions: {
acceptCookies: true,
}
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.gemini_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_options": {
"accept_cookies": True,
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartGeminiComputerUseTaskParams, CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.gemini_computer_use.start_and_wait(
StartGeminiComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_options=CreateSessionParams(
accept_cookies=True,
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/gemini-computer-use \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"sessionOptions": {
"acceptCookies": true
}
}'
```
`sessionOptions` only applies when creating a new session. If you provide a `sessionId`, these options are ignored.
Proxies and CAPTCHA solving add latency to page navigation. Only enable them when necessary for your use case.
## Best Practices
Be explicit about what you want Gemini to do. Instead of "check the website", say "go to example.com, find the pricing page, and extract the cost of the Enterprise plan".
Simple tasks need 10-20 steps. Complex multi-page workflows might need 50+ steps. Monitor failed tasks and adjust accordingly.
It is usually better to split up complex tasks into smaller, more manageable ones and execute them as separate agent calls on the same session.
# Grok Computer Use
Source: https://hyperbrowser.ai/docs/agents/grok-computer-use
Automate browser tasks with xAI's Grok 4.7 computer use capabilities
Grok Computer Use allows Grok 4.7 to directly interact with a browser to perform tasks much like a human. This capability lets Grok move the cursor, click buttons, type text, and navigate the web, automating complex, multi-step workflows.
Hyperbrowser makes it simple to run Grok Computer Use tasks in managed cloud browsers. Start a task with a single API call, then poll for results or use our SDK's blocking methods that handle everything automatically.
You can view your Grok Computer Use tasks in the [dashboard](https://app.hyperbrowser.ai/features/agents/grok-computer-use).
## How It Works
You can use Grok Computer Use in two ways:
1. **Start and Wait**: SDKs provide a `startAndWait()` method that blocks until the task completes and returns the result
2. **Async Pattern**: Start a task, get a job ID, then poll for status and results—useful for long-running tasks or when you want more control
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The simplest way to run a Grok Computer Use task is with the `startAndWait()` method, which handles everything for you:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.grokComputerUse.startAndWait({
task: "Go to Hacker News and tell me the title of the top post",
maxSteps: 20,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.grok_computer_use.start_and_wait(
params={
"task": "Go to Hacker News and tell me the title of the top post",
"max_steps": 20,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartGrokComputerUseTaskParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.grok_computer_use.start_and_wait(
params=StartGrokComputerUseTaskParams(
task="Go to Hacker News and tell me the title of the top post", max_steps=20
)
)
print(f"Output:\n{result.data.final_result}")
```
```bash cURL theme={null}
# Start the task
curl -X POST https://api.hyperbrowser.ai/api/task/grok-computer-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"task": "Go to Hacker News and tell me the title of the top post",
"maxSteps": 20
}'
# Response: {"jobId": "abc123", "liveUrl": "https://..."}
# Check status
curl https://api.hyperbrowser.ai/api/task/grok-computer-use/abc123/status \
-H "x-api-key: YOUR_API_KEY"
# Get full results
curl https://api.hyperbrowser.ai/api/task/grok-computer-use/abc123 \
-H "x-api-key: YOUR_API_KEY"
```
## Async Pattern
When you need more control, use the async pattern to start a task and poll for results:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
try {
// Start the task
const task = await client.agents.grokComputerUse.start({
task: "What is the title of the first post on Hacker News today?",
maxSteps: 20,
});
console.log(`Task started: ${task.jobId}`);
console.log(`Watch live: ${task.liveUrl}`);
// Poll for completion
let result;
while (true) {
result = await client.agents.grokComputerUse.getStatus(task.jobId);
console.log(`Status: ${result.status}`);
if (result.status === "completed" || result.status === "failed") {
break;
}
await new Promise((resolve) => setTimeout(resolve, 5000)); // Wait 5s
}
const fullResult = await client.agents.grokComputerUse.get(task.jobId);
if (fullResult.status === "completed") {
console.log("Result:", fullResult.data?.finalResult);
console.log("Steps taken:", fullResult.data?.steps?.length);
} else {
console.error("Task failed:", fullResult.error);
}
} catch (err) {
console.error(`Error: ${err.message}`);
}
}
main();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.grok_computer_use.start(
params={
"task": "What is the title of the first post on Hacker News today?",
"max_steps": 20,
}
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.grok_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.grok_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import StartGrokComputerUseTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.grok_computer_use.start(
params=StartGrokComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
max_steps=20,
)
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.grok_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.grok_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
## Stop a Running Task
Stop a task before it completes:
```typescript Node.js theme={null}
await client.agents.grokComputerUse.stop("job-id");
```
```python Python theme={null}
client.agents.grok_computer_use.stop("job-id")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/task/grok-computer-use/job-id/stop \
-H "x-api-key: YOUR_API_KEY"
```
## Parameters
Natural language description of what you want Grok to accomplish. Be specific for best results.
Grok model to use. `"grok-4.7"` is the supported model. `"grok-4.6"` and `"grok-4.5"` are accepted and run as `"grok-4.7"`.
Reasoning effort for Grok. Available options:
* `"low"`
* `"medium"`
* `"high"`
Maximum number of actions Grok can take (clicks, typing, navigation, etc.). Increase for complex tasks.
Maximum consecutive failures before the task is aborted.
ID of an existing browser session to reuse. Useful for multi-step workflows that need to maintain the same browser session.
Keep the browser session alive after task completion.
Allow the agent to interact by executing actions on the actual computer, not just within the page. Allows the agent to see the entire screen instead of just the page contents.
[Session configuration](/docs/api-reference/start-a-grok-computer-use-task#body-session-options) (proxy, stealth, captcha solving, etc.). Only applies when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Use your own xAI API key instead of consuming Hyperbrowser credits for LLM calls. You will only be charged for browser usage.
API key for `xai`. Required when `useCustomApiKeys` is `true`.
```typescript theme={null}
{
xai: "..."
}
```
The agent may not complete the task within the specified `maxSteps`. If that happens, try increasing the `maxSteps` parameter.
Additionally, the browser session used by the AI Agent will time out based on your team's default Session Timeout settings or the session's `timeoutMinutes` parameter if provided. You can adjust the default Session Timeout in the [Settings page](https://app.hyperbrowser.ai/settings).
`useComputerAction` can often be better for completing tasks but may require more steps. It is especially useful when the agent needs to interact with elements on the page that might not be accessible by or visible to Playwright. Since it allows the agent to see and interact with the entire screen, it is much more powerful. Instead of executing actions with Playwright which can only interact with the page via CDP, computer actions allow the agent to interact directly with computer primitives (direct clicks, typing, scroll, etc.).
## Reuse Browser Sessions
You can pass in an existing `sessionId` to the Grok Computer Use task so that it can execute the task on an existing session. Also, if you want to keep the session open after executing the task, you can supply the `keepBrowserOpen` parameter.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const session = await client.sessions.create();
try {
const result = await client.agents.grokComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionId: session.id,
keepBrowserOpen: true,
});
console.log(`Output:\n${result.data?.finalResult}`);
const result2 = await client.agents.grokComputerUse.startAndWait({
task: "Tell me how many upvotes the first post has.",
sessionId: session.id,
});
console.log(`\nOutput:\n${result2.data?.finalResult}`);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.grok_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_id": session.id,
"keep_browser_open": True,
}
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.grok_computer_use.start_and_wait(
{
"task": "Tell me how many upvotes the first post has.",
"session_id": session.id,
}
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartGrokComputerUseTaskParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.grok_computer_use.start_and_wait(
StartGrokComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_id=session.id,
keep_browser_open=True,
)
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.grok_computer_use.start_and_wait(
StartGrokComputerUseTaskParams(
task="Tell me how many upvotes the first post has.",
session_id=session.id,
)
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
Always set `keepBrowserOpen: true` on tasks that you want to reuse the session from. Otherwise, the session will be automatically closed when the task completes.
## Using Your Own API Keys
Bring your own xAI API key to avoid consuming Hyperbrowser credits for LLM calls. You'll still be charged for browser session usage, but save on token costs.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.grokComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
useCustomApiKeys: true,
apiKeys: {
xai: "",
},
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.grok_computer_use.start_and_wait(
{
"task": "What is the title of the first post on HackerNews today?",
"use_custom_api_keys": True,
"api_keys": {
"xai": "",
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartGrokComputerUseTaskParams, GrokComputerUseApiKeys
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.grok_computer_use.start_and_wait(
StartGrokComputerUseTaskParams(
task="What is the title of the first post on HackerNews today?",
use_custom_api_keys=True,
api_keys=GrokComputerUseApiKeys(
xai="",
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/grok-computer-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_HYPERBROWSER_API_KEY" \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"useCustomApiKeys": true,
"apiKeys": {
"xai": "YOUR_XAI_API_KEY"
}
}'
```
## Session Configuration
Customize the browser session used by Grok Computer Use with session options.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.grokComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionOptions: {
acceptCookies: true,
}
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.grok_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_options": {
"accept_cookies": True,
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartGrokComputerUseTaskParams, CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.grok_computer_use.start_and_wait(
StartGrokComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_options=CreateSessionParams(
accept_cookies=True,
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/grok-computer-use \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"sessionOptions": {
"acceptCookies": true
}
}'
```
`sessionOptions` only applies when creating a new session. If you provide a `sessionId`, these options are ignored.
Proxies and CAPTCHA solving add latency to page navigation. Only enable them when necessary for your use case.
## Best Practices
Be explicit about what you want Grok to do. Instead of "check the website", say "go to example.com, find the pricing page, and extract the cost of the Enterprise plan".
Simple tasks need 10-20 steps. Complex multi-page workflows might need 50+ steps. Monitor failed tasks and adjust accordingly.
It is usually better to split up complex tasks into smaller, more manageable ones and execute them as separate agent calls on the same session.
# HyperAgent
Source: https://hyperbrowser.ai/docs/agents/hyperagent
Execute AI agent tasks using HyperAgent, our open-source Playwright-powered framework
This page covers **running HyperAgent via the Hyperbrowser Cloud API**. For using the HyperAgent SDK directly in your own code (local or self-hosted), see the [HyperAgent SDK documentation](/docs/hyperagent/overview).
HyperAgent is our open-source tool that supercharges Playwright with AI. To view the full details on HyperAgent, check out the [HyperAgent Github Repo](https://github.com/hyperbrowserai/HyperAgent). Here, we will just go over using one of the features of HyperAgent which is being able to have it automatically execute tasks on your behalf on the web with just a simple call.
By default, HyperAgent tasks are handled in an asynchronous manner of first starting the task and then checking its status until it is completed. However, if you don't want to handle the monitoring yourself, our SDKs provide a simple function that handles the whole flow and returns the data once the task is completed.
You can view your HyperAgent tasks in the [dashboard](https://app.hyperbrowser.ai/features/agents/hyper-agent).
## How It Works
You can use HyperAgent in two ways:
1. **Start and Wait**: SDKs provide a `startAndWait()` method that blocks until the task completes and returns the result
2. **Async Pattern**: Start a task, get a job ID, then poll for status and results—useful for long-running tasks or when you want more control
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The simplest way to run a HyperAgent task is with the `startAndWait()` method, which handles everything for you:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.hyperAgent.startAndWait({
version: "1.1.0",
task: "Go to Hacker News and tell me the title of the top post",
llm: "gemini-3-flash-preview",
maxSteps: 20,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.hyper_agent.start_and_wait(
params={
"version": "1.1.0",
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "gemini-3-flash-preview",
"max_steps": 20,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartHyperAgentTaskParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.hyper_agent.start_and_wait(
params=StartHyperAgentTaskParams(
version="1.1.0",
task="Go to Hacker News and tell me the title of the top post",
llm="gemini-3-flash-preview",
max_steps=20,
)
)
print(f"Output:\n{result.data.final_result}")
```
```bash cURL theme={null}
# Start the task
curl -X POST https://api.hyperbrowser.ai/api/task/hyper-agent \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "gpt-4o",
"maxSteps": 20
}'
# Response: {"jobId": "abc123", "liveUrl": "https://..."}
# Check status
curl https://api.hyperbrowser.ai/api/task/hyper-agent/abc123/status \
-H "x-api-key: YOUR_API_KEY"
# Get full results
curl https://api.hyperbrowser.ai/api/task/hyper-agent/abc123 \
-H "x-api-key: YOUR_API_KEY"
```
## Async Pattern
When you need more control, use the async pattern to start a task and poll for results:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
try {
// Start the task
const task = await client.agents.hyperAgent.start({
version: "1.1.0",
task: "What is the title of the first post on Hacker News today?",
llm: "gemini-3-flash-preview",
maxSteps: 20,
});
console.log(`Task started: ${task.jobId}`);
console.log(`Watch live: ${task.liveUrl}`);
// Poll for completion
let result;
while (true) {
result = await client.agents.hyperAgent.getStatus(task.jobId);
console.log(`Status: ${result.status}`);
if (result.status === "completed" || result.status === "failed") {
break;
}
await new Promise((resolve) => setTimeout(resolve, 5000)); // Wait 5s
}
const fullResult = await client.agents.hyperAgent.get(task.jobId);
if (fullResult.status === "completed") {
console.log("Result:", fullResult.data?.finalResult);
console.log("Steps taken:", fullResult.data?.steps?.length);
} else {
console.error("Task failed:", fullResult.error);
}
} catch (err) {
console.error(`Error: ${err.message}`);
}
}
main();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.hyper_agent.start(
params={
"version": "1.1.0",
"task": "What is the title of the first post on Hacker News today?",
"llm": "gemini-3-flash-preview",
"max_steps": 20,
}
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.hyper_agent.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.hyper_agent.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import StartHyperAgentTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.hyper_agent.start(
params=StartHyperAgentTaskParams(
version="1.1.0",
task="What is the title of the first post on Hacker News today?",
llm="gemini-3-flash-preview",
max_steps=20,
)
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.hyper_agent.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.hyper_agent.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
## Stop a Running Task
Stop a task before it completes:
```typescript Node.js theme={null}
await client.agents.hyperAgent.stop("job-id");
```
```python Python theme={null}
client.agents.hyper_agent.stop("job-id")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/task/hyper-agent/job-id/stop \
-H "x-api-key: YOUR_API_KEY"
```
## Parameters
Natural language description of what you want HyperAgent to accomplish. Be specific for best results.
Version of HyperAgent to use.
Options: `0.8.0`, `1.1.0`
We highly recommend using the `1.1.0` version of HyperAgent.
LLM model to use. Available options:
* `"gpt-5.6-luna"` - GPT-5.6 Luna
* `"gpt-5.5"` - GPT-5.5
* `"gpt-5.2"` - GPT-5.2
* `"gpt-5.1"` - GPT-5.1
* `"gpt-5"` - GPT-5
* `"gpt-5-mini"` - GPT-5 Mini
* `"gpt-4o"` - GPT-4o (default)
* `"gpt-4o-mini"` - GPT-4o Mini
* `"gpt-4.1"` - GPT-4.1
* `"gpt-4.1-mini"` - GPT-4.1 Mini
* `"claude-sonnet-5"` - Claude Sonnet 5
* `"claude-sonnet-4-6"` - Claude Sonnet 4.6
* `"claude-sonnet-4-5"` - Claude Sonnet 4.5
* `"gemini-3.8-flash"` - Gemini 3.8 Flash
* `"gemini-3.7-flash"` - Gemini 3.7 Flash
* `"gemini-3-flash-preview"` - Gemini 3 Flash Preview
Claude and Gemini models are not supported on the `0.8.0` version of HyperAgent.
Maximum number of actions HyperAgent can take (clicks, typing, navigation, etc.). Increase for complex tasks.
ID of an existing browser session to reuse. Useful for multi-step workflows that need to maintain the same browser session.
Keep the browser session alive after task completion.
[Session configuration](/docs/api-reference/start-a-hyperagent-task#body-session-options) (proxy, stealth, captcha solving, etc.). Only applies when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Use your own LLM API keys instead of consuming Hyperbrowser credits for LLM calls. You will only be charged for browser usage.
API keys for `openai`, `anthropic`, and `google`. Required when `useCustomApiKeys` is `true`. Must provide keys based on the LLM you are using.
```typescript theme={null}
{
openai: "...",
anthropic: "...",
google: "..."
}
```
The agent may not complete the task within the specified `maxSteps`. If that happens, try increasing the `maxSteps` parameter.
Additionally, the browser session used by the AI Agent will time out based on your team's default Session Timeout settings or the session's `timeoutMinutes` parameter if provided. You can adjust the default Session Timeout in the [Settings page](https://app.hyperbrowser.ai/settings).
## Reuse Browser Sessions
You can pass in an existing `sessionId` to the HyperAgent task so that it can execute the task on an existing session. Also, if you want to keep the session open after executing the task, you can supply the `keepBrowserOpen` parameter.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const session = await client.sessions.create();
try {
const result = await client.agents.hyperAgent.startAndWait({
version: "1.1.0",
task: "What is the title of the first post on Hacker News today?",
llm: "gemini-3-flash-preview",
sessionId: session.id,
keepBrowserOpen: true,
});
console.log(`Output:\n${result.data?.finalResult}`);
const result2 = await client.agents.hyperAgent.startAndWait({
version: "1.1.0",
llm: "gemini-3-flash-preview",
task: "Tell me how many upvotes the first post has.",
sessionId: session.id,
});
console.log(`\nOutput:\n${result2.data?.finalResult}`);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.hyper_agent.start_and_wait(
{
"version": "1.1.0",
"task": "What is the title of the first post on Hacker News today?",
"llm": "gemini-3-flash-preview",
"session_id": session.id,
"keep_browser_open": True,
}
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.hyper_agent.start_and_wait(
{
"version": "1.1.0",
"llm": "gemini-3-flash-preview",
"task": "Tell me how many upvotes the first post has.",
"session_id": session.id,
}
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartHyperAgentTaskParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.hyper_agent.start_and_wait(
StartHyperAgentTaskParams(
version="1.1.0",
task="What is the title of the first post on Hacker News today?",
llm="gemini-3-flash-preview",
session_id=session.id,
keep_browser_open=True,
)
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.hyper_agent.start_and_wait(
StartHyperAgentTaskParams(
version="1.1.0",
llm="gemini-3-flash-preview",
task="Tell me how many upvotes the first post has.",
session_id=session.id,
)
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
Always set `keepBrowserOpen: true` on tasks that you want to reuse the session from. Otherwise, the session will be automatically closed when the task completes.
## Using Your Own API Keys
Bring your own LLM API keys to avoid consuming Hyperbrowser credits for LLM calls. You'll still be charged for browser session usage, but save on token costs.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.hyperAgent.startAndWait({
version: "1.1.0",
task: "What is the title of the first post on Hacker News today?",
llm: "gpt-5.2",
useCustomApiKeys: true,
apiKeys: {
openai: "",
},
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.hyper_agent.start_and_wait(
{
"version": "1.1.0",
"task": "What is the title of the first post on HackerNews today?",
"llm": "gpt-5.2",
"use_custom_api_keys": True,
"api_keys": {
"openai": "",
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartHyperAgentTaskParams, HyperAgentApiKeys
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.hyper_agent.start_and_wait(
StartHyperAgentTaskParams(
version="1.1.0",
task="What is the title of the first post on HackerNews today?",
llm="gpt-5.2",
use_custom_api_keys=True,
api_keys=HyperAgentApiKeys(
openai="",
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/hyper-agent \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_HYPERBROWSER_API_KEY" \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"llm": "gpt-4o",
"useCustomApiKeys": true,
"apiKeys": {
"openai": "YOUR_OPENAI_API_KEY"
}
}'
```
## Session Configuration
Customize the browser session used by HyperAgent with session options.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.hyperAgent.startAndWait({
version: "1.1.0",
task: "What is the title of the first post on Hacker News today?",
llm: "gemini-3-flash-preview",
sessionOptions: {
acceptCookies: true,
}
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.hyper_agent.start_and_wait(
{
"version": "1.1.0",
"task": "What is the title of the first post on Hacker News today?",
"llm": "gemini-3-flash-preview",
"session_options": {
"accept_cookies": True,
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartHyperAgentTaskParams, CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.hyper_agent.start_and_wait(
StartHyperAgentTaskParams(
version="1.1.0",
task="What is the title of the first post on Hacker News today?",
llm="gemini-3-flash-preview",
session_options=CreateSessionParams(
accept_cookies=True,
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/hyper-agent \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"llm": "gpt-4o",
"sessionOptions": {
"acceptCookies": true
}
}'
```
`sessionOptions` only applies when creating a new session. If you provide a `sessionId`, these options are ignored.
Proxies and CAPTCHA solving add latency to page navigation. Only enable them when necessary for your use case.
## Best Practices
Be explicit about what you want HyperAgent to do. Instead of "check the website", say "go to example.com, find the pricing page, and extract the cost of the Enterprise plan".
Simple tasks need 10-20 steps. Complex multi-page workflows might need 50+ steps. Monitor failed tasks and adjust accordingly.
It is usually better to split up complex tasks into smaller, more manageable ones and execute them as separate agent calls on the same session.
# Jev Computer Use
Source: https://hyperbrowser.ai/docs/agents/jev-computer-use
Automate browser tasks with Jev's TypeSafe decision models
Jev Computer Use is a native CDP browser agent. It observes the current page, chooses one operation and target, then executes that action. Jev works from page text and DOM controls rather than screenshots.
Hyperbrowser runs Jev tasks in managed cloud browsers. Start a task with a single API call, then poll for results or use our SDK's blocking methods that handle everything automatically.
You can view your Jev tasks in the [dashboard](https://app.hyperbrowser.ai/features/agents/jev).
## How It Works
You can use Jev in two ways:
1. **Start and Wait**: SDKs provide a `startAndWait()` method that blocks until the task completes and returns the result
2. **Async Pattern**: Start a task, get a job ID, then poll for status and results—useful for long-running tasks or when you want more control
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The simplest way to run a Jev task is with the `startAndWait()` method, which handles everything for you:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.jevComputerUse.startAndWait({
task: "Go to Hacker News and tell me the title of the top post",
maxSteps: 20,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.jev_computer_use.start_and_wait(
params={
"task": "Go to Hacker News and tell me the title of the top post",
"max_steps": 20,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartJevComputerUseTaskParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.jev_computer_use.start_and_wait(
params=StartJevComputerUseTaskParams(
task="Go to Hacker News and tell me the title of the top post", max_steps=20
)
)
print(f"Output:\n{result.data.final_result}")
```
```bash cURL theme={null}
# Start the task
curl -X POST https://api.hyperbrowser.ai/api/task/jev \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"task": "Go to Hacker News and tell me the title of the top post",
"maxSteps": 20
}'
# Response: {"jobId": "abc123", "liveUrl": "https://..."}
# Check status
curl https://api.hyperbrowser.ai/api/task/jev/abc123/status \
-H "x-api-key: YOUR_API_KEY"
# Get full results
curl https://api.hyperbrowser.ai/api/task/jev/abc123 \
-H "x-api-key: YOUR_API_KEY"
```
## Async Pattern
When you need more control, use the async pattern to start a task and poll for results:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
try {
// Start the task
const task = await client.agents.jevComputerUse.start({
task: "What is the title of the first post on Hacker News today?",
maxSteps: 20,
});
console.log(`Task started: ${task.jobId}`);
console.log(`Watch live: ${task.liveUrl}`);
// Poll for completion
let result;
while (true) {
result = await client.agents.jevComputerUse.getStatus(task.jobId);
console.log(`Status: ${result.status}`);
if (result.status === "completed" || result.status === "failed") {
break;
}
await new Promise((resolve) => setTimeout(resolve, 5000)); // Wait 5s
}
const fullResult = await client.agents.jevComputerUse.get(task.jobId);
if (fullResult.status === "completed") {
console.log("Result:", fullResult.data?.finalResult);
console.log("Steps taken:", fullResult.data?.steps?.length);
} else {
console.error("Task failed:", fullResult.error);
}
} catch (err) {
console.error(`Error: ${err.message}`);
}
}
main();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.jev_computer_use.start(
params={
"task": "What is the title of the first post on Hacker News today?",
"max_steps": 20,
}
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.jev_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.jev_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import StartJevComputerUseTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.jev_computer_use.start(
params=StartJevComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
max_steps=20,
)
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.jev_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.jev_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
## Stop a Running Task
Stop a task before it completes:
```typescript Node.js theme={null}
await client.agents.jevComputerUse.stop("job-id");
```
```python Python theme={null}
client.agents.jev_computer_use.stop("job-id")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/task/jev/job-id/stop \
-H "x-api-key: YOUR_API_KEY"
```
## Parameters
Natural language description of what you want Jev to accomplish. Be specific for best results. The task must be at most 8000 UTF-8 bytes.
Jev decision model to use. Available options:
* `"jev-1.13.0"` - Pinned Jev 1.13.0 decision model (default)
* `"jev-latest"` - Latest hosted Jev decision model
Text helper model used for field values and the final result. Currently `"gemini-3.5-flash-lite"` is the only supported option.
Maximum number of executed policy actions. Allowed range is 1-300.
Accepted for compatibility with other agents. Jev does not add controller-level retries from this value.
ID of an existing browser session to reuse. Useful for multi-step workflows that need to maintain the same browser session.
Keep the browser session alive after task completion.
[Session configuration](/docs/api-reference/start-a-jev-computer-use-task#body-session-options) (proxy, stealth, captcha solving, etc.). Only applies when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Use your own Jev and Google API keys instead of consuming Hyperbrowser credits for model calls. You will only be charged for browser usage.
API keys for `jev` and `google`. Both are required when `useCustomApiKeys` is `true`.
```typescript theme={null}
{
jev: "...",
google: "..."
}
```
The agent may not complete the task within the specified `maxSteps`. If that happens, try increasing the `maxSteps` parameter.
Additionally, the browser session used by the AI Agent will time out based on your team's default Session Timeout settings or the session's `timeoutMinutes` parameter if provided. You can adjust the default Session Timeout in the [Settings page](https://app.hyperbrowser.ai/settings).
Jev does not send screenshots to either model. It reads visible page text and common HTML/ARIA controls, then dispatches native CDP clicks, typing, and selects.
## Reuse Browser Sessions
You can pass in an existing `sessionId` to the Jev task so that it can execute the task on an existing session. Also, if you want to keep the session open after executing the task, you can supply the `keepBrowserOpen` parameter.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const session = await client.sessions.create();
try {
const result = await client.agents.jevComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionId: session.id,
keepBrowserOpen: true,
});
console.log(`Output:\n${result.data?.finalResult}`);
const result2 = await client.agents.jevComputerUse.startAndWait({
task: "Tell me how many upvotes the first post has.",
sessionId: session.id,
});
console.log(`\nOutput:\n${result2.data?.finalResult}`);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.jev_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_id": session.id,
"keep_browser_open": True,
}
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.jev_computer_use.start_and_wait(
{
"task": "Tell me how many upvotes the first post has.",
"session_id": session.id,
}
)
print(f"\nOutput:\n{resp2.data.final_result}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartJevComputerUseTaskParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.jev_computer_use.start_and_wait(
StartJevComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_id=session.id,
keep_browser_open=True,
)
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.jev_computer_use.start_and_wait(
StartJevComputerUseTaskParams(
task="Tell me how many upvotes the first post has.",
session_id=session.id,
)
)
print(f"\nOutput:\n{resp2.data.final_result}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
Always set `keepBrowserOpen: true` on tasks that you want to reuse the session from. Otherwise, the session will be automatically closed when the task completes.
## Using Your Own API Keys
Bring your own Jev and Google API keys to avoid consuming Hyperbrowser credits for model calls. You'll still be charged for browser session usage, but save on token costs. Jev BYOK requires both keys: `jev` for decisions and `google` for text generation.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.jevComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
useCustomApiKeys: true,
apiKeys: {
jev: "",
google: "",
},
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.jev_computer_use.start_and_wait(
{
"task": "What is the title of the first post on HackerNews today?",
"use_custom_api_keys": True,
"api_keys": {
"jev": "",
"google": "",
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartJevComputerUseTaskParams, JevComputerUseApiKeys
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.jev_computer_use.start_and_wait(
StartJevComputerUseTaskParams(
task="What is the title of the first post on HackerNews today?",
use_custom_api_keys=True,
api_keys=JevComputerUseApiKeys(
jev="",
google="",
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/jev \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_HYPERBROWSER_API_KEY" \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"useCustomApiKeys": true,
"apiKeys": {
"jev": "YOUR_JEV_API_KEY",
"google": "YOUR_GOOGLE_API_KEY"
}
}'
```
## Session Configuration
Customize the browser session used by Jev with session options.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.jevComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionOptions: {
acceptCookies: true,
}
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.jev_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_options": {
"accept_cookies": True,
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartJevComputerUseTaskParams, CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.jev_computer_use.start_and_wait(
StartJevComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_options=CreateSessionParams(
accept_cookies=True,
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/jev \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"sessionOptions": {
"acceptCookies": true
}
}'
```
`sessionOptions` only applies when creating a new session. If you provide a `sessionId`, these options are ignored.
Proxies and CAPTCHA solving add latency to page navigation. Only enable them when necessary for your use case.
## Best Practices
Be explicit about what you want Jev to do. Instead of "check the website", say "go to example.com, find the pricing page, and extract the cost of the Enterprise plan".
Simple tasks often finish in well under the default of 100 steps. Complex multi-page workflows can use up to 300. Monitor failed tasks and adjust accordingly.
It is usually better to split up complex tasks into smaller, more manageable ones and execute them as separate agent calls on the same session.
# Meta Computer Use
Source: https://hyperbrowser.ai/docs/agents/meta-computer-use
Automate browser tasks with Meta's Muse Spark computer use capabilities
Meta Computer Use allows Muse Spark to directly interact with a browser to perform tasks much like a human. This capability lets the agent move the cursor, click buttons, type text, and navigate the web, automating complex, multi-step workflows.
Hyperbrowser makes it simple to run Meta Computer Use tasks in managed cloud browsers. Start a task with a single API call, then poll for results or use our SDK's blocking methods that handle everything automatically.
You can view your Meta Computer Use tasks in the [dashboard](https://app.hyperbrowser.ai/features/agents/meta-computer-use).
## How It Works
You can use Meta Computer Use in two ways:
1. **Start and Wait**: SDKs provide a `startAndWait()` method that blocks until the task completes and returns the result
2. **Async Pattern**: Start a task, get a job ID, then poll for status and results—useful for long-running tasks or when you want more control
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The simplest way to run a Meta Computer Use task is with the `startAndWait()` method, which handles everything for you:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.metaComputerUse.startAndWait({
task: "Go to Hacker News and tell me the title of the top post",
maxSteps: 20,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.meta_computer_use.start_and_wait(
params={
"task": "Go to Hacker News and tell me the title of the top post",
"max_steps": 20,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartMetaComputerUseTaskParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.meta_computer_use.start_and_wait(
params=StartMetaComputerUseTaskParams(
task="Go to Hacker News and tell me the title of the top post", max_steps=20
)
)
print(f"Output:\n{result.data.final_result}")
```
```bash cURL theme={null}
# Start the task
curl -X POST https://api.hyperbrowser.ai/api/task/meta-computer-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"task": "Go to Hacker News and tell me the title of the top post",
"maxSteps": 20
}'
# Response: {"jobId": "abc123", "liveUrl": "https://..."}
# Check status
curl https://api.hyperbrowser.ai/api/task/meta-computer-use/abc123/status \
-H "x-api-key: YOUR_API_KEY"
# Get full results
curl https://api.hyperbrowser.ai/api/task/meta-computer-use/abc123 \
-H "x-api-key: YOUR_API_KEY"
```
## Async Pattern
When you need more control, use the async pattern to start a task and poll for results:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
try {
// Start the task
const task = await client.agents.metaComputerUse.start({
task: "What is the title of the first post on Hacker News today?",
maxSteps: 20,
});
console.log(`Task started: ${task.jobId}`);
console.log(`Watch live: ${task.liveUrl}`);
// Poll for completion
let result;
while (true) {
result = await client.agents.metaComputerUse.getStatus(task.jobId);
console.log(`Status: ${result.status}`);
if (result.status === "completed" || result.status === "failed") {
break;
}
await new Promise((resolve) => setTimeout(resolve, 5000)); // Wait 5s
}
const fullResult = await client.agents.metaComputerUse.get(task.jobId);
if (fullResult.status === "completed") {
console.log("Result:", fullResult.data?.finalResult);
console.log("Steps taken:", fullResult.data?.steps?.length);
} else {
console.error("Task failed:", fullResult.error);
}
} catch (err) {
console.error(`Error: ${err.message}`);
}
}
main();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.meta_computer_use.start(
params={
"task": "What is the title of the first post on Hacker News today?",
"max_steps": 20,
}
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.meta_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.meta_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import StartMetaComputerUseTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.meta_computer_use.start(
params=StartMetaComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
max_steps=20,
)
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.meta_computer_use.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.meta_computer_use.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
## Stop a Running Task
Stop a task before it completes:
```typescript Node.js theme={null}
await client.agents.metaComputerUse.stop("job-id");
```
```python Python theme={null}
client.agents.meta_computer_use.stop("job-id")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/task/meta-computer-use/job-id/stop \
-H "x-api-key: YOUR_API_KEY"
```
## Parameters
Natural language description of what you want the agent to accomplish. Be specific for best results.
Meta model to use. Available options:
* `"muse-spark-1.3"`
* `"muse-spark-1.2"`
* `"muse-spark-1.1"`
Reasoning effort for Muse Spark. Available options:
* `"minimal"`
* `"low"`
* `"medium"`
* `"high"`
* `"xhigh"`
* `"max"` — only supported by `"muse-spark-1.3"`
Maximum number of actions the agent can take (clicks, typing, navigation, etc.). Increase for complex tasks.
Maximum consecutive failures before the task is aborted.
ID of an existing browser session to reuse. Useful for multi-step workflows that need to maintain the same browser session.
Keep the browser session alive after task completion.
Allow the agent to interact by executing actions on the actual computer, not just within the page. Allows the agent to see the entire screen instead of just the page contents.
[Session configuration](/docs/api-reference/start-a-meta-computer-use-task#body-session-options) (proxy, stealth, captcha solving, etc.). Only applies when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Use your own Meta API key instead of consuming Hyperbrowser credits for LLM calls. You will only be charged for browser usage.
API key for `meta`. Required when `useCustomApiKeys` is `true`.
```typescript theme={null}
{
meta: "..."
}
```
The agent may not complete the task within the specified `maxSteps`. If that happens, try increasing the `maxSteps` parameter.
Additionally, the browser session used by the AI Agent will time out based on your team's default Session Timeout settings or the session's `timeoutMinutes` parameter if provided. You can adjust the default Session Timeout in the [Settings page](https://app.hyperbrowser.ai/settings).
`useComputerAction` can often be better for completing tasks but may require more steps. It is especially useful when the agent needs to interact with elements on the page that might not be accessible by or visible to Playwright. Since it allows the agent to see and interact with the entire screen, it is much more powerful. Instead of executing actions with Playwright which can only interact with the page via CDP, computer actions allow the agent to interact directly with computer primitives (direct clicks, typing, scroll, etc.).
## Reuse Browser Sessions
You can pass in an existing `sessionId` to the Meta Computer Use task so that it can execute the task on an existing session. Also, if you want to keep the session open after executing the task, you can supply the `keepBrowserOpen` parameter.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const session = await client.sessions.create();
try {
const result = await client.agents.metaComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionId: session.id,
keepBrowserOpen: true,
});
console.log(`Output:\n${result.data?.finalResult}`);
const result2 = await client.agents.metaComputerUse.startAndWait({
task: "Tell me how many upvotes the first post has.",
sessionId: session.id,
});
console.log(`\nOutput:\n${result2.data?.finalResult}`);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.meta_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_id": session.id,
"keep_browser_open": True,
}
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.meta_computer_use.start_and_wait(
{
"task": "Tell me how many upvotes the first post has.",
"session_id": session.id,
}
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartMetaComputerUseTaskParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.meta_computer_use.start_and_wait(
StartMetaComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_id=session.id,
keep_browser_open=True,
)
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.meta_computer_use.start_and_wait(
StartMetaComputerUseTaskParams(
task="Tell me how many upvotes the first post has.",
session_id=session.id,
)
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
Always set `keepBrowserOpen: true` on tasks that you want to reuse the session from. Otherwise, the session will be automatically closed when the task completes.
## Using Your Own API Keys
Bring your own Meta API key to avoid consuming Hyperbrowser credits for LLM calls. You'll still be charged for browser session usage, but save on token costs.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.metaComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
useCustomApiKeys: true,
apiKeys: {
meta: "",
},
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.meta_computer_use.start_and_wait(
{
"task": "What is the title of the first post on HackerNews today?",
"use_custom_api_keys": True,
"api_keys": {
"meta": "",
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartMetaComputerUseTaskParams, MetaComputerUseApiKeys
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.meta_computer_use.start_and_wait(
StartMetaComputerUseTaskParams(
task="What is the title of the first post on HackerNews today?",
use_custom_api_keys=True,
api_keys=MetaComputerUseApiKeys(
meta="",
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/meta-computer-use \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_HYPERBROWSER_API_KEY" \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"useCustomApiKeys": true,
"apiKeys": {
"meta": "YOUR_META_API_KEY"
}
}'
```
## Session Configuration
Customize the browser session used by Meta Computer Use with session options.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.metaComputerUse.startAndWait({
task: "What is the title of the first post on Hacker News today?",
sessionOptions: {
acceptCookies: true,
}
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.meta_computer_use.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"session_options": {
"accept_cookies": True,
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartMetaComputerUseTaskParams, CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.meta_computer_use.start_and_wait(
StartMetaComputerUseTaskParams(
task="What is the title of the first post on Hacker News today?",
session_options=CreateSessionParams(
accept_cookies=True,
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/meta-computer-use \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"sessionOptions": {
"acceptCookies": true
}
}'
```
`sessionOptions` only applies when creating a new session. If you provide a `sessionId`, these options are ignored.
Proxies and CAPTCHA solving add latency to page navigation. Only enable them when necessary for your use case.
## Best Practices
Be explicit about what you want Meta Computer Use to do. Instead of "check the website", say "go to example.com, find the pricing page, and extract the cost of the Enterprise plan".
Simple tasks need 10-20 steps. Complex multi-page workflows might need 50+ steps. Monitor failed tasks and adjust accordingly.
It is usually better to split up complex tasks into smaller, more manageable ones and execute them as separate agent calls on the same session.
# OpenAI CUA
Source: https://hyperbrowser.ai/docs/agents/openai-cua
Execute AI agent tasks using OpenAI's Computer-Using Agent
OpenAI's Computer-Using Agent (CUA) is the AI model powering Operator—OpenAI's agent that can navigate websites, fill forms, and complete multi-step workflows in a browser. CUA interacts with web interfaces like a human, clicking buttons, typing text, and handling complex tasks without needing specialized APIs.
With Hyperbrowser, you can leverage CUA to automate browser tasks with a simple API call. Simply provide a task description, and CUA handles the rest—navigating pages, extracting information, and completing your objective.
You can view your CUA tasks in the [dashboard](https://app.hyperbrowser.ai/features/agents/openai-cua).
## How It Works
You can use CUA in two ways:
1. **Start and Wait**: SDKs provide a `startAndWait()` method that blocks until the task completes and returns the result
2. **Async Pattern**: Start a task, get a job ID, then poll for status and results—useful for long-running tasks or when you want more control
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The simplest way to run a CUA task is with the `startAndWait()` method, which handles everything for you:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.cua.startAndWait({
task: "Go to Hacker News and tell me the title of the top post",
llm: "gpt-5.4",
useComputerAction: true,
maxSteps: 25,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.cua.start_and_wait(
params={
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "gpt-5.4",
"use_computer_action": True,
"max_steps": 25,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartCuaTaskParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.cua.start_and_wait(
params=StartCuaTaskParams(
task="Go to Hacker News and tell me the title of the top post",
llm="gpt-5.4",
use_computer_action=True,
max_steps=25,
)
)
print(f"Output:\n{result.data.final_result}")
```
```bash cURL theme={null}
# Start the task
curl -X POST https://api.hyperbrowser.ai/api/task/cua \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "gpt-5.4",
"useComputerAction": true,
"maxSteps": 25
}'
# Response: {"jobId": "abc123", "liveUrl": "https://..."}
# Check status
curl https://api.hyperbrowser.ai/api/task/cua/abc123/status \
-H "x-api-key: YOUR_API_KEY"
# Get full results
curl https://api.hyperbrowser.ai/api/task/cua/abc123 \
-H "x-api-key: YOUR_API_KEY"
```
## Async Pattern
When you need more control, use the async pattern to start a task and poll for results:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
try {
// Start the task
const task = await client.agents.cua.start({
task: "What is the title of the first post on Hacker News today?",
llm: "gpt-5.4",
useComputerAction: true,
maxSteps: 25,
});
console.log(`Task started: ${task.jobId}`);
console.log(`Watch live: ${task.liveUrl}`);
// Poll for completion
let result;
while (true) {
result = await client.agents.cua.getStatus(task.jobId);
console.log(`Status: ${result.status}`);
if (result.status === "completed" || result.status === "failed") {
break;
}
await new Promise((resolve) => setTimeout(resolve, 5000)); // Wait 5s
}
const fullResult = await client.agents.cua.get(task.jobId);
if (fullResult.status === "completed") {
console.log("Result:", fullResult.data?.finalResult);
console.log("Steps taken:", fullResult.data?.steps?.length);
} else {
console.error("Task failed:", fullResult.error);
}
} catch (err) {
console.error(`Error: ${err.message}`);
}
}
main();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.cua.start(
params={
"task": "What is the title of the first post on Hacker News today?",
"llm": "gpt-5.4",
"use_computer_action": True,
"max_steps": 25,
}
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.cua.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.cua.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import StartCuaTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
try:
# Start the task
task = await client.agents.cua.start(
params=StartCuaTaskParams(
task="What is the title of the first post on Hacker News today?",
llm="gpt-5.4",
use_computer_action=True,
max_steps=25,
)
)
print(f"Task started: {task.job_id}")
print(f"Watch live: {task.live_url}")
# Poll for completion
while True:
result = await client.agents.cua.get_status(task.job_id)
print(f"Status: {result.status}")
if result.status in ["completed", "failed"]:
break
await asyncio.sleep(5) # Wait 5s
full_result = await client.agents.cua.get(task.job_id)
if full_result.status == "completed":
print("Result:", full_result.data.final_result)
print(
"Steps taken:",
len(full_result.data.steps) if full_result.data.steps else 0,
)
else:
print("Task failed:", full_result.error)
except Exception as e:
print(f"Error: {e}")
if __name__ == "__main__":
asyncio.run(main())
```
## Stop a Running Task
Stop a task before it completes:
```typescript Node.js theme={null}
await client.agents.cua.stop("job-id");
```
```python Python theme={null}
client.agents.cua.stop("job-id")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/task/cua/job-id/stop \
-H "x-api-key: YOUR_API_KEY"
```
## Parameters
Natural language description of what you want CUA to accomplish. Be specific for best results.
OpenAI model to use. Available options:
* `"computer-use-preview"` - OpenAI's Computer-Using Agent model (default)
* `"gpt-6-astra"` - GPT-6 Astra computer-use model, OpenAI's most capable model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-6.1-sol"` - GPT-6.1 Sol computer-use model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-6-sol"` - GPT-6 Sol computer-use model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-6-luna"` - GPT-6 Luna computer-use model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-5.6-sol"` - GPT-5.6 Sol computer-use model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-5.6-terra"` - GPT-5.6 Terra computer-use model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-5.6-luna"` - GPT-5.6 Luna computer-use model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-5.5"` - GPT-5.5 computer-use model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-5.4"` - GPT-5.4 computer-use model. It is highly recommended to enable `useComputerAction` when using this model.
* `"gpt-5.4-mini"` - Smaller, faster, and more cost-effective version of GPT-5.4. It is highly recommended to enable `useComputerAction` when using this model.
Reasoning effort for OpenAI CUA. Available options:
* `"none"` — not supported by `"gpt-6-astra"` or `"gpt-6.1-sol"`
* `"low"`
* `"medium"`
* `"high"`
* `"xhigh"`
* `"max"` — only supported by `"gpt-6-astra"`, `"gpt-6.1-sol"`, `"gpt-6-sol"`, `"gpt-6-luna"`, `"gpt-5.6-sol"`, `"gpt-5.6-terra"`, and `"gpt-5.6-luna"`
Maximum number of actions CUA can take (clicks, typing, navigation, etc.). Increase for complex tasks.
Maximum consecutive failures before the task is aborted.
ID of an existing browser session to reuse. Useful for multi-step workflows that need to maintain the same browser session.
Keep the browser session alive after task completion.
Allow the agent to interact by executing actions on the actual computer not just within the page. Allows the agent to see the entire screen instead of just the page contents.
[Session configuration](/docs/api-reference/start-a-cua-task#body-session-options) (proxy, stealth, captcha solving, etc.). Only applies when creating a new session. If you provide an existing `sessionId`, these options are ignored.
Use your own OpenAI API key instead of consuming Hyperbrowser credits for LLM calls. You will only be charged for browser usage.
API key for `openai`. Required when `useCustomApiKeys` is `true`.
```typescript theme={null}
{
openai: "..."
}
```
The agent may not complete the task within the specified `maxSteps`. If that happens, try increasing the `maxSteps` parameter.
Additionally, the browser session used by the AI Agent will time out based on your team's default Session Timeout settings or the session's `timeoutMinutes` parameter if provided. You can adjust the default Session Timeout in the [Settings page](https://app.hyperbrowser.ai/settings).
`useComputerAction` can often be better for completing tasks but may require more steps. It is especially useful when the agent needs to interact with elements on the page that might not be accessible by or visible to Playwright. Since it allows the agent to see and interact with the entire screen, it is much more powerful. Instead of executing actions with Playwright which can only interact with the page via CDP, computer actions allow the agent to interact directly with computer primitives (direct clicks, typing, scroll, etc.).
## Reuse Browser Sessions
You can pass in an existing `sessionId` to the CUA task so that it can execute the task on an existing session. Also, if you want to keep the session open after executing the task, you can supply the `keepBrowserOpen` parameter.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const session = await client.sessions.create();
try {
const result = await client.agents.cua.startAndWait({
task: "What is the title of the first post on Hacker News today?",
llm: "gpt-5.4",
useComputerAction: true,
sessionId: session.id,
keepBrowserOpen: true,
});
console.log(`Output:\n${result.data?.finalResult}`);
const result2 = await client.agents.cua.startAndWait({
task: "Tell me how many upvotes the first post has.",
llm: "gpt-5.4",
useComputerAction: true,
sessionId: session.id,
});
console.log(`\nOutput:\n${result2.data?.finalResult}`);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.cua.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"llm": "gpt-5.4",
"use_computer_action": True,
"session_id": session.id,
"keep_browser_open": True,
}
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.cua.start_and_wait(
{
"task": "Tell me how many upvotes the first post has.",
"llm": "gpt-5.4",
"use_computer_action": True,
"session_id": session.id,
}
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartCuaTaskParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create()
try:
resp = client.agents.cua.start_and_wait(
StartCuaTaskParams(
task="What is the title of the first post on Hacker News today?",
llm="gpt-5.4",
use_computer_action=True,
session_id=session.id,
keep_browser_open=True,
)
)
print(f"Output:\n{resp.data.final_result}")
resp2 = client.agents.cua.start_and_wait(
StartCuaTaskParams(
task="Tell me how many upvotes the first post has.",
llm="gpt-5.4",
use_computer_action=True,
session_id=session.id,
)
)
print(f"\nOutput:\n{resp2.data.final_result}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
Always set `keepBrowserOpen: true` on tasks that you want to reuse the session from. Otherwise, the session will be automatically closed when the task completes.
## Using Your Own API Keys
Bring your own OpenAI API key to avoid consuming Hyperbrowser credits for LLM calls. You'll still be charged for browser session usage, but save on token costs.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.cua.startAndWait({
task: "What is the title of the first post on Hacker News today?",
llm: "gpt-5.4",
useComputerAction: true,
useCustomApiKeys: true,
apiKeys: {
openai: "",
},
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.cua.start_and_wait(
{
"task": "What is the title of the first post on HackerNews today?",
"llm": "gpt-5.4",
"use_computer_action": True,
"use_custom_api_keys": True,
"api_keys": {
"openai": "",
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartCuaTaskParams, CuaApiKeys
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.cua.start_and_wait(
StartCuaTaskParams(
task="What is the title of the first post on HackerNews today?",
llm="gpt-5.4",
use_computer_action=True,
use_custom_api_keys=True,
api_keys=CuaApiKeys(
openai="",
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/cua \
-H "Content-Type: application/json" \
-H "x-api-key: YOUR_HYPERBROWSER_API_KEY" \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"llm": "gpt-5.4",
"useComputerAction": true,
"useCustomApiKeys": true,
"apiKeys": {
"openai": "YOUR_OPENAI_API_KEY"
}
}'
```
## Session Configuration
Customize the browser session used by CUA with session options.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const result = await client.agents.cua.startAndWait({
task: "What is the title of the first post on Hacker News today?",
llm: "gpt-5.4",
useComputerAction: true,
sessionOptions: {
acceptCookies: true,
}
});
console.log(`Output:\n\n${result.data?.finalResult}`);
};
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.cua.start_and_wait(
{
"task": "What is the title of the first post on Hacker News today?",
"llm": "gpt-5.4",
"use_computer_action": True,
"session_options": {
"accept_cookies": True,
},
}
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartCuaTaskParams, CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
resp = client.agents.cua.start_and_wait(
StartCuaTaskParams(
task="What is the title of the first post on Hacker News today?",
llm="gpt-5.4",
use_computer_action=True,
session_options=CreateSessionParams(
accept_cookies=True,
),
)
)
print(f"Output:\n\n{resp.data.final_result}")
if __name__ == "__main__":
try:
main()
except Exception as e:
print(f"Error: {e}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/cua \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "What is the title of the first post on Hacker News today?",
"llm": "gpt-5.4",
"useComputerAction": true,
"sessionOptions": {
"acceptCookies": true
}
}'
```
`sessionOptions` only applies when creating a new session. If you provide a `sessionId`, these options are ignored.
Proxies and CAPTCHA solving add latency to page navigation. Only enable them when necessary for your use case.
## Best Practices
Be explicit about what you want CUA to do. Instead of "check the website", say "go to example.com, find the pricing page, and extract the cost of the Enterprise plan".
Simple tasks need 10-20 steps. Complex multi-page workflows might need 50+ steps. Monitor failed tasks and adjust accordingly.
It is usually better to split up complex tasks into smaller, more manageable ones and execute them as separate agent calls on the same session.
# Overview
Source: https://hyperbrowser.ai/docs/agents/overview
Use AI agents to easily automate browser tasks and create agentic workflows
Hyperbrowser lets you run powerful, AI-driven browser agents in managed cloud sessions. Whether you prefer open-source frameworks or cutting‑edge model-native agents, you can start tasks with a single API call and watch them execute live.
All agents share the same operational model: start a task, optionally poll for
status, and fetch the final result. Use our SDK helpers like `startAndWait()`
for a simple blocking workflow, or the async pattern for full control.
## Which agent should I use?
Open-source, fast, and efficient framework for browser automation with
strong defaults and great performance. Interacts with the dom using
Playwright.
Anthropic’s Claude with full computer-use capabilities for robust, reliable
real‑world interaction. Has wide array of actions and computer tools that it
can choose from.
OpenAI’s Computer‑Using Agent (Operator tech) for task execution across the
web UI. Usually slower but can be more reliable/accurate.
Google’s Gemini with computer-use for fast agentic tasks. Has more broad
tools that can incorporate multiple actions into a single step.
Meta’s Muse Spark models with computer-use capabilities and configurable
reasoning effort for multi-step browser automation.
xAI’s Grok 4.7 with computer-use capabilities and stateless reasoning
replay for robust multi-step browser automation.
Jev’s TypeSafe decision models for DOM-based browser automation that is
fast and cheap.
Our open-source, Playwright‑powered agent framework designed for control and
extensibility.
Run Stagehand on Hyperbrowser-managed browsers. Attach via CDP, reuse
stealth sessions, and monitor tasks live without self-hosting Chrome.
## Quickstart
The simplest way to run any agent is to call its `startAndWait()` method from our SDKs. Here’s a representative example using Browser‑Use:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const result = await client.agents.browserUse.startAndWait({
task: "Go to Hacker News and tell me the title of the top post",
llm: "gemini-2.0-flash",
maxSteps: 20,
});
console.log(`Output:\n${result.data?.finalResult}`);
}
main().catch((err) => {
console.error(`Error: ${err.message}`);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.browser_use.start_and_wait(
params={
"task": "Go to Hacker News and tell me the title of the top post",
"llm": "gemini-2.0-flash",
"max_steps": 20,
}
)
print(f"Output:\n{result.data.final_result}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartBrowserUseTaskParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.agents.browser_use.start_and_wait(
params=StartBrowserUseTaskParams(
task="Go to Hacker News and tell me the title of the top post",
llm="gemini-2.0-flash",
max_steps=20,
)
)
print(f"Output:\n{result.data.final_result}")
```
Switch the agent family by swapping the SDK path, e.g.
`client.agents.claudeComputerUse.startAndWait(...)`,
`client.agents.cua.startAndWait(...)`,
`client.agents.geminiComputerUse.startAndWait(...)`,
`client.agents.metaComputerUse.startAndWait(...)`,
`client.agents.grokComputerUse.startAndWait(...)`, or
`client.agents.hyperAgent.startAndWait(...)`.
## Best practices
Be explicit about the goal and constraints. Prefer “go to example.com, open
pricing, extract Enterprise monthly price” to vague prompts.
Simple tasks typically succeed within 10–20 steps; complex multi‑page flows
may need 50+. Monitor failures and adjust `maxSteps` and `maxFailures`.
Create a session once, pass `sessionId` to successive tasks, and set
`keepBrowserOpen: true` where you need continuity.
Set `useCustomApiKeys: true` and provider your own API Keys to pass calls to
your own organization.
For best results with AI Agents, it is highly recommended to keep the default
screen configuration of 1280 x 720. The models can behave poorly with larger
screen sizes.
## Explore agents
* [Browser‑Use](/docs/agents/browser-use)
* [Claude Computer Use](/docs/agents/claude-computer-use)
* [OpenAI CUA](/docs/agents/openai-cua)
* [Gemini Computer Use](/docs/agents/gemini-computer-use)
* [Meta Computer Use](/docs/agents/meta-computer-use)
* [Grok Computer Use](/docs/agents/grok-computer-use)
* [Jev Computer Use](/docs/agents/jev-computer-use)
* [HyperAgent](/docs/agents/hyperagent)
# Add a new extension
Source: https://hyperbrowser.ai/docs/api-reference/add-a-new-extension
openapi.json POST /api/extensions/add
# Cancel an image build
Source: https://hyperbrowser.ai/docs/api-reference/cancel-an-image-build
openapi.json POST /api/images/builds/{buildId}/cancel
# Complete an image build upload
Source: https://hyperbrowser.ai/docs/api-reference/complete-an-image-build-upload
openapi.json POST /api/images/builds/{buildId}/complete
# Create a sandbox memory snapshot
Source: https://hyperbrowser.ai/docs/api-reference/create-a-sandbox-memory-snapshot
openapi.json POST /api/sandbox/{id}/snapshot
# Create a volume
Source: https://hyperbrowser.ai/docs/api-reference/create-a-volume
openapi.json POST /api/volume
# Create an image build
Source: https://hyperbrowser.ai/docs/api-reference/create-an-image-build
openapi.json POST /api/images/builds
# Create new sandbox
Source: https://hyperbrowser.ai/docs/api-reference/create-new-sandbox
openapi.json POST /api/sandbox
# Create new scrape job
Source: https://hyperbrowser.ai/docs/api-reference/create-new-scrape-job
openapi.json POST /api/scrape
# Create new session
Source: https://hyperbrowser.ai/docs/api-reference/create-new-session
openapi.json POST /api/session
# Creates a new profile
Source: https://hyperbrowser.ai/docs/api-reference/creates-a-new-profile
openapi.json POST /api/profile
# Delete a volume
Source: https://hyperbrowser.ai/docs/api-reference/delete-a-volume
openapi.json DELETE /api/volume/{key}
# Delete profile by ID
Source: https://hyperbrowser.ai/docs/api-reference/delete-profile-by-id
openapi.json DELETE /api/profile/{id}
# Delete sandbox image
Source: https://hyperbrowser.ai/docs/api-reference/delete-sandbox-image
openapi.json DELETE /api/images/{key}
# Delete sandbox snapshot
Source: https://hyperbrowser.ai/docs/api-reference/delete-sandbox-snapshot
openapi.json DELETE /api/snapshots/{key}
# Expose a sandbox port
Source: https://hyperbrowser.ai/docs/api-reference/expose-a-sandbox-port
openapi.json POST /api/sandbox/{id}/expose
# Fetch a web page
Source: https://hyperbrowser.ai/docs/api-reference/fetch-a-web-page
openapi.json POST /api/web/fetch
Fetches a web page and returns the content in various formats (HTML, Markdown, JSON, screenshot, etc.)
# Fetch a web page with X402 payment
Source: https://hyperbrowser.ai/docs/api-reference/fetch-a-web-page-with-x402-payment
openapi.json POST /x402/web/fetch
X402 payment endpoint. First request returns 402 with payment requirements. Retry with PAYMENT-SIGNATURE header containing cryptographic proof to get the actual data. See https://x402.gitbook.io/x402 for protocol details.
# Get batch scrape job status
Source: https://hyperbrowser.ai/docs/api-reference/get-batch-scrape-job-status
openapi.json GET /api/scrape/batch/{id}/status
# Get batch scrape job status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-batch-scrape-job-status-and-results
openapi.json GET /api/scrape/batch/{id}
# Get browser use task status
Source: https://hyperbrowser.ai/docs/api-reference/get-browser-use-task-status
openapi.json GET /api/task/browser-use/{id}/status
# Get browser use task status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-browser-use-task-status-and-results
openapi.json GET /api/task/browser-use/{id}
# Get claude computer use task status
Source: https://hyperbrowser.ai/docs/api-reference/get-claude-computer-use-task-status
openapi.json GET /api/task/claude-computer-use/{id}/status
# Get claude computer use task status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-claude-computer-use-task-status-and-results
openapi.json GET /api/task/claude-computer-use/{id}
# Get crawl job status
Source: https://hyperbrowser.ai/docs/api-reference/get-crawl-job-status
openapi.json GET /api/crawl/{id}/status
# Get crawl job status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-crawl-job-status-and-results
openapi.json GET /api/crawl/{id}
# Get CUA task status
Source: https://hyperbrowser.ai/docs/api-reference/get-cua-task-status
openapi.json GET /api/task/cua/{id}/status
# Get CUA task status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-cua-task-status-and-results
openapi.json GET /api/task/cua/{id}
# Get extract job status
Source: https://hyperbrowser.ai/docs/api-reference/get-extract-job-status
openapi.json GET /api/extract/{id}/status
# Get extract job status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-extract-job-status-and-results
openapi.json GET /api/extract/{id}
# Get gemini computer use task status
Source: https://hyperbrowser.ai/docs/api-reference/get-gemini-computer-use-task-status
openapi.json GET /api/task/gemini-computer-use/{id}/status
# Get gemini computer use task status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-gemini-computer-use-task-status-and-results
openapi.json GET /api/task/gemini-computer-use/{id}
# Get grok computer use task status
Source: https://hyperbrowser.ai/docs/api-reference/get-grok-computer-use-task-status
openapi.json GET /api/task/grok-computer-use/{id}/status
# Get grok computer use task status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-grok-computer-use-task-status-and-results
openapi.json GET /api/task/grok-computer-use/{id}
# Get HyperAgent task status
Source: https://hyperbrowser.ai/docs/api-reference/get-hyperagent-task-status
openapi.json GET /api/task/hyper-agent/{id}/status
# Get HyperAgent task status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-hyperagent-task-status-and-results
openapi.json GET /api/task/hyper-agent/{id}
# Get image build by ID
Source: https://hyperbrowser.ai/docs/api-reference/get-image-build-by-id
openapi.json GET /api/images/builds/{buildId}
# Get jev computer use task status
Source: https://hyperbrowser.ai/docs/api-reference/get-jev-computer-use-task-status
openapi.json GET /api/task/jev/{id}/status
# Get jev computer use task status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-jev-computer-use-task-status-and-results
openapi.json GET /api/task/jev/{id}
# Get list of image builds
Source: https://hyperbrowser.ai/docs/api-reference/get-list-of-image-builds
openapi.json GET /api/images/builds
# Get list of profiles
Source: https://hyperbrowser.ai/docs/api-reference/get-list-of-profiles
openapi.json GET /api/profiles
# Get list of sandbox images
Source: https://hyperbrowser.ai/docs/api-reference/get-list-of-sandbox-images
openapi.json GET /api/images
# Get list of sandbox snapshots
Source: https://hyperbrowser.ai/docs/api-reference/get-list-of-sandbox-snapshots
openapi.json GET /api/snapshots
# Get list of sandboxes
Source: https://hyperbrowser.ai/docs/api-reference/get-list-of-sandboxes
openapi.json GET /api/sandboxes
# Get list of sessions
Source: https://hyperbrowser.ai/docs/api-reference/get-list-of-sessions
openapi.json GET /api/sessions
# Get list of volumes
Source: https://hyperbrowser.ai/docs/api-reference/get-list-of-volumes
openapi.json GET /api/volume
# Get meta computer use task status
Source: https://hyperbrowser.ai/docs/api-reference/get-meta-computer-use-task-status
openapi.json GET /api/task/meta-computer-use/{id}/status
# Get meta computer use task status and results
Source: https://hyperbrowser.ai/docs/api-reference/get-meta-computer-use-task-status-and-results
openapi.json GET /api/task/meta-computer-use/{id}
# Get profile by ID
Source: https://hyperbrowser.ai/docs/api-reference/get-profile-by-id
openapi.json GET /api/profile/{id}
# Get sandbox by ID
Source: https://hyperbrowser.ai/docs/api-reference/get-sandbox-by-id
openapi.json GET /api/sandbox/{id}
# Get sandbox snapshot
Source: https://hyperbrowser.ai/docs/api-reference/get-sandbox-snapshot
openapi.json GET /api/snapshots/{key}
# Get scrape job status
Source: https://hyperbrowser.ai/docs/api-reference/get-scrape-job-status
openapi.json GET /api/scrape/{id}/status
# Get scrape job status and result
Source: https://hyperbrowser.ai/docs/api-reference/get-scrape-job-status-and-result
openapi.json GET /api/scrape/{id}
# Get session by ID
Source: https://hyperbrowser.ai/docs/api-reference/get-session-by-id
openapi.json GET /api/session/{id}
# Get session downloads URL
Source: https://hyperbrowser.ai/docs/api-reference/get-session-downloads-url
openapi.json GET /api/session/{id}/downloads-url
# Get session recording URL
Source: https://hyperbrowser.ai/docs/api-reference/get-session-recording-url
openapi.json GET /api/session/{id}/recording-url
# Get session video recording URL
Source: https://hyperbrowser.ai/docs/api-reference/get-session-video-recording-url
openapi.json GET /api/session/{id}/video-recording-url
# Get volume by ID
Source: https://hyperbrowser.ai/docs/api-reference/get-volume-by-id
openapi.json GET /api/volume/{id}
# Get web crawl job results
Source: https://hyperbrowser.ai/docs/api-reference/get-web-crawl-job-results
openapi.json GET /api/web/crawl/{id}
Retrieves the status and results of a web crawl job. Results are paginated.
# Get web crawl job status
Source: https://hyperbrowser.ai/docs/api-reference/get-web-crawl-job-status
openapi.json GET /api/web/crawl/{id}/status
Retrieves just the status of a web crawl job without the full results.
# List all extensions
Source: https://hyperbrowser.ai/docs/api-reference/list-all-extensions
openapi.json GET /api/extensions/list
# Run manual CAPTCHA evaluation
Source: https://hyperbrowser.ai/docs/api-reference/manually-evaluate-captcha
openapi.json POST /api/session/{id}/captcha/evaluate
Trigger a bounded CAPTCHA evaluation on an active browser session. Supported CAPTCHA targets are turnstile, cloudflare-challenge, aliexpress, recaptcha, and amazon.
# Remove an exposed sandbox port
Source: https://hyperbrowser.ai/docs/api-reference/remove-an-exposed-sandbox-port
openapi.json POST /api/sandbox/{id}/unexpose
# Search the web
Source: https://hyperbrowser.ai/docs/api-reference/search-the-web
openapi.json POST /api/web/search
Performs a web search and returns search results with titles, URLs, and descriptions
# Search the web with X402 payment
Source: https://hyperbrowser.ai/docs/api-reference/search-the-web-with-x402-payment
openapi.json POST /x402/web/search
X402 payment endpoint. First request returns 402 with payment requirements. Retry with PAYMENT-SIGNATURE header containing cryptographic proof to get search results. See https://x402.gitbook.io/x402 for protocol details.
# Start a batch scrape job
Source: https://hyperbrowser.ai/docs/api-reference/start-a-batch-scrape-job
openapi.json POST /api/scrape/batch
# Start a browser use task
Source: https://hyperbrowser.ai/docs/api-reference/start-a-browser-use-task
openapi.json POST /api/task/browser-use
# Start a claude computer use task
Source: https://hyperbrowser.ai/docs/api-reference/start-a-claude-computer-use-task
openapi.json POST /api/task/claude-computer-use
# Start a crawl job
Source: https://hyperbrowser.ai/docs/api-reference/start-a-crawl-job
openapi.json POST /api/crawl
# Start a CUA task
Source: https://hyperbrowser.ai/docs/api-reference/start-a-cua-task
openapi.json POST /api/task/cua
# Start a gemini computer use task
Source: https://hyperbrowser.ai/docs/api-reference/start-a-gemini-computer-use-task
openapi.json POST /api/task/gemini-computer-use
# Start a grok computer use task
Source: https://hyperbrowser.ai/docs/api-reference/start-a-grok-computer-use-task
openapi.json POST /api/task/grok-computer-use
# Start a HyperAgent task
Source: https://hyperbrowser.ai/docs/api-reference/start-a-hyperagent-task
openapi.json POST /api/task/hyper-agent
# Start a jev computer use task
Source: https://hyperbrowser.ai/docs/api-reference/start-a-jev-computer-use-task
openapi.json POST /api/task/jev
# Start a meta computer use task
Source: https://hyperbrowser.ai/docs/api-reference/start-a-meta-computer-use-task
openapi.json POST /api/task/meta-computer-use
# Start a web crawl job
Source: https://hyperbrowser.ai/docs/api-reference/start-a-web-crawl-job
openapi.json POST /api/web/crawl
Starts an asynchronous crawl job that follows links from a starting URL and returns content from each page in the specified formats.
# Start an extract job
Source: https://hyperbrowser.ai/docs/api-reference/start-an-extract-job
openapi.json POST /api/extract
# Stop a browser use task
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-browser-use-task
openapi.json PUT /api/task/browser-use/{id}/stop
# Stop a claude computer use task
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-claude-computer-use-task
openapi.json PUT /api/task/claude-computer-use/{id}/stop
# Stop a CUA task
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-cua-task
openapi.json PUT /api/task/cua/{id}/stop
# Stop a gemini computer use task
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-gemini-computer-use-task
openapi.json PUT /api/task/gemini-computer-use/{id}/stop
# Stop a grok computer use task
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-grok-computer-use-task
openapi.json PUT /api/task/grok-computer-use/{id}/stop
# Stop a HyperAgent task
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-hyperagent-task
openapi.json PUT /api/task/hyper-agent/{id}/stop
# Stop a jev computer use task
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-jev-computer-use-task
openapi.json PUT /api/task/jev/{id}/stop
# Stop a meta computer use task
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-meta-computer-use-task
openapi.json PUT /api/task/meta-computer-use/{id}/stop
# Stop a sandbox
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-sandbox
openapi.json PUT /api/sandbox/{id}/stop
# Stop a session
Source: https://hyperbrowser.ai/docs/api-reference/stop-a-session
openapi.json PUT /api/session/{id}/stop
# Update a running session
Source: https://hyperbrowser.ai/docs/api-reference/update-a-running-session
openapi.json PUT /api/session/{id}/update
Update supported settings on an active browser session. Supported update types are profile, proxy, screen, and solveCaptchas.
# Update sandbox network policy
Source: https://hyperbrowser.ai/docs/api-reference/update-sandbox-network-policy
openapi.json PUT /api/sandbox/{id}/network
# Home
Source: https://hyperbrowser.ai/docs/home
Fast Cloud Browsers for AI Agents and Automation
# Automate with
Fast, reliable cloud browsers and sandboxes for AI automation, large-scale web scraping, and agentic code execution.
Get up and running with cloud browser sessions in 5 minutes
Get up and running with Hyperbrowser's Web API in 5 minutes
Get started with AI-powered browser automation agents in 5 minutes
## Resources
Learn how to create and manage cloud browser sessions at scale
Connect your existing automation scripts to cloud browsers
Master anti-detection features and proxy configuration
Manage sandboxes, volumes, images, and snapshots from your terminal
Get started with our Python SDK for seamless integration
Integrate Hyperbrowser into your Node.js applications
Complete API documentation for all available endpoints
# Action Caching
Source: https://hyperbrowser.ai/docs/hyperagent/action-cache
Record and replay browser automation tasks without LLM calls for fast, deterministic execution
Action Caching records every action taken during a `page.ai()` call. Replay these recordings later for deterministic, LLM-free automation—dramatically reducing costs and improving execution speed.
## Why Use Action Caching?
| Without Caching | With Caching |
| - | - |
| LLM call every run | LLM call once, replay free |
| Variable behavior | Deterministic execution |
| Higher latency | Near-instant replay |
| Higher cost | Pay once |
## Recording Actions
Every `page.ai()` call automatically returns an `actionCache`:
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
import fs from "fs";
const agent = new HyperAgent();
const page = await agent.newPage();
await page.goto("https://flights.google.com");
// Execute task - actionCache is automatically generated
const { output, actionCache } = await page.ai(
"Search for flights from Miami to LAX on December 15"
);
console.log(`Recorded ${actionCache.actionCache.steps.length} steps`);
```
## Replaying Actions
Use `agent.runFromActionCache()` to replay recorded actions:
```typescript theme={null}
import { HyperAgent, ActionCacheOutput } from "@hyperbrowser/agent";
import fs from "fs";
const agent = new HyperAgent();
// Replay without LLM calls
const result = await agent.runFromActionCache(actionCache.steps, {
maxXPathRetries: 3,
});
console.log("Replay status:", result.status);
await agent.closeAgent();
```
## Generate Script from Action Cache
Instead of replaying actions programmatically, you can generate a standalone TypeScript script from recorded actions using `agent.createScriptFromActionCache()`:
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
const agent = new HyperAgent({
llm: { provider: "anthropic", model: "claude-sonnet-4-0" },
});
const page = await agent.newPage();
// Record the automation
const { actionCache } = await page.ai(
"Go to https://demo.automationtesting.in/Frames.html, " +
"select the iframe within iframe tab, " +
"and fill in the text box in the nested iframe"
);
// Generate a reusable script
const script = agent.createScriptFromActionCache(actionCache.steps);
console.log(script);
await agent.closeAgent();
```
This outputs a standalone script you can save and run directly—no LLM calls needed:
```typescript theme={null}
// Generated script
import { HyperAgent } from "@hyperbrowser/agent";
async function main() {
const agent = new HyperAgent();
const page = await agent.newPage();
await page.goto("https://demo.automationtesting.in/Frames.html");
await page.performClick("/html/body/section/div/div/div/ul/li[2]/a", {
performInstruction: "Click the iframe within iframe tab"
});
await page.performFill("/html/body/section/div/div/div/input", "Hello", {
performInstruction: "Fill in the text box",
frameIndex: 2
});
await agent.closeAgent();
}
main();
```
The generated script uses the cache perform actions to execute the task without LLM calls.
## How Replay Works
1. **XPath First**: Attempts to find elements using cached XPaths
2. **Retry on Failure**: Retries up to `maxXPathRetries` times
3. **LLM Fallback**: If XPath fails, falls back to AI using the cached instruction
4. **Continue or Stop**: Stops on first failure by default
```typescript theme={null}
const result = await page.runFromActionCache(cache, {
maxXPathRetries: 3, // Retry XPath 3 times before LLM fallback
debug: true, // Log execution details
});
// Check what happened
for (const step of result.steps) {
console.log(`Step ${step.stepIndex}:`, {
usedXPath: step.usedXPath,
fallbackUsed: step.fallbackUsed,
success: step.success,
});
}
```
## Action Cache Format
The cache is a JSON structure containing all recorded steps:
```json theme={null}
{
"taskId": "abc-123",
"createdAt": "2025-01-15T10:30:00Z",
"status": "completed",
"steps": [
{
"stepIndex": 0,
"actionType": "actElement",
"instruction": "Click the departure city input",
"method": "click",
"arguments": [],
"frameIndex": 0,
"xpath": "/html/body/div[2]/div[4]/input[1]",
"success": true
},
{
"stepIndex": 1,
"actionType": "actElement",
"instruction": "Type 'Miami' into the input",
"method": "fill",
"arguments": ["Miami"],
"frameIndex": 0,
"xpath": "/html/body/div[2]/div[4]/input[1]",
"success": true
}
]
}
```
## Direct XPath Execution
For maximum control, use the perform helpers to execute actions directly:
```typescript theme={null}
const page = await agent.newPage();
await page.goto("https://example.com");
// Execute by XPath with LLM fallback
await page.performClick(
"/html/body/button[1]",
{ performInstruction: "Click the submit button" }
);
await page.performFill(
"/html/body/input[1]",
"user@example.com",
{ performInstruction: "Fill the email field" }
);
```
### Available Perform Actions
| Helper | Description |
| - | - |
| `performClick(xpath)` | Click an element |
| `performFill(xpath, text)` | Clear and fill an input |
| `performType(xpath, text)` | Type into an element |
| `performPress(xpath, key)` | Press a keyboard key |
| `performSelectOption(xpath, option)` | Select from dropdown |
| `performCheck(xpath)` | Check a checkbox |
| `performUncheck(xpath)` | Uncheck a checkbox |
| `performHover(xpath)` | Hover over an element |
| `performScrollToElement(xpath)` | Scroll element into view |
Each helper accepts an options object:
```typescript theme={null}
await page.performClick(xpath, {
performInstruction: "Click the login button", // Fallback instruction
frameIndex: 0, // Target iframe (0 = main frame)
maxSteps: 3, // Retries before fallback
});
```
## When to Use Action Caching
| Scenario | Recommendation |
| - | - |
| Repetitive tasks (daily scraping, scheduled jobs) | ✅ Record once, replay indefinitely |
| E2E testing | ✅ Fast, deterministic test runs |
| High-volume automation | ✅ Eliminate per-run LLM costs |
| Stable page structures | ✅ XPaths remain valid longer |
| Dynamic pages with frequent layout changes | ⚠️ May require frequent re-recording |
| One-time tasks | ❌ Just use `page.ai()` directly |
## Monitoring Fallback Rates
When a cached XPath no longer matches the page, HyperAgent falls back to the LLM to find the element if the performInstruction is provided. You'll see logs like this:
```
⚠️ [runCachedStep] Cached action failed. Falling back to LLM...
Instruction: "Select the LATAM/Delta flight with the lowest carbon emissions"
❌ Cached XPath Failed: "/html[1]/body[1]/c-wiz[2]/div[1]/.../li[5]/div[1]/div[1]"
✅ LLM Resolved New XPath: "/html[1]/body[1]/c-wiz[2]/div[1]/.../li[4]/div[1]/div[1]"
```
**What this means:**
* The cached XPath pointed to `li[5]` but the element moved to `li[4]`
* The LLM successfully found the correct element using the instruction
* The action completed, but with added latency and cost
**When to re-record:**
* If you see fallback warnings frequently, the page structure has changed
* Re-run the original `page.ai()` task to capture fresh XPaths
* Save the new `actionCache` to replace your stale recording
Enable `debug: true` on your agent to see more detailed logging.
# HyperBrowser Provider
Source: https://hyperbrowser.ai/docs/hyperagent/browser-providers
Run HyperAgent locally or scale to the cloud with Hyperbrowser
HyperAgent supports two browser providers: local Chromium for development and Hyperbrowser for production scale.
## Local Browser (Default)
By default, HyperAgent launches a local Chromium browser:
```typescript theme={null}
const agent = new HyperAgent({
// browserProvider defaults to "Local"
});
```
**Pros:**
* No API key required
* Free to use
* Full control over browser instance
* Works offline
**Cons:**
* Limited to your machine's resources
* Manual scaling required
* No built-in stealth or proxy features
## Hyperbrowser Cloud
You can use Hyperbrowser to scale your production needs [Hyperbrowser](/docs/sessions/create):
```typescript theme={null}
const agent = new HyperAgent({
browserProvider: "Hyperbrowser",
});
```
**Environment variable:** `HYPERBROWSER_API_KEY`
Get your free API key at [app.hyperbrowser.ai](https://app.hyperbrowser.ai).
**Pros:**
* Scale to hundreds of concurrent sessions
* Built-in stealth mode and proxy rotation
* CAPTCHA solving support
* Session recordings and live view
* No local resources consumed
**Cons:**
* Requires API key
* Network latency
* Usage-based pricing
## Hyperbrowser Configuration
When using Hyperbrowser, you can configure session options:
```typescript theme={null}
const agent = new HyperAgent({
browserProvider: "Hyperbrowser",
// Session options passed to Hyperbrowser
});
```
For advanced session configuration (proxies, stealth, CAPTCHA solving), see:
* [Session Configuration](/docs/sessions/create)
* [Stealth Mode](/docs/sessions/stealth)
* [Proxy Settings](/docs/sessions/proxy)
* [CAPTCHA Solving](/docs/sessions/captcha-solving)
## Next Steps
Configure your AI provider
Record and replay automations
# Custom Actions
Source: https://hyperbrowser.ai/docs/hyperagent/custom-actions
Extend HyperAgent with custom actions for specialized workflows
Custom actions let you extend HyperAgent's capabilities beyond browser automation. Add integrations with external APIs, databases, or any custom logic.
## Defining a Custom Action
A custom action requires three things:
1. **type**: A descriptive name for the action
2. **actionParams**: A Zod schema describing the parameters
3. **run**: A function that executes the action
```typescript theme={null}
import { HyperAgent, AgentActionDefinition, ActionContext, ActionOutput } from "@hyperbrowser/agent";
import { z } from "zod";
const SendEmailAction: AgentActionDefinition = {
type: "send_email",
actionParams: z.object({
to: z.string().describe("Email recipient address"),
subject: z.string().describe("Email subject line"),
body: z.string().describe("Email body content"),
}).describe("Send an email to a specified recipient"),
run: async function(
ctx: ActionContext,
params: { to: string; subject: string; body: string }
): Promise {
// Your email sending logic here
await sendEmail(params.to, params.subject, params.body);
return {
success: true,
message: `Successfully sent email to ${params.to}`,
};
},
};
```
## Using Custom Actions
Pass custom actions when creating the agent:
```typescript theme={null}
const agent = new HyperAgent({
customActions: [SendEmailAction],
});
const result = await agent.executeTask(
"Go to my inbox, find the latest newsletter, summarize it, and send the summary to boss@company.com"
);
```
The AI will automatically use your custom action when appropriate.
## Real-World Example: Web Search
Integrate with a search API like Exa:
```typescript theme={null}
import Exa from "exa-js";
const exaClient = new Exa(process.env.EXA_API_KEY);
const WebSearchAction: AgentActionDefinition = {
type: "web_search",
actionParams: z.object({
query: z.string().describe("Search query - keep it concise and specific"),
}).describe("Search the web and return relevant results"),
run: async function(ctx, params): Promise {
const results = await exaClient.search(params.query, {
numResults: 5,
});
const formatted = results.results
.map(r => `- ${r.title}: ${r.url}`)
.join("\n");
return {
success: true,
message: `Search results for "${params.query}":\n${formatted}`,
};
},
};
const agent = new HyperAgent({
customActions: [WebSearchAction],
});
await agent.executeTask(
"Search for the latest news about AI and summarize the top 3 stories"
);
```
## Action Context
The `ActionContext` provides access to:
```typescript theme={null}
interface ActionContext {
page: Page; // Current Playwright page
agent: HyperAgent; // Agent instance
taskId: string; // Current task ID
}
```
Use it to interact with the browser or agent state:
```typescript theme={null}
const SaveScreenshotAction: AgentActionDefinition = {
type: "save_screenshot",
actionParams: z.object({
filename: z.string().describe("Filename to save the screenshot as"),
}),
run: async function(ctx, params): Promise {
await ctx.page.screenshot({ path: params.filename });
return {
success: true,
message: `Screenshot saved to ${params.filename}`,
};
},
};
```
## Action Output
Return an `ActionOutput` object:
```typescript theme={null}
interface ActionOutput {
success: boolean; // Whether the action succeeded
message: string; // Result message shown to the AI
}
```
The message helps the AI understand what happened and plan next steps.
## Multiple Custom Actions
Combine multiple actions for complex workflows:
```typescript theme={null}
const agent = new HyperAgent({
customActions: [
WebSearchAction,
SendEmailAction,
SaveToNotionAction,
SlackNotifyAction,
],
});
await agent.executeTask(
"Research competitors, save findings to Notion, and notify the team on Slack"
);
```
## Next Steps
Connect to MCP servers for more tools
Learn about task execution
# page.extract()
Source: https://hyperbrowser.ai/docs/hyperagent/extract
Extract structured data from web pages using natural language and Zod schemas
The `page.extract()` method pulls structured data from web pages. Define what you want using natural language and optionally enforce a schema with Zod for type-safe results.
## Basic Usage
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
import { z } from "zod";
const agent = new HyperAgent();
const page = await agent.newPage();
await page.goto("https://news.ycombinator.com");
// Simple extraction (returns string)
const topStory = await page.extract("what is the title of the top story?");
console.log(topStory); // "Show HN: I built a..."
// Structured extraction (returns typed object)
const stories = await page.extract(
"get the top 5 stories",
z.object({
stories: z.array(z.object({
title: z.string(),
points: z.number(),
author: z.string(),
}))
})
);
console.log(stories.stories[0].title);
```
## Parameters
Natural language description of what data to extract.
Optional Zod schema for structured, typed output. Without a schema, returns a string.
## Why Use Schemas?
Schemas provide three key benefits:
1. **Type Safety**: Get full TypeScript autocompletion and type checking
2. **Validation**: Ensures the AI returns data in the correct format
3. **Documentation**: Use `.describe()` to guide the AI on what each field means
```typescript theme={null}
const productSchema = z.object({
name: z.string().describe("The product name"),
price: z.number().describe("Price in USD, numbers only"),
inStock: z.boolean().describe("Whether the item is available"),
reviews: z.array(z.object({
rating: z.number().min(1).max(5),
text: z.string(),
})).describe("Customer reviews"),
});
const product = await page.extract("get this product's details", productSchema);
// product is fully typed: { name: string, price: number, inStock: boolean, reviews: [...] }
```
## Common Extraction Patterns
### Product Information
```typescript theme={null}
const product = await page.extract(
"extract the product details",
z.object({
name: z.string(),
price: z.number(),
originalPrice: z.number().optional(),
rating: z.number(),
reviewCount: z.number(),
availability: z.enum(["in_stock", "out_of_stock", "limited"]),
})
);
```
### Table Data
```typescript theme={null}
const tableData = await page.extract(
"extract all rows from the pricing table",
z.object({
rows: z.array(z.object({
plan: z.string(),
price: z.string(),
features: z.array(z.string()),
}))
})
);
```
### Article Content
```typescript theme={null}
const article = await page.extract(
"extract the article content",
z.object({
title: z.string(),
author: z.string(),
publishDate: z.string(),
content: z.string(),
tags: z.array(z.string()),
})
);
```
### Lists and Rankings
```typescript theme={null}
const rankings = await page.extract(
"get the top 10 items from this list",
z.object({
items: z.array(z.object({
rank: z.number(),
name: z.string(),
score: z.number().optional(),
}))
})
);
```
## Error Handling
```typescript theme={null}
try {
const data = await page.extract("get the user profile", schema);
} catch (error) {
if (error.message.includes("validation")) {
console.error("Data didn't match schema:", error);
} else {
console.error("Extraction failed:", error);
}
}
```
## Best Practices
**Good:** "extract the price shown next to the 'Buy Now' button"
**Bad:** "get the price"
```typescript theme={null}
// ✅ Good: Clear field names and descriptions
z.object({
priceUsd: z.number().describe("Price in US dollars"),
stockCount: z.number().describe("Number of items in stock"),
})
// ❌ Bad: Ambiguous fields
z.object({
p: z.number(),
n: z.number(),
})
```
Don't create overly complex schemas for simple data. If you only need a single value:
```typescript theme={null}
const price = await page.extract("what is the price?", z.number());
```
```typescript theme={null}
z.object({
status: z.enum(["pending", "shipped", "delivered"]),
category: z.enum(["electronics", "clothing", "home"]),
})
```
## Next Steps
Perform actions before extracting
Single-action execution
# LLM Providers
Source: https://hyperbrowser.ai/docs/hyperagent/llm-providers
Configure OpenAI, Anthropic, Google Gemini as your LLM provider
HyperAgent supports multiple LLM providers with native SDK integration. All models from each provider are supported—use whichever fits your needs.
## Supported Providers
| Provider | Models | Environment Variable |
| - | - | - |
| OpenAI | [All models](https://platform.openai.com/docs/models) | `OPENAI_API_KEY` |
| Anthropic | [All models](https://docs.anthropic.com/en/docs/about-claude/models) | `ANTHROPIC_API_KEY` |
| Google Gemini | [All models](https://ai.google.dev/gemini-api/docs/models/gemini) | `GEMINI_API_KEY` |
## Configuration
### OpenAI
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
const agent = new HyperAgent({
llm: {
provider: "openai",
model: "gpt-5.1",
},
});
```
### Anthropic
```typescript theme={null}
const agent = new HyperAgent({
llm: {
provider: "anthropic",
model: "claude-sonnet-4-5",
},
});
```
### Google Gemini
```typescript theme={null}
const agent = new HyperAgent({
llm: {
provider: "gemini",
model: "gemini-3-pro-preview",
},
});
```
## Environment Setup
Create a `.env` file in your project root:
```bash theme={null}
# Add the API key for your chosen provider
OPENAI_API_KEY=sk-...
ANTHROPIC_API_KEY=sk-ant-...
GEMINI_API_KEY=...
```
Load it in your code:
```typescript theme={null}
import { config } from "dotenv";
config();
const agent = new HyperAgent({
llm: {
provider: "openai",
model: "gpt-5.1",
},
});
```
## Switching Providers
Create agents with different providers for different tasks:
```typescript theme={null}
// Use Claude for complex reasoning
const complexAgent = new HyperAgent({
llm: { provider: "anthropic", model: "claude-sonnet-4-5" },
});
// Use a smaller model for simple tasks
const simpleAgent = new HyperAgent({
llm: { provider: "openai", model: "gpt-5-mini" },
});
```
## Cost Optimization
Token costs vary by provider and model. To reduce costs:
1. Use `page.perform()` instead of `page.ai()` when possible (single LLM call vs. multiple)
2. Use [Action Caching](/docs/hyperagent/action-cache) to replay without LLM calls
3. Use smaller/cheaper models for simple tasks
## Next Steps
Configure local vs cloud browsers
Replay automations without LLM calls
# MCP Integration
Source: https://hyperbrowser.ai/docs/hyperagent/mcp
Connect HyperAgent to MCP servers for extended capabilities
HyperAgent functions as a fully functional MCP (Model Context Protocol) client. Connect to MCP servers to extend capabilities with tools like Google Sheets, Notion, Slack, and more.
## What is MCP?
MCP (Model Context Protocol) is a standard for connecting AI models to external tools and data sources. HyperAgent can connect to any MCP-compatible server.
## Setting Up MCP
Initialize the MCP client with your server configuration:
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
const agent = new HyperAgent({
llm: {
provider: "openai",
model: "gpt-4o", // Recommended for MCP
},
});
await agent.initializeMCPClient({
servers: [
{
command: "npx",
args: [
"@composio/mcp@latest",
"start",
"--url",
"https://mcp.composio.dev/googlesheets/your-connection-id",
],
env: {
npm_config_yes: "true",
},
},
],
});
```
## Example: Web Data to Google Sheets
Scrape data from a website and write it to Google Sheets:
```typescript theme={null}
const agent = new HyperAgent({
llm: {
provider: "openai",
model: "gpt-4o",
},
debug: true,
});
await agent.initializeMCPClient({
servers: [
{
command: "npx",
args: [
"@composio/mcp@latest",
"start",
"--url",
"https://mcp.composio.dev/googlesheets/...",
],
env: {
npm_config_yes: "true",
},
},
],
});
const response = await agent.executeTask(
"Go to https://en.wikipedia.org/wiki/List_of_U.S._states_and_territories_by_population " +
"and get the data on the top 5 most populous states from the table. " +
"Then insert that data into a Google Sheet."
);
console.log(response);
await agent.closeAgent();
```
## Available MCP Servers
Popular MCP servers you can connect to:
| Server | Use Case |
| - | - |
| Google Sheets | Read/write spreadsheet data |
| Notion | Create and update pages |
| Slack | Send messages and notifications |
| GitHub | Create issues, PRs |
| Airtable | Database operations |
Find more at [Composio MCP](https://composio.dev/mcp) or build your own.
## Multiple Servers
Connect to multiple MCP servers simultaneously:
```typescript theme={null}
await agent.initializeMCPClient({
servers: [
{
command: "npx",
args: ["@composio/mcp@latest", "start", "--url", "sheets-url"],
env: { npm_config_yes: "true" },
},
{
command: "npx",
args: ["@composio/mcp@latest", "start", "--url", "slack-url"],
env: { npm_config_yes: "true" },
},
],
});
await agent.executeTask(
"Scrape the pricing from competitor.com, save to Google Sheets, and notify #sales on Slack"
);
```
# Multi-Page Management
Source: https://hyperbrowser.ai/docs/hyperagent/multi-page
Manage multiple browser pages and tabs simultaneously
HyperAgent supports managing multiple browser pages within a single agent instance. This is useful for workflows that require data from multiple sources or parallel processing.
## Creating Multiple Pages
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
const agent = new HyperAgent();
// Create multiple pages
const page1 = await agent.newPage();
const page2 = await agent.newPage();
// Each page is independent
await page1.goto("https://news.ycombinator.com");
await page2.goto("https://reddit.com");
await agent.closeAgent();
```
## Shared Browser, Isolated Agent History
All pages share the same browser context (cookies, storage, session state), but the agent's history is tracked separately per page. This means:
* **Shared**: Cookies, localStorage, authentication state
* **Isolated**: The agent's memory of what actions it took on each page
```typescript theme={null}
const agent = new HyperAgent();
const page1 = await agent.newPage();
const page2 = await agent.newPage();
// page1: Agent tracks history for this page
await page1.goto("https://example.com/login");
await page1.ai("Log in with user@example.com");
// page2: Agent has separate history, but shares the login session
await page2.goto("https://example.com/dashboard");
await page2.ai("Find my account settings");
// ^ Works because cookies are shared, but agent doesn't know about page1's actions
```
To pass data between pages explicitly, extract it and use it in your code:
```typescript theme={null}
const destination = await page1.ai("Find the recommended destination");
await page2.ai(`Search for hotels in ${destination.output}`);
```
## Getting All Pages
```typescript theme={null}
const pages = await agent.getPages();
console.log(`Active pages: ${pages.length}`);
```
## Parallel Execution
Run tasks on multiple pages simultaneously:
```typescript theme={null}
const agent = new HyperAgent();
const pages = await Promise.all([
agent.newPage(),
agent.newPage(),
agent.newPage(),
]);
// Navigate all pages in parallel
await Promise.all([
pages[0].goto("https://amazon.com"),
pages[1].goto("https://ebay.com"),
pages[2].goto("https://walmart.com"),
]);
// Search on all pages in parallel
const results = await Promise.all([
pages[0].ai("search for 'wireless mouse' and find the cheapest option"),
pages[1].ai("search for 'wireless mouse' and find the cheapest option"),
pages[2].ai("search for 'wireless mouse' and find the cheapest option"),
]);
// Extract prices from all pages
const prices = await Promise.all([
pages[0].extract("get the lowest price", z.number()),
pages[1].extract("get the lowest price", z.number()),
pages[2].extract("get the lowest price", z.number()),
]);
console.log("Prices:", prices);
await agent.closeAgent();
```
# Overview
Source: https://hyperbrowser.ai/docs/hyperagent/overview
Use HyperAgent to automate browser tasks with AI
HyperAgent is an open-source browser automation framework that extends Playwright with AI capabilities. Write natural language commands instead of complex selectors, and let HyperAgent handle the tedious parts of web automation.
View source code
Install node SDK
View curated list of templates
## The Challenge with Browser Automation
Browser automation tools like Puppeteer and Playwright offer powerful functionality for scripting clicks, typing, scrolling, and more. But they require you to understand the DOM structure and locate elements through HTML attributes, CSS selectors, or complex XPath queries.
This gets harder fast:
* **Selectors break** when websites update their markup
* **Iframes isolate content**, requiring nested queries to reach elements inside them
* **Shadow DOM encapsulation** makes elements even harder to access
* **Dynamic content** means selectors that worked yesterday might fail today
You end up spending more time maintaining selectors than building features.
## What HyperAgent Does
HyperAgent lets you describe what you want in plain English. The AI figures out how to interact with the page—no matter how the DOM is structured.
```typescript theme={null}
// Instead of this:
await page.locator('/html[1]/body[1]/c-wiz[2]/div[1]/div[2]/c-wiz[1]/div[1]/c-wiz[1]/div[2]/div[1]/div[1]/div[1]/div[1]/div[2]/div[1]/div[6]/div[2]/div[2]/div[1]/div[1]/input[1]').fill('Miami');
// Write this:
await page.perform("type Miami into the departure city field");
```
HyperAgent handles iframes, shadow DOM, and dynamic content automatically. When a site changes, your automation keeps working.
## Core Methods
Execute complex multi-step tasks with natural language
Fast, single-action execution
Pull structured data with Zod schemas
Use standard Playwright when you need deterministic control
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
import { z } from "zod";
const agent = new HyperAgent();
const page = await agent.newPage();
await page.goto("https://flights.google.com");
// AI handles the complexity
await page.ai("search for flights from Miami to LAX on Dec 15");
// Single actions when you know what you need
await page.perform("click the first result");
// Extract structured data
const flight = await page.extract(
"get the price and duration of the selected flight",
z.object({
price: z.number(),
duration: z.string(),
})
);
// use Playwright
await page.locator('css=button').click();
await agent.closeAgent();
```
## Key Features
Describe the element in natural language. HyperAgent finds it regardless of DOM structure, iframes, or shadow DOM.
Record your automation once, replay it without LLM calls. Deterministic execution at a fraction of the cost.
Use OpenAI, Anthropic, Google Gemini. Switch providers with one line of code.
Run locally for development, scale to hundreds of sessions with [Hyperbrowser](/docs/sessions/create) in production.
Native Chrome DevTools Protocol integration for precise coordinates, deep iframe tracking, and automatic ad filtering.
## Get Started
```bash npm theme={null}
npm install @hyperbrowser/agent
```
```bash yarn theme={null}
yarn add @hyperbrowser/agent
```
```bash pnpm theme={null}
pnpm add @hyperbrowser/agent
```
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
const agent = new HyperAgent();
const page = await agent.newPage();
await page.goto("https://news.ycombinator.com");
await page.ai("find the top story and summarize it");
await agent.closeAgent();
```
Build your first automation
# page.ai()
Source: https://hyperbrowser.ai/docs/hyperagent/page-ai
Execute complex multi-step browser tasks with AI-powered automation
The `page.ai()` method executes multi-step browser tasks using natural language. It runs an agentic loop that observes the page, makes decisions, and executes actions until the goal is complete.
`agent.executeTask()` does the same thing—it's a convenience method that creates a page for you. Use whichever fits your workflow.
## Basic Usage
Execute a multi-step task on a specific page:
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
const agent = new HyperAgent({
llm: { provider: "openai", model: "gpt-4o" },
});
const page = await agent.newPage();
await page.goto("https://flights.google.com");
const { output, actionCache } = await page.ai(
"Search for round-trip flights from Miami to LAX, leaving Dec 15 and returning Dec 22"
);
console.log(output);
// The actionCache can be saved and replayed later
console.log(`Completed in ${actionCache.steps.length} steps`);
await agent.closeAgent();
```
### Parameters
Natural language description of what you want to accomplish. Be specific for best results.
Configuration options for the task execution.
Maximum number of actions the AI can take. Increase for complex tasks.
Reuse DOM snapshots across steps for faster execution.
Enable screenshots with element overlays for visual understanding.
Stream DOM updates for more responsive execution.
Define a Zod schema for structured output extraction at the end of the task.
### Example with Options
```typescript theme={null}
import { z } from "zod";
const { output, actionCache } = await page.ai(
"Find the cheapest flight from NYC to London next month",
{
maxSteps: 30,
useDomCache: true,
outputSchema: z.object({
airline: z.string(),
price: z.number(),
departure: z.string(),
arrival: z.string(),
}),
}
);
console.log(output); // Typed as { airline, price, departure, arrival }
```
## agent.executeTask()
`executeTask()` is a shorthand that creates a page and runs `page.ai()` for you:
```typescript theme={null}
// This:
const result = await agent.executeTask("Go to amazon.com and find the top seller");
// Is equivalent to:
const page = await agent.newPage();
const result = await page.ai("Go to amazon.com and find the top seller");
```
It accepts the same parameters as `page.ai()`:
### With Output Schema
```typescript theme={null}
import { z } from "zod";
const result = await agent.executeTask(
"Navigate to imdb.com, search for 'The Matrix', and extract the movie details",
{
outputSchema: z.object({
director: z.string().describe("The name of the movie director"),
releaseYear: z.number().describe("The year the movie was released"),
rating: z.string().describe("The IMDb rating of the movie"),
}),
}
);
console.log(result.output);
// { director: "Lana Wachowski, Lilly Wachowski", releaseYear: 1999, rating: "8.7/10" }
```
## Return Value
Both methods return a `TaskOutput` object:
```typescript theme={null}
interface TaskOutput {
taskId: string; // Unique identifier for this task
status: TaskStatus; // "completed" | "failed" | "cancelled"
output: string | T; // Result (or typed if outputSchema provided)
steps: AgentStep[]; // Array of steps taken
actionCache: ActionCacheOutput; // Recorded actions for replay
}
```
### Task Status
| Status | Description |
| - | - |
| `completed` | Task finished successfully |
| `failed` | Task encountered an error |
| `cancelled` | Task was cancelled before completion |
## Visual Mode
Enable visual mode when the AI needs to understand page layout or when dealing with complex visual elements:
```typescript theme={null}
const { output } = await page.ai(
"Find the product image and describe what's shown",
{
enableVisualMode: true,
}
);
```
Visual mode uses screenshots which increases token usage and latency. Only enable when visual understanding is necessary.
## Real-World Examples
### Flight Search
```typescript theme={null}
const page = await agent.newPage();
await page.goto("https://flights.google.com");
const { output, actionCache } = await page.ai(
"Search for round-trip flights from Rio de Janeiro to Los Angeles, " +
"leaving December 11, 2025 and returning December 22, 2025. " +
"Select the option with the lowest carbon emissions.",
{
useDomCache: true,
enableDomStreaming: true,
}
);
// Save actionCache for later replay
console.log(JSON.stringify(actionCache, null, 2));
```
### E-commerce Price Comparison
```typescript theme={null}
import { z } from "zod";
const result = await agent.executeTask(
"Go to amazon.com, search for 'mechanical keyboard', and compare the top 3 results",
{
outputSchema: z.object({
products: z.array(z.object({
name: z.string(),
price: z.number(),
rating: z.number(),
reviewCount: z.number(),
})),
recommendation: z.string(),
}),
}
);
console.log(result.output.products);
console.log(result.output.recommendation);
```
### Google Form Submission
```typescript theme={null}
const agent = new HyperAgent({
llm: { provider: "openai", model: "gpt-4o" },
});
const page = await agent.newPage();
await page.goto("https://docs.google.com/forms/d/e/1FAIpQLScPkE8wNLpPSkP2d__Ee7xx5Pj7_XDuZ0p16geYWrp73Nutmw/viewform?usp=dialog");
// Fill each field
await page.perform("fill the name field with John Doe");
await page.perform("fill the email field with john@example.com");
await page.perform("fill the feedback text area with This is a test submission");
await page.perform("select 5 rating option");
// Submit
await page.perform("click the submit button");
await agent.closeAgent();
```
## Action Cache Output
Every `page.ai()` call returns an `actionCache` that records all actions taken:
```json theme={null}
{
"taskId": "abc-123",
"createdAt": "2025-01-15T10:30:00Z",
"status": "completed",
"steps": [
{
"stepIndex": 0,
"actionType": "actElement",
"instruction": "Click the source location input",
"method": "click",
"arguments": [],
"frameIndex": 0,
"xpath": "/html/body/div[1]/input[1]",
"success": true,
"message": "Successfully clicked element"
}
]
}
```
This cache can be saved and [replayed later](/docs/hyperagent/action-cache) for deterministic execution without LLM calls.
## Error Handling
```typescript theme={null}
try {
const result = await page.ai("Complete the checkout process", {
maxSteps: 50,
});
if (result.status === "failed") {
console.error("Task failed:", result.output);
}
} catch (error) {
console.error("Execution error:", error);
}
```
## Best Practices
Instead of "search for flights", say "search for round-trip flights from Miami to LAX, departing December 15 and returning December 22, 2025".
Simple tasks: 10-20 steps. Complex multi-page workflows: 50+ steps. Monitor task outputs and adjust as needed.
When you need specific data extracted, define a Zod schema to get typed, validated output.
Store the returned `actionCache` to replay the same automation later without LLM calls.
## Next Steps
Fast single-action execution
Extract structured data
Replay automations without LLM calls
Configure LLM providers
# page.perform()
Source: https://hyperbrowser.ai/docs/hyperagent/page-perform
Execute fast, single-action browser interactions using natural language
The `page.perform()` method executes single, granular actions on a web page. It's optimized for speed and reliability, using the accessibility tree instead of screenshots.
## Overview
| Characteristic | Description |
| - | - |
| **Speed** | ⚡ Fast - Uses accessibility tree (no screenshots) |
| **Cost** | 💰 Cheap - Single LLM call per action |
| **Reliability** | 🎯 Direct element finding and execution |
| **Efficiency** | 📊 Text-based DOM analysis with automatic ad-frame filtering |
## Basic Usage
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
const agent = new HyperAgent({
llm: { provider: "openai", model: "gpt-4o" },
});
const page = await agent.newPage();
await page.goto("https://example.com/login");
// Execute single actions
await page.perform("fill email with user@example.com");
await page.perform("fill password with mypassword");
await page.perform("click the login button");
await agent.closeAgent();
```
## Common Actions
### Click Elements
```typescript theme={null}
await page.perform("click the login button");
await page.perform("click the first search result");
await page.perform("click the 'Add to Cart' button");
await page.perform("click the menu icon in the top right");
```
### Fill or Type Inputs
```typescript theme={null}
await page.perform("fill email with test@example.com");
await page.perform("type 'mechanical keyboard' into the search box");
await page.perform("fill the password field with MySecurePass123");
```
### Form Interactions
```typescript theme={null}
await page.perform("check the 'Remember me' checkbox");
await page.perform("uncheck the newsletter subscription");
await page.perform("select 'United States' from the country dropdown");
```
### Scrolling
```typescript theme={null}
// Scroll to a specific element
await page.perform("scroll to the pricing section");
await page.perform("scroll the reviews section into view");
// Scroll by percentage
await page.perform("scroll to 50% of the page");
await page.perform("scroll to the bottom of the page");
// Chunk-based scrolling (useful for infinite scroll or long pages)
await page.perform("scroll to the next chunk");
await page.perform("scroll to the previous chunk");
```
### Hover
```typescript theme={null}
await page.perform("hover over the user profile menu");
await page.perform("hover over the dropdown to reveal options");
```
### Keyboard Actions
```typescript theme={null}
await page.perform("press Enter");
await page.perform("press Escape to close the modal");
await page.perform("press Tab to move to the next field");
```
## When to Use perform() vs ai()
* Single, specific actions
* When you know exactly what action is needed
* Fast, reliable execution
* Lower token cost
* Complex multi-step workflows
* When visual context is needed
* Tasks requiring decision making
* When next action depends on page state
### Example: Combining Both
```typescript theme={null}
const page = await agent.newPage();
await page.goto("https://amazon.com");
// Use perform for known, simple actions
await page.perform("click the search box");
await page.perform("type 'laptop' into the search box");
await page.perform("click the search button");
// Use ai() when complex decision-making is needed
await page.ai("find the best-rated laptop under $1000 and add it to cart");
```
## Return Value
`page.perform()` returns a `TaskOutput` object:
```typescript theme={null}
interface TaskOutput {
taskId: string;
status: TaskStatus; // "completed" | "failed"
output: string; // Result message
steps: AgentStep[]; // Steps taken (usually 1 for perform)
}
```
### Checking Success
```typescript theme={null}
const result = await page.perform("click the submit button");
if (result.status === "completed") {
console.log("Action successful:", result.output);
} else {
console.error("Action failed:", result.output);
}
```
## Error Handling
```typescript theme={null}
try {
await page.perform("click the non-existent button");
} catch (error) {
console.error("Failed to perform action:", error);
}
```
## Tips for Writing Effective Instructions
**Good:** "click the blue 'Sign Up' button at the bottom of the form"
**Bad:** "click the button"
**Good:** "fill the email input in the login form with [user@example.com](mailto:user@example.com)"
**Bad:** "fill email"
**Good:** "click the link that says 'Learn More'"
**Bad:** "click the third link"
**Good:** "type 'search query' into the search box"
**Bad:** "search for something"
## CDP Actions
HyperAgent uses Chrome DevTools Protocol (CDP) for precise element interactions by default. This provides:
* Exact coordinate-based clicks
* Deep iframe support
* Auto-filtering of ad frames
To disable CDP and use Playwright locators instead:
```typescript theme={null}
const agent = new HyperAgent({
cdpActions: false,
});
```
## Next Steps
Complex multi-step automation
Extract structured data
Record and replay automations
# Quickstart
Source: https://hyperbrowser.ai/docs/hyperagent/quickstart
Build your first HyperAgent automation in under a minute
## Installation
```bash npm theme={null}
npm install @hyperbrowser/agent
```
```bash yarn theme={null}
yarn add @hyperbrowser/agent
```
```bash pnpm theme={null}
pnpm add @hyperbrowser/agent
```
## Choose Your LLM Provider
HyperAgent needs an LLM to power its AI capabilities. We recommend setting environment variables for your API keys.
```bash OpenAI theme={null}
export OPENAI_API_KEY="sk-..."
```
```bash Anthropic theme={null}
export ANTHROPIC_API_KEY="sk-ant-..."
```
```bash Gemini theme={null}
export GOOGLE_API_KEY="..."
```
## Your First Automation
Create a file called `demo.ts`:
```typescript theme={null}
import { HyperAgent } from "@hyperbrowser/agent";
async function main() {
// Initialize the agent
const agent = new HyperAgent({
llm: {
provider: "openai",
model: "gpt-5.1",
},
});
// Create a new browser page
const page = await agent.newPage();
// Navigate and automate
await page.goto("https://news.ycombinator.com");
const result = await page.ai(
"Find the top story and tell me the title and point count"
);
console.log(result.output);
// Remember to close the agent when you're done, otherwise the session will stay open.
await agent.closeAgent();
}
main();
```
Run it:
```bash theme={null}
npx tsx demo.ts
```
## CLI Mode
For quick one-off tasks, use the CLI:
```bash theme={null}
npx @hyperbrowser/agent -c "Go to google.com and search for 'best pizza in NYC'"
```
**CLI Options:**
| Flag | Description |
| - | - |
| `-c, --command ` | Natural language command to run |
| `-d, --debug` | Enable debug mode with verbose output |
| `--hyperbrowser` | Use Hyperbrowser cloud instead of local browser |
## Get Started with 3 Core Methods
HyperAgent gives you three ways to interact with pages:
### 1. `page.ai()` — Complex Tasks
For multi-step workflows where AI makes decisions:
```typescript theme={null}
await page.ai("search for flights from Miami to LAX, select the cheapest option");
```
### 2. `page.perform()` — Single Actions
For fast, specific actions:
```typescript theme={null}
await page.perform("click the login button");
await page.perform("fill email with user@example.com");
```
### 3. `page.extract()` — Data Extraction
For pulling structured data:
```typescript theme={null}
import { z } from "zod";
const data = await page.extract(
"get the product details",
z.object({
name: z.string(),
price: z.number(),
rating: z.number(),
})
);
```
## Mix and Match
The real power comes from combining these methods:
```typescript theme={null}
const page = await agent.newPage();
await page.goto("https://amazon.com");
// Fast, single-action execution
await page.perform("click the search box");
await page.perform("type 'mechanical keyboard'");
await page.perform("press Enter");
// AI handles the complex part
await page.ai("search for keyboards under $100");
// Extract structured data
const product = await page.extract(
"Select the first product from the search results",
z.object({
name: z.string(),
price: z.number(),
rating: z.number(),
})
);
console.log(product);
```
## Scale to the Cloud
When you're ready to run at scale, switch to [Hyperbrowser](/docs/sessions/create):
```typescript theme={null}
const agent = new HyperAgent({
browserProvider: "Hyperbrowser", // That's it!
});
```
You'll need a `HYPERBROWSER_API_KEY`. Get one free at [app.hyperbrowser.ai](https://app.hyperbrowser.ai).
## Next Steps
Deep dive into multi-step automation
Learn about single-action execution
Extract structured data with schemas
Configure LLMs and browser settings
# AI Function Calling
Source: https://hyperbrowser.ai/docs/integrations/ai-function-calling
Integrate Hyperbrowser scrape, crawl, and extract tools with OpenAI and Anthropic function calling
Hyperbrowser integrates seamlessly with OpenAI and Anthropic's function calling APIs, enabling you to enhance your AI applications with web scraping and crawling capabilities. This guide shows how to set up and use Hyperbrowser's scrape, crawl, and extract tools with OpenAI (and Anthropic-compatible definitions).
## Setup
### Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv openai
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv openai
```
```bash pip theme={null}
pip install hyperbrowser openai python-dotenv
```
```bash uv theme={null}
uv add @hyperbrowser/sdk dotenv openai
```
### Setup your Environment
To use Hyperbrowser with your code, you need an API Key from the [dashboard](https://app.hyperbrowser.ai). Add it to your `.env` as `HYPERBROWSER_API_KEY`. You will also need an `OPENAI_API_KEY`.
## Code
```typescript Node.js theme={null}
import OpenAI from "openai";
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { WebsiteCrawlTool, WebsiteScrapeTool,WebsiteExtractTool } from "@hyperbrowser/sdk/tools";
import { config } from "dotenv";
config();
// Initialize clients
const hb = new Hyperbrowser({ apiKey: process.env.HYPERBROWSER_API_KEY });
const oai = new OpenAI({
apiKey: process.env.OPENAI_API_KEY,
});
async function handleToolCall(tc) {
console.log("Handling tool call");
try {
const args = JSON.parse(tc.function.arguments);
console.log(`Tool call ID: ${tc.id}`);
console.log(`Function name: ${tc.function.name}`);
console.log(`Function args: ${JSON.stringify(args, null, 2)}`);
console.log("-".repeat(50));
if (
tc.function.name === WebsiteCrawlTool.openaiToolDefinition.function.name
) {
const response = await WebsiteCrawlTool.runnable(hb, args);
return {
tool_call_id: tc.id,
content: response,
role: "tool",
};
} else if (
tc.function.name === WebsiteScrapeTool.openaiToolDefinition.function.name
) {
const response = await WebsiteScrapeTool.runnable(hb, args);
return {
tool_call_id: tc.id,
content: response,
role: "tool",
};
} else if (
tc.function.name === WebsiteExtractTool.openaiToolDefinition.function.name
) {
const response = await WebsiteExtractTool.runnable(hb, args);
return {
tool_call_id: tc.id,
content: response,
role: "tool",
};
} else {
return {
tool_call_id: tc.id,
content: "Unknown tool call",
role: "tool",
};
}
} catch (e) {
console.error(e);
return {
role: "tool",
tool_call_id: tc.id,
content: `Error occurred: ${e}`,
};
}
}
const messages = [
{
role: "user",
content: "What does Hyperbrowser.ai do? Provide citations.",
},
];
async function openaiChat() {
while (true) {
const resp = await oai.beta.chat.completions.parse({
model: "gpt-4o-mini",
messages: messages,
tools: [
WebsiteCrawlTool.openaiToolDefinition,
WebsiteScrapeTool.openaiToolDefinition,
WebsiteExtractTool.openaiToolDefinition,
],
});
const choice = resp.choices[0];
messages.push(choice.message);
if (choice.finish_reason === "tool_calls") {
for (const tc of choice.message.tool_calls) {
const result = await handleToolCall(tc);
messages.push(result);
}
} else if (choice.finish_reason === "stop") {
console.log(choice.message.content);
break;
} else {
throw new Error("Unknown Error Occurred");
}
}
}
openaiChat();
```
```python Python theme={null}
import json
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.tools import WebsiteCrawlTool, WebsiteScrapeTool, WebsiteExtractTool
from openai import OpenAI
from openai.types.chat import (
ChatCompletionMessageToolCall,
ChatCompletionMessageParam,
ChatCompletionToolMessageParam,
)
from dotenv import load_dotenv
load_dotenv()
oai = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
hb = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def handle_tool_call(
tc: ChatCompletionMessageToolCall,
) -> ChatCompletionToolMessageParam:
print("Handling tool call")
try:
args = json.loads(tc.function.arguments)
print(f"Tool call ID: {tc.id}")
print(f"Function name: {tc.function.name}")
print(f"Function args: {args}")
print("-" * 50)
if (
tc.function.name
== WebsiteCrawlTool.openai_tool_definition["function"]["name"]
):
response = WebsiteCrawlTool.runnable(hb, args)
return {"tool_call_id": tc.id, "content": response, "role": "tool"}
elif (
tc.function.name
== WebsiteScrapeTool.openai_tool_definition["function"]["name"]
):
response = WebsiteScrapeTool.runnable(hb, args)
return {"tool_call_id": tc.id, "content": response, "role": "tool"}
elif (
tc.function.name
== WebsiteExtractTool.openai_tool_definition["function"]["name"]
):
response = WebsiteExtractTool.runnable(hb, args)
return {"tool_call_id": tc.id, "content": response, "role": "tool"}
else:
return {
"tool_call_id": tc.id,
"content": "Unknown tool call",
"role": "tool",
}
except Exception as e:
print(e)
return {"role": "tool", "tool_call_id": tc.id, "content": f"Error occurred: {e}"}
messages: list[ChatCompletionMessageParam] = [
{
"role": "user",
"content": "What does Hyperbrowser.ai do? Provide citations.",
},
]
while True:
response = oai.beta.chat.completions.parse(
messages=messages,
model="gpt-4o-mini",
tools=[
WebsiteCrawlTool.openai_tool_definition,
WebsiteScrapeTool.openai_tool_definition,
WebsiteExtractTool.openai_tool_definition,
],
)
choice = response.choices[0]
messages.append(choice.message) # type: ignore
if choice.finish_reason == "tool_calls":
for tc in choice.message.tool_calls: # type: ignore
result = handle_tool_call(tc=tc)
messages.append(result)
elif choice.finish_reason == "stop":
print(choice.message.content)
break
else:
raise Exception("Unknown Error Occured")
```
Hyperbrowser exposes a `WebsiteCrawlTool`, a `WebsiteScrapeTool`, and a `WebsiteExtractTool` to be used for OpenAI and Anthropic function calling. Each class provides a tool definition for both OpenAI and Anthropic that defines the schema which can be passed into the tools argument. Then, in the `handleToolCall`/`handle_tool_call` function, we parse the tool call arguments and dispatch the appropriate tool class `runnable` function based on the function name which returns the result of the scrape or crawl in formatted markdown.
# LangChain
Source: https://hyperbrowser.ai/docs/integrations/langchain
Integrate Hyperbrowser with LangChain for web scraping and document loading
Hyperbrowser provides a Document Loader integration with LangChain via the `langchain-hyperbrowser` package. It can be used to load the metadata and contents(in formatted markdown or html) of any site as a LangChain `Document`.
## Installation and Setup
To get started with `langchain-hyperbrowser`, you can install the package using pip:
```bash theme={null}
pip install langchain-hyperbrowser
```
And you should configure credentials by setting the following environment variables:
`HYPERBROWSER_API_KEY=`
You can get an API Key easily from the [dashboard](https://app.hyperbrowser.ai). Once you have your API Key, add it to your `.env` file as `HYPERBROWSER_API_KEY` or you can pass it via the `api_key` argument in the constructor.
## Document Loader
The `HyperbrowserLoader` class in `langchain-hyperbrowser` can easily be used to load content from any single page or multiple pages as well as crawl an entire site. The content can be loaded as markdown or html.
```python theme={null}
from langchain_hyperbrowser import HyperbrowserLoader
loader = HyperbrowserLoader(urls="https://example.com")
docs = loader.load()
print(docs[0])
```
## Advanced Usage
You can specify the operation to be performed by the loader. The default operation is `scrape`. For `scrape`, you can provide a single URL or a list of URLs to be scraped. For `crawl`, you can only provide a single URL. The `crawl` operation will crawl the provided page and subpages and return a document for each page.
```python theme={null}
loader = HyperbrowserLoader(
urls="https://hyperbrowser.ai", api_key="YOUR_API_KEY", operation="crawl"
)
```
Optional params for the loader can also be provided in the `params` argument. For more information on the supported params, you can see the params for [scraping](/docs/api-reference/create-new-scrape-job) or [crawling](/docs/api-reference/start-a-crawl-job).
```python theme={null}
loader = HyperbrowserLoader(
urls="https://example.com",
api_key="YOUR_API_KEY",
operation="scrape",
params={"scrape_options": {"include_tags": ["h1", "h2", "p"]}}
)
```
# LlamaIndex
Source: https://hyperbrowser.ai/docs/integrations/llamaindex
Integrate Hyperbrowser with LlamaIndex for web scraping and document loading
## Installation and Setup
To get started with LlamaIndex and Hyperbrowser, you can install the necessary packages using pip:
```bash theme={null}
pip install llama-index-core llama-index-readers-web hyperbrowser
```
And you should configure credentials by setting the following environment variables:
`HYPERBROWSER_API_KEY=`
You can get an API Key easily from the [dashboard](https://app.hyperbrowser.ai). Once you have your API Key, add it to your `.env` file as `HYPERBROWSER_API_KEY` or you can pass it via the `api_key` argument in the `HyperbrowserWebReader` constructor.
## Usage
Once you have your API Key and have installed the packages you can load webpages into LlamaIndex using `HyperbrowserWebReader`.
```python theme={null}
from llama_index.readers.web import HyperbrowserWebReader
reader = HyperbrowserWebReader(api_key="your_api_key_here")
```
To load data, you can specify the operation to be performed by the loader. The default operation is `scrape`. For `scrape`, you can provide a single URL or a list of URLs to be scraped. For `crawl`, you can only provide a single URL. The `crawl` operation will crawl the provided page and subpages and return a document for each page. `HyperbrowserWebReader` supports loading and lazy loading data in both sync and async modes.
```python theme={null}
documents = reader.load_data(
urls=["https://example.com"],
operation="scrape",
)
```
Optional params for the loader can also be provided in the `params` argument. For more information on the supported params, you can see the params on the [scraping guide](/docs/web-scraping/scrape).
The params will be Snake case for python code, so here for example, it is `max_pages` instead of `maxPages`.
```python theme={null}
# Scrape
documents = reader.load_data(
urls=["https://example.com"],
operation="scrape",
params={"scrape_options": {"include_tags": ["h1", "h2", "p"]}},
)
# Crawl
documents = reader.load_data(
urls=["https://example.com"],
operation="crawl",
params={
"max_pages": 10,
"scrape_options": {
"formats": ["markdown"],
},
"session_options": {
"use_stealth": True,
}
}
)
```
# Model Context Protocol
Source: https://hyperbrowser.ai/docs/integrations/model-context-protocol
Use Hyperbrowser with the Model Context Protocol (MCP) to enable web automation tools for AI models
## Overview
The Hyperbrowser MCP server provides a standardized interface for AI models to access powerful web automation capabilities like scraping, structured extraction, and crawling.
You can find the server implementation on GitHub: [hyperbrowserai/mcp](https://github.com/hyperbrowserai/mcp).

*With the Hyperbrowser MCP, Claude can browse the web.*
## Installation
### Prerequisites
* Node.js (v14 or later)
* npm or yarn
### Setup
1. Install the Hyperbrowser MCP server:
```bash theme={null}
npx hyperbrowser-mcp
```
## Configuration
### Client Setup
Configure your MCP client to launch the Hyperbrowser MCP server:
```json theme={null}
{
"mcpServers": {
"hyperbrowser": {
"command": "npx",
"args": ["hyperbrowser-mcp"],
"env": {
"HYPERBROWSER_API_KEY": "your-api-key"
}
}
}
}
```
### Alternative: Shell Script Wrapper
For clients that don't support an `env` field (for example, Cursor):
```json theme={null}
{
"mcpServers": {
"hyperbrowser": {
"command": "bash",
"args": ["/path/to/hyperbrowser-mcp/run_server.sh"]
}
}
}
```
Create `run_server.sh` and add your API key:
```bash theme={null}
#!/bin/bash
export HYPERBROWSER_API_KEY="your-api-key"
npx hyperbrowser-mcp
```
## Tools
### Scrape Webpage
Retrieve content from a URL in various formats.
* **Method**: `scrape_webpage`
* **Parameters**:
* `url`: `string` – The URL to scrape
* `outputFormat`: `string[]` – Output formats (`markdown`, `html`, `links`, `screenshot`)
* `apiKey`: `string` (optional) – API key override
* `sessionOptions`: `object` (optional) – Browser session configuration
Example:
```json theme={null}
{
"url": "https://example.com",
"outputFormat": ["markdown", "screenshot"],
"sessionOptions": {
"useStealth": true,
"acceptCookies": true
}
}
```
### Extract Structured Data
Extract data from webpages according to a prompt and optional schema.
* **Method**: `extract_structured_data`
* **Parameters**:
* `urls`: `string[]` – URLs to extract from (supports wildcards)
* `prompt`: `string` – Instructions for extraction
* `schema`: `object` (optional) – JSON schema for extracted data
* `apiKey`: `string` (optional) – API key override
* `sessionOptions`: `object` (optional) – Browser session configuration
Example:
```json theme={null}
{
"urls": ["https://example.com/products/*"],
"prompt": "Extract product name, price, and description",
"schema": {
"type": "object",
"properties": {
"name": { "type": "string" },
"price": { "type": "number" },
"description": { "type": "string" }
}
},
"sessionOptions": {
"useStealth": true
}
}
```
### Crawl Webpages
Navigate through multiple pages on a site, optionally following links.
* **Method**: `crawl_webpages`
* **Parameters**:
* `url`: `string` – Starting URL
* `outputFormat`: `string[]` – Desired output formats
* `followLinks`: `boolean` – Whether to follow links
* `maxPages`: `number` (default: 10) – Max pages to crawl
* `ignoreSitemap`: `boolean` (optional) – Skip the site's sitemap
* `apiKey`: `string` (optional) – API key override
* `sessionOptions`: `object` (optional) – Browser session configuration
Example:
```json theme={null}
{
"url": "https://example.com",
"outputFormat": ["markdown", "links"],
"followLinks": true,
"maxPages": 5,
"sessionOptions": {
"acceptCookies": true
}
}
```
## Session Options
All tools support these common session configuration options:
* `useStealth`: `boolean` – Make browser detection more difficult
* `useProxy`: `boolean` – Route traffic through proxies
* `solveCaptchas`: `boolean` – Automatically solve CAPTCHA challenges
* `acceptCookies`: `boolean` – Automatically handle cookie consent popups
# Stagehand
Source: https://hyperbrowser.ai/docs/integrations/stagehand
Run Stagehand automations on Hyperbrowser's managed browser sessions.
Stagehand can remote-control any Chrome session that exposes a CDP endpoint. Hyperbrowser supplies the always-on, privacy-hardened sessions—complete with stealth, proxies, ad blocking, and observability—so you can connect Stagehand to a reliable browser in seconds without standing up Chrome yourself.
## Why Hyperbrowser for your Stagehand project?
* **Consistent environments** – launch Chrome with stealth, custom profiles, and residential/static proxies through a single API call.
* **Enterprise controls** – restrict egress with managed proxies, enable audit logs via recordings/download inspection, and reuse sessions for multi-step flows.
* **Faster swaps** – migrate existing Browserbase + Stagehand projects by replacing the session creation call; the Stagehand client still points to the CDP URL it receives.
* **Live debugging** – every session automatically exposes `liveUrl`, letting you watch Stagehand steps in real time.
## Prerequisites
1. `HYPERBROWSER_API_KEY` loaded via `.env` or your secret manager.
2. Access to Stagehand via `@browserbasehq/stagehand` (or clone from docs.stagehand.dev).
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk @browserbasehq/stagehand dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk @browserbasehq/stagehand dotenv
```
## Quickstart (TypeScript)
This example mirrors the reference gist and shows Stagehand driving a Hyperbrowser session. It enables stealth, ad blocking, and accepts cookies out of the box, then attaches Stagehand to the returned CDP endpoint.
To see all the available session parameters, check out the [Session Parameters](/docs/sessions/parameters) page.
This example uses the new Stagehand v3 implementation.
```typescript Node.js theme={null}
import { config } from "dotenv";
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { Stagehand } from "@browserbasehq/stagehand";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function sleep(ms: number) {
return new Promise((resolve) => setTimeout(resolve, ms));
}
async function main() {
const session = await client.sessions.create({
useStealth: true,
adblock: true,
acceptCookies: true,
});
console.log(`Watch live: ${session.liveUrl}`);
const stagehand = new Stagehand({
env: "LOCAL",
localBrowserLaunchOptions: {
cdpUrl: session.wsEndpoint,
},
});
try {
await stagehand.init();
const page = stagehand.context.pages()[0];
await page.goto("https://hyperbrowser.ai");
await stagehand.act("Click on the 'Launch Browser' button");
console.log("Page title:", await page.title());
console.log("Waiting 5 seconds before closing");
await sleep(5_000);
} catch (err) {
console.error("Stagehand task failed:", err);
} finally {
await stagehand.close();
await client.sessions.stop(session.id);
}
}
main().catch((error) => {
console.error("Stagehand task failed:", error);
process.exit(1);
});
```
## Switch from Browserbase
| Browserbase setup | Hyperbrowser replacement |
| - | - |
| `const session = await browserbase.sessions.create({...});` | `const session = await hyperbrowser.sessions.create({...});` |
| `session.connectUrl` | `session.wsEndpoint` |
| `liveViewLinks.debuggerFullscreenUrl` | `session.liveUrl` |
| `BROWSERBASE_API_KEY` | `HYPERBROWSER_API_KEY` |
**Migration checklist from Browserbase to Hyperbrowser**
* Swap the SDK import to `@hyperbrowser/sdk`.
* Rename environment variables and CI secrets to `HYPERBROWSER_API_KEY`.
* Map Browserbase session flags to Hyperbrowser equivalents (stealth, proxies, uploads, downloads, custom extensions).
* Update dashboards/links to use `session.liveUrl`.
* Stop sessions explicitly with `client.sessions.stop(session.id)` to free capacity when Stagehand finishes.
## Troubleshooting & resources
* [Hyperbrowser sessions guide](/docs/sessions/create)
* [Stagehand Docs](https://docs.stagehand.dev)
* Need Browserbase parity (extensions, uploads, downloads)? Reuse the same API shape via `sessions.create` with the Hyperbrowser parameters documented under [Session Parameters](/docs/sessions/parameters).
# X402
Source: https://hyperbrowser.ai/docs/integrations/x402
Enable autonomous agent payments with X402 protocol
Hyperbrowser supports the X402 payment protocol, enabling AI agents to programmatically pay for web powered APIs at request time using cryptocurrency. Payment is handled inline via HTTP 402 responses, allowing truly autonomous agentic workflows.
## What is X402?
X402 is an open payment standard built around the HTTP `402 Payment Required` status code. It enables services to charge for access to their APIs and content directly over HTTP, allowing clients to programmatically pay for resources without accounts, sessions, or credential management. When an agent makes a request to an X402 endpoint, if payment is required, the server responds with `402 Payment Required` along with payment instructions. The agent can then complete the payment and retry the request.
To learn more about X402 and get started, check out the [X402 Documentation](https://x402.gitbook.io/x402).
***
## Supported Endpoints
Hyperbrowser provides X402-enabled versions of these API endpoints:
### Fetch
Fetch any page and get data in the formats you choose.
**X402 Endpoint:**
```
https://api.hyperbrowser.ai/x402/web/fetch
```
**Documentation:** See the [Fetch API documentation](/docs/web/fetch) for full parameter reference and examples.
### Search
Query the web and get clean, structured search results.
**X402 Endpoint:**
```
https://api.hyperbrowser.ai/x402/web/search
```
**Documentation:** See the [Search API documentation](/docs/web/search) for full parameter reference and examples.
***
## How It Works
1. **Agent makes request**: Your AI agent sends a request to an X402 endpoint (e.g., `https://api.hyperbrowser.ai/x402/web/fetch`)
2. **Payment required response**: If payment is needed, Hyperbrowser responds with `402 Payment Required` and payment instructions.
3. **Agent completes payment**: The agent uses the X402 protocol to complete the cryptocurrency payment
4. **Request retry**: After payment confirmation, the agent retries the original request
5. **Service delivered**: Hyperbrowser processes the request and returns the results
***
## Benefits for Agentic Workflows
* **Autonomous operation**: Agents can pay for services without human intervention
* **No accounts or credentials**: Access paid services without account creation or API key management
* **Cost transparency**: Agents know the exact cost before committing
* **Programmable budgets**: Agents can be configured with spending limits
***
## Getting Started
To integrate X402 into your agentic workflows:
1. Review the [X402 protocol documentation](https://x402.gitbook.io/x402)
2. Implement X402 client functionality in your agent
3. Use Hyperbrowser's X402 endpoints instead of standard API endpoints
4. Configure your agent with a crypto wallet and spending parameters
# Introduction
Source: https://hyperbrowser.ai/docs/introduction
Fast Cloud Browsers for your AI Agents
Hyperbrowser is a cloud browser platform for running automated browser sessions at scale. Control Chrome browsers in the cloud using Puppeteer, Playwright, or our SDKs—no infrastructure management required.
Try Hyperbrowser's advanced features:
**Ultra Stealth Mode** - The most advanced stealth mode for extra evasion from bot detection. [Learn more](/docs/sessions/stealth).
**AI Agent Capabilities** - Built in support for Claude, OpenAI, Gemini, Grok, and BrowserUse agents. [Learn more](/docs/agents/overview).
## Get started
Get up and running with Hyperbrowser in 5 minutes.
Learn about cloud browser sessions, and which configurations work best for you.
Discover and experiment with examples with a developer friendly tool.
***
## Key capabilities
Extract structured data from websites with powerful scraping APIs.
Review session recordings to debug and analyze behavior.
Integrate with MCP for enhanced AI capabilities.
Leverage advanced AI agentic models like Claude Computer Use.
Run code in fast, isolated cloud VMs built for agentic workflows.
***
## Resources
Complete API documentation for all available endpoints.
Connect with LangChain, LlamaIndex, and AI function calling.
Official SDKs for Node.js and Python.
# Pricing
Source: https://hyperbrowser.ai/docs/pricing
Understand how Hyperbrowser credits work and pricing for all services
Hyperbrowser tracks your usage via `credits` which can be acquired through a subscription or through a direct purchase. If you are on a subscription plan, then your plan credits will refresh when your plan renews. If you directly purchase credits, then those credits will expire after 12 months. You can purchase credits and subscribe to a plan on the [billing page](https://app.hyperbrowser.ai/settings?tab=billing).
> 1 credit = \$0.001
>
> 1,000 credits = \$1.00
Here is a breakdown of how credits are used:
## Browser Sessions
| Service | Credits | Cost |
| - | - | - |
| Browser Usage | 100 credits / hour | \$0.10 / hour |
| Proxy Data Usage | 10,000 credits / GB | \$10 / GB |
## Sandboxes
Sandbox usage is billed from both vCPU and memory for the time the sandbox is running. Memory is billed per allocated GiB (1024 MiB).
| Service | Credits | Cost |
| - | - | - |
| vCPU Usage | 45 credits / vCPU-hour | \$0.045 / vCPU-hour |
| Memory Usage | 15 credits / GiB-hour | \$0.015 / GiB-hour |
## Web API
| Service | Credits | Cost |
| - | - | - |
| Fetch | 1 credit / page | \$0.001 / page |
| Fetch (With Proxy) | +9 credits / page | +\$0.009 / page |
| Fetch (With JSON Output) | +5 credits / page | +\$0.005 / page |
| Search | 5 credits / query | \$0.005 / query |
| Search (With Location) | 10 credits / query | \$0.01 / query |
## Web Scraping
| Service | Credits | Cost |
| - | - | - |
| Scrape | 1 credit / page | \$0.001 / page |
| Scrape (With Proxy) | 10 credits / page | \$0.01 / page |
| AI Extract | 0.03 credits / output token | \$30 / million output tokens |
## AI Agents
Browser sessions used by AI Agents are tracked and consume credits according to the [Browser Sessions](#browser-sessions) rates listed above.
### HyperAgent
| Service | Credits | Cost |
| - | - | - |
| Per Step | 20 credits / step | \$0.02 / step |
### Browser Use
| Service | Credits | Cost |
| - | - | - |
| Per Step | 20 credits / step | \$0.02 / step |
### OpenAI CUA
| Model | Token Type | Credits | Cost |
| - | - | - | - |
| computer-use-preview | Input Tokens | 0.00315 credits / token | \$3.15 / million tokens |
| computer-use-preview | Output Tokens | 0.0126 credits / token | \$12.60 / million tokens |
| gpt-6-astra | Input Tokens | 0.0105 credits / token | \$10.50 / million tokens |
| gpt-6-astra | Output Tokens | 0.0525 credits / token | \$52.50 / million tokens |
| gpt-6.1-sol | Input Tokens | 0.0021 credits / token | \$2.10 / million tokens |
| gpt-6.1-sol | Output Tokens | 0.0105 credits / token | \$10.50 / million tokens |
| gpt-6-sol | Input Tokens | 0.0021 credits / token | \$2.10 / million tokens |
| gpt-6-sol | Output Tokens | 0.0105 credits / token | \$10.50 / million tokens |
| gpt-6-luna | Input Tokens | 0.000105 credits / token | \$0.105 / million tokens |
| gpt-6-luna | Output Tokens | 0.000525 credits / token | \$0.525 / million tokens |
| gpt-5.6-sol | Input Tokens | 0.00525 credits / token | \$5.25 / million tokens |
| gpt-5.6-sol | Output Tokens | 0.0315 credits / token | \$31.50 / million tokens |
| gpt-5.6-terra | Input Tokens | 0.002625 credits / token | \$2.625 / million tokens |
| gpt-5.6-terra | Output Tokens | 0.01575 credits / token | \$15.75 / million tokens |
| gpt-5.6-luna | Input Tokens | 0.00105 credits / token | \$1.05 / million tokens |
| gpt-5.6-luna | Output Tokens | 0.0063 credits / token | \$6.30 / million tokens |
| gpt-5.5 | Input Tokens | 0.00525 credits / token | \$5.25 / million tokens |
| gpt-5.5 | Output Tokens | 0.0315 credits / token | \$31.50 / million tokens |
| gpt-5.4 | Input Tokens | 0.002625 credits / token | \$2.625 / million tokens |
| gpt-5.4 | Output Tokens | 0.01575 credits / token | \$15.75 / million tokens |
| gpt-5.4-mini | Input Tokens | 0.0007875 credits / token | \$0.7875 / million tokens |
| gpt-5.4-mini | Output Tokens | 0.004725 credits / token | \$4.725 / million tokens |
### Claude Computer Use
| Model | Token Type | Credits | Cost |
| - | - | - | - |
| Haiku | Input Tokens | 0.00105 credits / token | \$1.05 / million tokens |
| Haiku | Output Tokens | 0.00525 credits / token | \$5.25 / million tokens |
| Sonnet 3.7–5 | Input Tokens | 0.00315 credits / token | \$3.15 / million tokens |
| Sonnet 3.7–5 | Output Tokens | 0.01575 credits / token | \$15.75 / million tokens |
| Sonnet 5.5 | Input Tokens | 0.0021 credits / token | \$2.10 / million tokens |
| Sonnet 5.5 | Output Tokens | 0.0105 credits / token | \$10.50 / million tokens |
| Opus 5.5 | Input Tokens | 0.0042 credits / token | \$4.20 / million tokens |
| Opus 5.5 | Output Tokens | 0.021 credits / token | \$21.00 / million tokens |
| Opus | Input Tokens | 0.00525 credits / token | \$5.25 / million tokens |
| Opus | Output Tokens | 0.02625 credits / token | \$26.25 / million tokens |
| Fable | Input Tokens | 0.0105 credits / token | \$10.50 / million tokens |
| Fable | Output Tokens | 0.0525 credits / token | \$52.50 / million tokens |
### Gemini Computer Use
| Model | Token Type | Credits | Cost |
| - | - | - | - |
| gemini-3.8-flash | Input Tokens | 0.001575 credits / token | \$1.575 / million tokens |
| gemini-3.8-flash | Output Tokens | 0.007875 credits / token | \$7.875 / million tokens |
| gemini-3.7-flash | Input Tokens | 0.001575 credits / token | \$1.575 / million tokens |
| gemini-3.7-flash | Output Tokens | 0.007875 credits / token | \$7.875 / million tokens |
| gemini-3.6-flash | Input Tokens | 0.001575 credits / token | \$1.575 / million tokens |
| gemini-3.6-flash | Output Tokens | 0.007875 credits / token | \$7.875 / million tokens |
| gemini-3.5-flash-lite | Input Tokens | 0.000315 credits / token | \$0.315 / million tokens |
| gemini-3.5-flash-lite | Output Tokens | 0.002625 credits / token | \$2.625 / million tokens |
| gemini-3.5-flash | Input Tokens | 0.001575 credits / token | \$1.575 / million tokens |
| gemini-3.5-flash | Output Tokens | 0.00945 credits / token | \$9.45 / million tokens |
| gemini-3-flash-preview | Input Tokens | 0.000525 credits / token | \$0.525 / million tokens |
| gemini-3-flash-preview | Output Tokens | 0.00315 credits / token | \$3.15 / million tokens |
### Meta Computer Use
| Model | Token Type | Credits | Cost |
| - | - | - | - |
| muse-spark-1.3 | Input Tokens | 0.0013125 credits / token | \$1.3125 / million tokens |
| muse-spark-1.3 | Output Tokens | 0.0044625 credits / token | \$4.4625 / million tokens |
| muse-spark-1.2 | Input Tokens | 0.0013125 credits / token | \$1.3125 / million tokens |
| muse-spark-1.2 | Output Tokens | 0.0044625 credits / token | \$4.4625 / million tokens |
| muse-spark-1.1 | Input Tokens | 0.0013125 credits / token | \$1.3125 / million tokens |
| muse-spark-1.1 | Output Tokens | 0.0044625 credits / token | \$4.4625 / million tokens |
### Grok Computer Use
| Model | Token Type | Credits | Cost |
| - | - | - | - |
| grok-4.7 | Input Tokens | 0.0021 credits / token | \$2.10 / million tokens |
| grok-4.7 | Output Tokens | 0.0063 credits / token | \$6.30 / million tokens |
### Jev Computer Use
| Model | Token Type | Credits | Cost |
| - | - | - | - |
| jev-1.13.0 | Input Tokens | 0.0000441 credits / token | \$0.0441 / million tokens |
| jev-1.13.0 | Output Tokens | 0 credits / token | \$0 / million tokens |
| jev-latest | Input Tokens | 0.0000441 credits / token | \$0.0441 / million tokens |
| jev-latest | Output Tokens | 0 credits / token | \$0 / million tokens |
| gemini-3.5-flash-lite (text helper) | Input Tokens | 0.000315 credits / token | \$0.315 / million tokens |
| gemini-3.5-flash-lite (text helper) | Output Tokens | 0.002625 credits / token | \$2.625 / million tokens |
# Quickstart
Source: https://hyperbrowser.ai/docs/quickstart
Start automating with Hyperbrowser Sessions
In this guide, you'll learn how to scrape Hacker News using scripts and powerful AI agents to automate with Hyperbrowser's cloud based browsers.
## Prerequisites
### Get Your API Key
Sign up and get your API key from the [Hyperbrowser Dashboard](https://app.hyperbrowser.ai/quickstart).
```bash theme={null}
export HYPERBROWSER_API_KEY=
```
### Install Dependencies
```bash npm theme={null}
npm install @hyperbrowser/sdk playwright-core dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk playwright-core dotenv
```
```bash pip theme={null}
pip install hyperbrowser playwright python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser playwright python-dotenv
```
```bash npm theme={null}
npm install @hyperbrowser/sdk puppeteer-core dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk puppeteer-core dotenv
```
## Part 1: Automate with Playwright or Puppeteer Scripts
Let's start by scraping the top story from Hacker News using Playwright or Puppeteer. This shows how you can launch Hyperbrowser sessions and connect to them with automation libraries like Playwright or Puppeteer.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function getTopStory() {
// Create a cloud browser session
const session = await client.sessions.create();
console.log("Session created:", session.id);
try {
// Connect Playwright to the cloud browser
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const context = browser.contexts()[0];
const page = context.pages()[0];
// Navigate to Hacker News
await page.goto("https://news.ycombinator.com/");
// Extract the first story title
const topStory = await page.evaluate(() => {
const titleElement = document.querySelector(".titleline > a");
return titleElement ? titleElement.textContent : null;
});
console.log("Top Story:", topStory);
return topStory;
} catch (err) {
console.error(`Encountered error: ${err}`);
} finally {
// Clean up the session
await client.sessions.stop(session.id);
}
}
getTopStory().catch(console.error);
```
```python Python theme={null}
import asyncio
import os
from playwright.async_api import async_playwright
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def get_top_story():
# Create a cloud browser session
session = await client.sessions.create()
print(f"Session created: {session.id}")
try:
# Connect Playwright to the cloud browser
async with async_playwright() as p:
browser = await p.chromium.connect_over_cdp(session.ws_endpoint)
context = browser.contexts[0]
page = context.pages[0]
# Navigate to Hacker News
await page.goto("https://news.ycombinator.com/")
# Extract the first story title
top_story = await page.evaluate("""
() => {
const titleElement = document.querySelector('.titleline > a');
return titleElement ? titleElement.textContent : null;
}
""")
print(f"Top Story: {top_story}")
return top_story
except Exception as e:
print(f"Error: {e}")
finally:
# Clean up the session
await client.sessions.stop(session.id)
if __name__ == "__main__":
asyncio.run(get_top_story())
```
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { connect } from "puppeteer-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function getTopStory() {
// Create a cloud browser session
const session = await client.sessions.create();
console.log("Session created:", session.id);
try {
// Connect Puppeteer to the cloud browser
const browser = await connect({
browserWSEndpoint: session.wsEndpoint,
defaultViewport: null,
});
const page = (await browser.pages())[0];
// Navigate to Hacker News
await page.goto("https://news.ycombinator.com/");
// Extract the first story title
const topStory = await page.evaluate(() => {
const titleElement = document.querySelector(".titleline > a");
return titleElement ? titleElement.textContent : null;
});
console.log("Top Story:", topStory);
return topStory;
} catch (err) {
console.error(`Encountered error: ${err}`);
} finally {
// Clean up the session
await client.sessions.stop(session.id);
}
}
getTopStory().catch(console.error);
```
## Part 2: AI-Powered Automation
Now let's solve the same problem using HyperAgent, our AI-powered browser automation tool. Instead of writing selectors and navigation logic, you simply describe what you want in natural language.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function getTopStoryWithAgent() {
const result = await client.agents.hyperAgent.startAndWait({
task: "Go to Hacker News and get the title of the first post",
});
console.log("Top Story:", result.data?.finalResult);
return result.data?.finalResult;
}
getTopStoryWithAgent().catch(console.error);
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def get_top_story_with_agent():
result = client.agents.hyper_agent.start_and_wait(
{"task": "Go to Hacker News and get the title of the first post"}
)
print(f"Top Story: {result.data.final_result}")
return result.data.final_result
if __name__ == "__main__":
get_top_story_with_agent()
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
from hyperbrowser.models import StartHyperAgentTaskParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def get_top_story_with_agent():
result = client.agents.hyper_agent.start_and_wait(
StartHyperAgentTaskParams(
task="Go to Hacker News and get the title of the first post"
)
)
print(f"Top Story: {result.data.final_result}")
return result.data.final_result
if __name__ == "__main__":
get_top_story_with_agent()
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/task/hyper-agent \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"task": "Go to Hacker News and get the title of the first post"
}'
```
### The Difference
With HyperAgent, you don't need to:
* Write CSS selectors
* Handle page navigation
* Manage browser state
* Debug DOM changes
Just describe what you want, and the AI handles the rest. It's perfect for complex workflows, dynamic sites, and rapid prototyping.
Hyperbrowser also integrates with other popular AI agent frameworks including [BrowserUse](/docs/agents/browser-use), [Claude Computer Use](/docs/agents/claude-computer-use), [Gemini Computer Use](/docs/agents/gemini-computer-use), [Meta Computer Use](/docs/agents/meta-computer-use), [Grok Computer Use](/docs/agents/grok-computer-use), [Jev Computer Use](/docs/agents/jev-computer-use), and [OpenAI CUA](/docs/agents/openai-cua). All agents benefit from Hyperbrowser's cloud infrastructure, stealth capabilities, and proxy support.
## Next Steps
Learn more about managing browsers sessions and other capabilities:
Learn about managing cloud browser sessions at scale
Extract data from websites with our scraping APIs
Dive deeper into HyperAgent and other AI automation tools
Explore the complete API documentation
## Additional Resources
### Browser Integrations
* [Puppeteer Integration](/docs/sessions/puppeteer) - Connect with Puppeteer for automation
* [Playwright Integration](/docs/sessions/playwright) - Connect with Playwright for automation
### SDKs
* [Node.js SDK](/docs/sdks/node) - TypeScript/JavaScript SDK reference
* [Python SDK](/docs/sdks/python) - Python SDK reference
### AI Agent Integrations
* [BrowserUse](/docs/agents/browser-use) - Open-source agent for fast browser automation
* [Claude Computer Use](/docs/agents/claude-computer-use) - Anthropic's Claude with computer capabilities
* [Gemini Computer Use](/docs/agents/gemini-computer-use) - Google's Gemini computer use agent
* [Meta Computer Use](/docs/agents/meta-computer-use) - Meta's Muse Spark computer use agent
* [Grok Computer Use](/docs/agents/grok-computer-use) - xAI's Grok 4.7 computer use agent
* [Jev Computer Use](/docs/agents/jev-computer-use) - Jev TypeSafe decision agent
* [OpenAI CUA](/docs/agents/openai-cua) - OpenAI's computer use agent
# Base Images
Source: https://hyperbrowser.ai/docs/sandboxes/base-images
Get started with preconfigured sandbox images. Base images are available for every team
and you can start building immediately.
## Quick Start
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
});
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "python",
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
)
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "node"
}'
```
## Available Images
| `imageName` | Best for | Includes |
| - | - | - |
| `node` | General Node.js and TypeScript workloads | Ubuntu 24.04, Node.js, `/home/ubuntu` working directory |
| `openclaw` | OpenClaw-powered agent and coding workflows | Ubuntu 24.04, Node.js, OpenClaw CLI, git, build tools |
| `claude-code` | Claude Code agent and coding workflows | Ubuntu 24.04, Node.js, Claude Code CLI, git, build tools |
| `codex` | Codex agent and coding workflows | Ubuntu 24.04, Node.js, Codex CLI, git, build tools |
| `python` | Python scripts, agents, and data workloads | Ubuntu 24.04, Python 3, pip, venv |
| `node-chromium` | Browser automation, desktop-style flows, and Playwright Chromium workloads | Node.js, Playwright Chromium, desktop session tooling |
## Runtime Defaults
All current base images share a few defaults:
* Ubuntu 24.04 as the base OS.
* A default `ubuntu` sudo user with home directory at `/home/ubuntu`.
* Default working directory set to `/home/ubuntu`.
## Which Image To Choose
* Choose `node` when you want a minimal JavaScript or TypeScript runtime.
* Choose `openclaw` when you want a Node runtime with the OpenClaw CLI preinstalled.
* Choose `claude-code` when you want a Node runtime with the Claude Code CLI preinstalled.
* Choose `codex` when you want a Node runtime with the Codex CLI preinstalled.
* Choose `python` when your workload is Python-first and you want `pip` and `venv` ready out of the box.
* Choose `node-chromium` when you need a browser-capable image with Chromium already installed and want the best starting point for browser automation or desktop-style workflows.
## Next Steps
* For launch parameters and runtime options, continue to [Creating Sandboxes](/docs/sandboxes/create).
* To package your own Docker image into a sandbox image, see [Build And Upload A Custom Image](/docs/sandboxes/custom-images).
# Sandbox CLI
Source: https://hyperbrowser.ai/docs/sandboxes/cli
Use the Hyperbrowser CLI to create and manage sandboxes
The `hx` CLI covers the sandbox workflows in this section: profile setup, sandbox creation, network policies, volume management, process execution, file transfer, terminals, snapshots, image builds, and port exposure.
Use the CLI when you want a shell-first workflow or need to script sandbox operations outside the SDKs. For the underlying API concepts, see [Creating Sandboxes](/docs/sandboxes/create), [Network Policies](/docs/sandboxes/networking), [Volumes](/docs/sandboxes/volumes), [Sandbox Lifecycle](/docs/sandboxes/lifecycle), [Sandbox Processes](/docs/sandboxes/processes), [Local Filesystem](/docs/sandboxes/filesystem/overview), [Base Images](/docs/sandboxes/base-images), [Build And Upload A Custom Image](/docs/sandboxes/custom-images), and [Sandbox Terminal](/docs/sandboxes/terminal).
## Install
See [Install](/docs/sandboxes/install) for macOS, Linux, Windows, and direct-download instructions.
## Configure Profiles
Login with your account
```bash theme={null}
hx auth login
```
Login with and save credentials to a profile:
```bash theme={null}
hx auth login --profile dev
```
List saved profiles and run a command with a specific profile:
```bash theme={null}
hx profile list
hx --profile dev vm list
```
The CLI uses the `default` profile unless you pass `--profile`. `hx configure`
prompts for an API key if `--api-key` is omitted and stores profiles in
`~/.hx_config/config`.
## Open The Dashboard
Launch the interactive dashboard:
```bash theme={null}
hx dash
```
`hx dash` requires an interactive terminal. Running `hx` without a subcommand
shows help; it does not open the dashboard automatically.
## Manage Volumes
Create, list, inspect, and delete persistent volumes:
```bash theme={null}
hx vm volumes create my-workspace
hx vm volumes ls
hx vm volumes get
hx vm volumes delete my-workspace --force
```
Attach a volume at sandbox launch:
```bash theme={null}
hx vm create node --mount :/mnt/workspace
hx vm create node --mount :/mnt/cache:ro
hx vm create node --mount source=,target=/mnt/workspace
hx vm create node --mount source=,target=/mnt/cache,readonly
```
See [Managing Volumes](/docs/sandboxes/volumes/managing) and
[Mounting Volumes](/docs/sandboxes/volumes/mounting) for API-level rules.
## Create Sandboxes
Launch from an image:
```bash theme={null}
hx vm create node
hx vm create node --port 3000 --port 9222:auth
hx vm create image:node
hx vm create --image-name node --region us --timeout-minutes 30
hx vm create --image-name node --image-id --port 3000 --auth
hx vm create --image-name node --image-id --enable-recording
```
Launch from a snapshot:
```bash theme={null}
hx vm create my_snapshot --snapshot
hx vm create snapshot:my_snapshot --snapshot-id
hx vm create --snapshot-name my_snapshot --snapshot-id
```
## Manage Outbound Network Access
Set a network policy when creating a sandbox:
```bash theme={null}
hx vm create node --allow-internet=false
hx vm create node \
--allow-internet=false \
--allow-out api.example.com \
--allow-out 1.1.1.1
```
Update a running sandbox or restore the default outbound policy:
```bash theme={null}
hx vm network set --deny-out 203.0.113.0/24
hx vm network set --allow-internet=false --allow-out api.example.com
hx vm network clear
```
Each `--allow-out` or `--deny-out` flag may be repeated. Network updates apply
to new connections; established connections may continue until they close. See
[Network Policies](/docs/sandboxes/networking) for rule precedence, target formats,
and domain behavior.
## Build And List Images
`hx vm image build` packages a local Docker image before uploading it, so image
builds require Docker to be available locally.
Build a Firecracker image from a local Docker image:
```bash theme={null}
hx vm image build node:22 custom_node
hx vm image build myorg/app:latest app_prod --env NODE_ENV=production --env PORT=3000
hx vm image build myorg/app:latest app_prod --entrypoint 'npm run start'
```
List available images:
```bash theme={null}
hx vm image list
hx --json vm image list
```
`hx vm image build` also supports `--wait-timeout` and `--poll-interval` when
you want to control how long the CLI waits and how often it polls build
status.
## Inspect, Stop, And Expose Sandboxes
List and inspect sandboxes:
```bash theme={null}
hx vm list
hx vm list --status closed --page 1 --limit 20
hx vm get
hx --json vm get
```
Stop a sandbox:
```bash theme={null}
hx vm stop
```
Expose or remove an exposed sandbox port:
```bash theme={null}
hx vm expose --port 3000
hx vm expose --port 3000 --auth
hx vm expose --port 3000:auth
hx vm unexpose --port 3000
```
`hx vm expose` currently accepts exactly one `--port` value. Use
`:auth` or `:public` when you want to override the default auth
mode from `--auth`.
Sandbox workflows are grouped under the `hx vm` namespace (for example
`hx vm create`, `hx vm process`, `hx vm connect`, and `hx vm volumes`).
## Run Commands
Run a one-shot command inside a sandbox:
```bash theme={null}
hx vm exec 'pwd'
hx vm exec 'echo hello'
hx vm exec --cwd /tmp --env FOO=bar 'echo $FOO'
```
## Manage Long-Running Processes
Start a background process:
```bash theme={null}
hx vm process start 'python3 -m http.server 3000'
hx --json vm process start 'sleep 30'
```
Inspect, wait for, and stream a process:
```bash theme={null}
hx vm process get
hx vm process list
hx vm process list --status running --limit 20
hx vm process wait
hx vm process stream
hx vm process stream --from-seq 120
```
Send input or terminate a process:
```bash theme={null}
printf 'hello\n' | hx vm process stdin --from-stdin --eof
hx vm process signal --signal TERM
hx vm process kill
```
## Work With Files
Copy files and directories between your machine and a sandbox:
```bash theme={null}
hx vm cp ./local.txt :/tmp/local.txt
hx vm cp :/tmp/local.txt ./downloaded.txt
hx vm cp ./dir/. :/tmp/dir-contents
hx vm cp :/tmp/a.txt :/tmp/b.txt
```
Read a file directly from a sandbox:
```bash theme={null}
hx vm read /tmp/file.txt
hx vm read --encoding raw /tmp/file.bin
hx vm read --encoding base64 /tmp/file.bin
hx vm read --offset 0 --length 128 /tmp/file.txt
```
The `:/path` shorthand is only supported by `hx vm cp`.
`hx vm read` expects the sandbox ID and path as separate arguments.
## Open An Interactive Terminal
Open a terminal in a selected sandbox:
```bash theme={null}
hx vm connect
```
Open a terminal in a specific sandbox:
```bash theme={null}
hx vm connect
```
Override the default command or set a working directory:
```bash theme={null}
hx vm connect --command python3 --arg -i
hx vm connect --cwd /tmp
hx vm connect --env FOO=bar
```
By default, `hx vm connect` starts `/bin/bash -i -l`.
## Create And Restore Snapshots
Create a memory snapshot from a running sandbox:
```bash theme={null}
hx vm snapshot create my_snapshot
```
List snapshots:
```bash theme={null}
hx vm snapshot list
hx vm snapshot list --limit 20
```
Restore from a snapshot:
```bash theme={null}
hx vm create my_snapshot --snapshot
hx vm create --snapshot-name my_snapshot --snapshot-id
```
## JSON Output
Most commands support `--json` for scripting:
```bash theme={null}
hx --json vm create node --port 3000:auth
hx --json vm get
hx --json vm expose --port 3000:public --auth
hx --json vm process start 'echo hi'
hx --json vm cp ./file.txt :/tmp/file.txt
```
# Creating Sandboxes
Source: https://hyperbrowser.ai/docs/sandboxes/create
Create and configure Hyperbrowser sandboxes
Sandboxes are isolated cloud environments built to scale with your agentic needs. Sandboxes have real runtime urls that you can call to run any kind of
workflow you need. Batteries are included.
Get started instantly with common use cases covered by [Base Images](/docs/sandboxes/base-images).
For more custom use cases, learn how to
[Build And Upload A Custom Image](/docs/sandboxes/custom-images).
## Quick Start
Create a sandbox from an image:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const sandbox = await client.sandboxes.create({
imageName: "node",
});
console.log("Sandbox ID:", sandbox.id);
console.log("Status:", sandbox.status);
console.log("Runtime URL:", sandbox.runtime.baseUrl);
console.log("Session URL:", sandbox.sessionUrl);
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
sandbox = client.sandboxes.create({"image_name": "python"})
print(f"Sandbox ID: {sandbox.id}")
print(f"Status: {sandbox.status}")
print(f"Runtime URL: {sandbox.runtime.base_url}")
print(f"Session URL: {sandbox.session_url}")
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSandboxParams
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
sandbox = client.sandboxes.create(CreateSandboxParams(image_name="python"))
print(f"Sandbox ID: {sandbox.id}")
print(f"Status: {sandbox.status}")
print(f"Runtime URL: {sandbox.runtime.base_url}")
print(f"Session URL: {sandbox.session_url}")
```
```bash CLI theme={null}
hx --json vm create node
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "node"
}'
```
Provide exactly one start source when creating a sandbox:
`imageName` or `snapshotName`.
## Start From a Snapshot
Start a new sandbox from a memory snapshot:
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
snapshotName: "my-node-snapshot",
snapshotId: "your-snapshot-id", // optional but recommended
});
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"snapshot_name": "my-python-snapshot",
"snapshot_id": "your-snapshot-id", # optional but recommended
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
snapshot_name="my-python-snapshot",
snapshot_id="your-snapshot-id", # optional but recommended
)
)
```
```bash CLI theme={null}
hx --json vm create snapshot:my-node-snapshot --snapshot-id your-snapshot-id
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"snapshotName": "my-node-snapshot",
"snapshotId": "your-snapshot-id"
}'
```
## Configuration Options
Customize the sandbox region, timeout, and recording settings:
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
region: "us-west",
timeoutMinutes: 30,
enableRecording: true,
});
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "python",
"region": "us-west",
"timeout_minutes": 30,
"enable_recording": True,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
region="us-west",
timeout_minutes=30,
enable_recording=True,
)
)
```
```bash CLI theme={null}
hx --json vm create node --region us-west --timeout-minutes 30 --enable-recording
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "node",
"region": "us-west",
"timeoutMinutes": 30,
"enableRecording": true
}'
```
## Resource Configuration
Set vCPU, memory, and disk size when launching from an image:
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
cpu: 2,
memoryMiB: 2048,
diskMiB: 8192,
});
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "python",
"cpu": 2,
"memory_mib": 2048,
"disk_mib": 8192,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
cpu=2,
memory_mib=2048,
disk_mib=8192,
)
)
```
```bash CLI theme={null}
hx --json vm create node --cpu 2 --memory 2048 --disk 8192
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "node",
"vcpus": 2,
"memMiB": 2048,
"diskSizeMiB": 8192
}'
```
For image launches, the default resource configuration is `2` vCPUs,
`2048` MiB of memory, and `8192` MiB of disk if you omit these fields.
Resource configuration is only supported when started from an image.
Snapshot launches use the snapshot's saved resource baseline.
## Mount Volumes
Attach persistent volumes at sandbox launch using the `mounts` field.
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
mounts: {
"/mnt/workspace": {
id: "550e8400-e29b-41d4-a716-446655440000",
type: "rw",
},
"/mnt/cache": {
id: "660e8400-e29b-41d4-a716-446655440000",
type: "ro",
},
},
});
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "node",
"mounts": {
"/mnt/workspace": {
"id": "550e8400-e29b-41d4-a716-446655440000",
"type": "rw",
},
"/mnt/cache": {
"id": "660e8400-e29b-41d4-a716-446655440000",
"type": "ro",
},
},
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams, SandboxVolumeMount
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="node",
mounts={
"/mnt/workspace": SandboxVolumeMount(
id="550e8400-e29b-41d4-a716-446655440000",
type="rw",
),
"/mnt/cache": SandboxVolumeMount(
id="660e8400-e29b-41d4-a716-446655440000",
type="ro",
),
},
)
)
```
```bash CLI theme={null}
hx vm create node --mount :/mnt/workspace
hx vm create node --mount :/mnt/cache:ro
hx vm create node --mount source=,target=/mnt/workspace
hx vm create node --mount source=,target=/mnt/cache,readonly
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "node",
"mounts": {
"/mnt/workspace": {
"id": "550e8400-e29b-41d4-a716-446655440000",
"type": "rw"
},
"/mnt/cache": {
"id": "660e8400-e29b-41d4-a716-446655440000",
"type": "ro"
}
}
}'
```
See [Managing Volumes](/docs/sandboxes/volumes/managing) and
[Mounting Volumes](/docs/sandboxes/volumes/mounting) for API details and mount
constraints.
### Common Parameters
Name of the sandbox image to start from. Provide this or `snapshotName`.
See [Base Images](/docs/sandboxes/base-images) for built-in values.
Optional specific image ID. Requires `imageName`.
Requested vCPU count for image launches. In the raw API, this field is
`vcpus`.
Requested memory in MiB for image launches. In the raw API, this field is
`memMiB`.
Requested disk size in MiB for image launches. In the raw API, this field is
`diskSizeMiB`.
Name of the snapshot to restore from. Provide this or `imageName`.
Optional specific snapshot ID. Requires `snapshotName`.
Region where the sandbox should start.
The `asia-south` region is only available on Enterprise and Select plans. [Get in touch](https://calendly.com/shri-hyperbrowser/demo) to learn more.
Maximum sandbox lifetime in minutes.
Enable sandbox recording.
Optional raw-API map of mount path to volume reference. Each key is the mount
path inside the sandbox (for example `/mnt/workspace`). Each value includes:
`id` (volume UUID), `type` (`rw` or `ro`, default `rw`), and optional
`shared` (currently reserved). This is available in the SDKs, via REST, and
via CLI mount flags.
Sets the fallback for outbound destinations that do not match an allow or
deny rule. Set it to `false` for a default-deny policy.
IPv4 addresses, IPv4 CIDRs, exact domains, or wildcard domains to allow.
IPv4 addresses or IPv4 CIDRs to deny. See
[Network Policies](/docs/sandboxes/networking) for precedence and domain rules.
## Sandbox Response
The create API returns a detailed sandbox object:
```json theme={null}
{
"id": "550e8400-e29b-41d4-a716-446655440000",
"status": "active",
"region": "us",
"sessionUrl": "https://app.hyperbrowser.ai/sandboxes/550e8400-e29b-41d4-a716-446655440000",
"runtime": {
"transport": "regional_proxy",
"host": "https://",
"baseUrl": "https:///sandbox/"
},
"network": {
"allowInternetAccess": true,
"allowOut": [],
"denyOut": []
},
"token": "eyJ...",
"tokenExpiresAt": "2026-03-12T21:14:00.000Z"
}
```
Unique sandbox identifier.
Current sandbox status.
Region where the sandbox is running.
Dashboard URL for the sandbox.
Runtime target used for direct runtime operations. Includes `transport`,
`host`, and `baseUrl`.
Effective outbound network policy. Includes `allowInternetAccess`,
`allowOut`, and `denyOut`.
Sandbox runtime bearer token.
Token expiration time in ISO 8601 format.
## Explore Sandbox Features
Manage running sandboxes, refresh handles, reconnect, and stop them cleanly
with [Sandbox Lifecycle](/docs/sandboxes/lifecycle).
Expose HTTP services, understand how sandbox URLs route to ports, and use
authenticated browser access with
[Sandbox Runtime URLs](/docs/sandboxes/runtime).
Restrict outbound traffic with IPv4, CIDR, and domain rules using
[Network Policies](/docs/sandboxes/networking).
Run one-shot commands, start background work, stream output, and manage
process state with [Sandbox Processes](/docs/sandboxes/processes).
Read, write, watch, upload, download, and presign file transfers with
[Local Filesystem](/docs/sandboxes/filesystem/overview).
Create persistent volumes and mount them at launch with
[Volumes](/docs/sandboxes/volumes).
Capture memory state and restore new sandboxes from it with
[Sandbox Snapshots](/docs/sandboxes/snapshots).
# Build And Upload A Custom Image
Source: https://hyperbrowser.ai/docs/sandboxes/custom-images
Package a local Docker image into a Hyperbrowser sandbox image with the CLI
Use a custom image when the built-in base images are close, but not enough. The CLI packages a local Docker image, uploads it, and dispatches the remote Firecracker image build. The build continues asynchronously after the command exits.
Custom image builds are currently CLI-first. You need Docker available on the machine running `hx vm image build`.
## Build A Custom Image
```bash theme={null}
build_id=$(hx --json vm image build node:22 my-node | jq -r '.id')
hx vm image wait "$build_id"
hx vm image build myorg/app:latest app-prod
hx vm image build myorg/app:latest app-prod --env NODE_ENV=production --env PORT=3000
hx vm image build myorg/app:latest app-prod --entrypoint 'npm run start'
```
The command does four things:
* Exports your local Docker image.
* Compresses the archive locally.
* Uploads it to the Hyperbrowser artifact store.
* Dispatches the remote image build and returns its build ID.
Use `hx vm image status ` for a single status check, or `hx vm image wait ` to poll until the build completes, fails, or is canceled.
## Launch A Sandbox From The Custom Image
After the image build completes, use the registered image name in normal sandbox creation flows.
```bash CLI theme={null}
hx vm create --image-name app-prod
```
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "app-prod",
});
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "app-prod",
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="app-prod",
)
)
```
## List Available Images
```bash theme={null}
hx vm image list
hx --json vm image list
```
Use this to confirm the uploaded image name before launch.
## Image Init Defaults
`hx vm image build` reads Docker-config environment variables and the image's `ENTRYPOINT`/`CMD`, then attaches them as launch-time defaults.
Use `--env KEY=VALUE` to add or override default environment variables.
Use `--entrypoint '...'` to override the detected startup command for sandboxes launched from that image.
Docker-defined environment variables and startup commands are auto-detected by default. Explicit `--env` and `--entrypoint` values take precedence over the detected values.
# Overview
Source: https://hyperbrowser.ai/docs/sandboxes/filesystem/overview
Understand the sandbox local filesystem surface and the operations available on it
This section covers the sandbox local filesystem: the files and directories inside a running sandbox.
Use it to:
* Inspect paths and metadata.
* Read and write text or bytes.
* Move, copy, and remove files.
* Watch for file system changes.
* Upload, download, and generate presigned transfer URLs.
This section is about the sandbox's local filesystem. Persistent volumes are a
separate surface and are documented in [Volumes](/docs/sandboxes/volumes).
## Quick Start
```typescript Node.js theme={null}
await sandbox.files.writeText("/tmp/hello.txt", "hello from sandbox");
const text = await sandbox.files.readText("/tmp/hello.txt");
console.log(text);
```
```python Python theme={null}
sandbox.files.write_text("/tmp/hello.txt", "hello from sandbox")
text = sandbox.files.read_text("/tmp/hello.txt")
print(text)
```
```bash CLI theme={null}
hx file write /tmp/hello.txt --data 'hello from sandbox'
hx file cat /tmp/hello.txt
```
# Read And Write Operations
Source: https://hyperbrowser.ai/docs/sandboxes/filesystem/read-write
Read, write, inspect, and manage files and directories inside a sandbox
Use the local filesystem API to inspect paths, read and write content, and manage files and directories inside a running sandbox.
## Read And Write Files
```typescript Node.js theme={null}
await sandbox.files.writeText("/tmp/hello.txt", "hello from sandbox");
const text = await sandbox.files.readText("/tmp/hello.txt");
const bytes = await sandbox.files.readBytes("/tmp/hello.txt");
console.log(text);
console.log(bytes.toString("utf8"));
```
```python Python theme={null}
sandbox.files.write_text("/tmp/hello.txt", "hello from sandbox")
text = sandbox.files.read_text("/tmp/hello.txt")
data = sandbox.files.read_bytes("/tmp/hello.txt")
print(text)
print(data.decode("utf-8"))
```
Batch writes are also supported:
```typescript Node.js theme={null}
await sandbox.files.write([
{ path: "/tmp/a.txt", data: "alpha" },
{ path: "/tmp/b.bin", data: Buffer.from([1, 2, 3]) },
]);
```
```python Python 1.0+ theme={null}
sandbox.files.write(
[
{"path": "/tmp/a.txt", "data": "alpha"},
{"path": "/tmp/b.bin", "data": bytes([1, 2, 3])},
]
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxFileWriteEntry
sandbox.files.write(
[
SandboxFileWriteEntry(path="/tmp/a.txt", data="alpha"),
SandboxFileWriteEntry(path="/tmp/b.bin", data=bytes([1, 2, 3])),
]
)
```
## Read Ranges And Different Formats
Use `read()` when you want to control offset, length, or representation.
Supported formats include:
* `text`
* `bytes`
* `blob`
* `stream`
## List And Inspect Paths
```typescript Node.js theme={null}
const exists = await sandbox.files.exists("/tmp/hello.txt");
const info = await sandbox.files.getInfo("/tmp/hello.txt");
const entries = await sandbox.files.list("/tmp", { depth: 1 });
console.log(exists);
console.log(info.permissions, info.owner, info.group);
console.log(entries.map((entry) => entry.path));
```
```python Python theme={null}
exists = sandbox.files.exists("/tmp/hello.txt")
info = sandbox.files.get_info("/tmp/hello.txt")
entries = sandbox.files.list("/tmp", depth=1)
print(exists)
print(info.permissions, info.owner, info.group)
print([entry.path for entry in entries])
```
```bash CLI theme={null}
hx file exists /tmp/hello.txt
hx file stat /tmp/hello.txt
hx file ls /tmp --depth 1
```
## Move, Copy, Remove, And Update Permissions
```typescript Node.js theme={null}
await sandbox.files.makeDir("/tmp/data", { parents: true });
await sandbox.files.writeText("/tmp/data/source.txt", "payload");
await sandbox.files.copy({
source: "/tmp/data/source.txt",
destination: "/tmp/data/copied.txt",
});
await sandbox.files.rename("/tmp/data/copied.txt", "/tmp/data/renamed.txt");
await sandbox.files.chmod({ path: "/tmp/data/renamed.txt", mode: "0644" });
await sandbox.files.remove("/tmp/data/source.txt");
```
```python Python theme={null}
sandbox.files.make_dir("/tmp/data", parents=True)
sandbox.files.write_text("/tmp/data/source.txt", "payload")
sandbox.files.copy(
source="/tmp/data/source.txt",
destination="/tmp/data/copied.txt",
)
sandbox.files.rename("/tmp/data/copied.txt", "/tmp/data/renamed.txt")
sandbox.files.chmod(path="/tmp/data/renamed.txt", mode="0644")
sandbox.files.remove("/tmp/data/source.txt")
```
```bash CLI theme={null}
hx file mkdir /tmp/data --parents
hx file cp /tmp/data/source.txt /tmp/data/copied.txt
hx file mv /tmp/data/copied.txt /tmp/data/renamed.txt
hx file chmod /tmp/data/renamed.txt 0644
hx file rm /tmp/data/source.txt
```
## Behavior Notes
The current filesystem implementation has a few practical guarantees worth knowing:
* `remove` is idempotent for missing paths.
* Removing a symlink removes the link, not the target.
* Directory listings do not recurse through symlink loops.
* Recursive copies preserve symlinks instead of expanding them indefinitely.
# Upload, Download, And Presigned URLs
Source: https://hyperbrowser.ai/docs/sandboxes/filesystem/transfers
Transfer file contents into and out of a sandbox with direct and presigned operations
Use direct transfers when your application already has the bytes in memory.
Use presigned URLs when you want a simple HTTP upload or download flow outside the SDK runtime client.
## Direct Upload And Download
```typescript Node.js theme={null}
const uploaded = await sandbox.files.upload("/tmp/data/upload.txt", "uploaded body");
const downloaded = await sandbox.files.download("/tmp/data/upload.txt");
console.log(uploaded.bytesWritten);
console.log(downloaded.toString("utf8"));
```
```python Python theme={null}
uploaded = sandbox.files.upload("/tmp/data/upload.txt", "uploaded body")
downloaded = sandbox.files.download("/tmp/data/upload.txt")
print(uploaded.bytes_written)
print(downloaded.decode("utf-8"))
```
```bash CLI theme={null}
hx file upload ./local.txt /tmp/local.txt
hx file download /tmp/local.txt ./downloaded.txt
```
## Presigned Upload And Download URLs
```typescript Node.js theme={null}
const upload = await sandbox.files.uploadUrl("/tmp/presigned.txt", {
oneTime: true,
expiresInSeconds: 60,
});
const download = await sandbox.files.downloadUrl("/tmp/presigned.txt", {
oneTime: true,
expiresInSeconds: 60,
});
console.log(upload.method, upload.url);
console.log(download.method, download.url);
```
```python Python theme={null}
upload = sandbox.files.upload_url(
"/tmp/presigned.txt",
one_time=True,
expires_in_seconds=60,
)
download = sandbox.files.download_url(
"/tmp/presigned.txt",
one_time=True,
expires_in_seconds=60,
)
print(upload.method, upload.url)
print(download.method, download.url)
```
```bash CLI theme={null}
hx file presign-upload /tmp/presigned.txt --expires-in-seconds 60 --one-time
hx file presign-download /tmp/presigned.txt --expires-in-seconds 60 --one-time
```
## Choosing Between Direct And Presigned Transfers
Use direct upload and download when:
* The SDK client already owns the bytes.
* You want the simplest in-process transfer path.
* You do not need to hand off the transfer to another service.
Use presigned URLs when:
* You want plain HTTP transfer behavior.
* Another service or worker should perform the upload or download.
* You want one-time or explicitly expiring transfer links.
## One-Time URLs
One-time presigned URLs are supported for both upload and download.
Only one concurrent request should succeed for a given one-time URL. Use this mode when you want handoff semantics rather than reusable access.
# Watch
Source: https://hyperbrowser.ai/docs/sandboxes/filesystem/watch
Stream file system events from a sandbox directory or path
Use file watches when you need change notifications from inside a running sandbox.
## Watch A Directory
```typescript Node.js theme={null}
const handle = await sandbox.files.watchDir(
"/tmp/watch",
async (event) => {
console.log(event.type, event.name);
},
{
recursive: true,
}
);
await sandbox.files.writeText("/tmp/watch/example.txt", "hello");
await handle.stop();
```
```python Python theme={null}
def on_event(event):
print(event.type, event.name)
handle = sandbox.files.watch_dir(
"/tmp/watch",
on_event,
recursive=True,
)
sandbox.files.write_text("/tmp/watch/example.txt", "hello")
handle.stop()
```
```bash CLI theme={null}
hx file watch /tmp/watch --recursive
```
## Event Model
Watch events are emitted as the filesystem changes.
The current runtime watch stream includes event sequence numbers, operation types, and the affected path.
The CLI prints them in this form:
```text theme={null}
```
## Resume From A Cursor
If you are consuming watch events from the CLI, you can resume from a known cursor:
```bash theme={null}
hx file watch /tmp/watch --recursive --from-cursor 120
```
## When To Use Watches
Use watches when you need to:
* React to generated outputs.
* Detect file writes from background processes.
* Build tail-like workflows around a sandbox workspace.
* Feed change streams into higher-level orchestration.
# Install
Source: https://hyperbrowser.ai/docs/sandboxes/install
Install the Hyperbrowser SDKs and the hx CLI
Set up the Hyperbrowser CLI (`hx`) and/or the SDKs to start working with sandboxes.
Latest release:
Install the latest release with the official install script:
```bash theme={null}
curl -fsSL https://www.hyperbrowser.ai/cli/install.sh | sh
```
In PowerShell, download and extract the latest Windows build:
```powershell theme={null}
$version = (Invoke-WebRequest -UseBasicParsing https://www.hyperbrowser.ai/cli/LATEST).Content.Trim()
$url = "https://www.hyperbrowser.ai/cli/hx_${version}_windows_amd64.tar.gz"
Invoke-WebRequest -UseBasicParsing $url -OutFile hx.tar.gz
tar -xzf hx.tar.gz
Move-Item .\hx.exe "$env:USERPROFILE\AppData\Local\Microsoft\WindowsApps\hx.exe"
```
For ARM64 Windows, replace `windows_amd64` with `windows_arm64`.
```bash theme={null}
curl -fsSL https://www.hyperbrowser.ai/cli/install.sh | sh -s -- --version
```
Each release is published as a tarball at:
```
https://www.hyperbrowser.ai/cli/hx__.tar.gz
```
| Platform | Target |
| - | - |
| Linux | `linux_amd64`, `linux_arm64` |
| macOS | `darwin_amd64`, `darwin_arm64` |
| Windows | `windows_amd64`, `windows_arm64` |
The snippet below resolves the current release and downloads the artifact that matches your platform:
```bash theme={null}
version=$(curl -fsSL https://www.hyperbrowser.ai/cli/LATEST)
curl -fsSL "https://www.hyperbrowser.ai/cli/hx_${version}_darwin_arm64.tar.gz" -o hx.tar.gz
tar -xzf hx.tar.gz
```
Confirm the CLI is on your `PATH`:
```bash theme={null}
hx version
```
```bash theme={null}
npm install @hyperbrowser/sdk
```
```bash theme={null}
pip install hyperbrowser
# or
uv add hyperbrowser
```
Grab your API key from the [Hyperbrowser Dashboard](https://app.hyperbrowser.ai/quickstart).
Export as an environment variable:
```bash theme={null}
export HYPERBROWSER_API_KEY=
```
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
```
```python Python theme={null}
import os
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
```
Save your default profile:
```bash theme={null}
hx auth login
```
See [Sandbox CLI](/docs/sandboxes/cli) for the full command reference, including named profiles and per-command overrides.
# Introduction
Source: https://hyperbrowser.ai/docs/sandboxes/introduction
Run your agents in the cloud fully harnessed
Hyperbrowser Sandboxes are isolated cloud environments purposefully built for agentic workflows.
Hyperbrowser Sandboxes are the fastest sandboxes with less than 50ms startup time.
Get started with a quick example in less than 30 seconds.
## Prerequisites
Install the SDK for your language or the `hx` CLI, then configure your API key. See [Install](/docs/sandboxes/install) for the full Node.js, Python, macOS, Linux, and Windows instructions.
## Quick Example
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
timeoutMinutes: 30,
});
const result = await sandbox.exec("node -e 'console.log(\"hello world\")'");
console.log(result.stdout);
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "python",
"timeout_minutes": 30,
}
)
result = sandbox.exec("""python3 -c 'print("hello world")'""")
print(result.stdout)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
timeout_minutes=30,
)
)
result = sandbox.exec("""python3 -c 'print("hello world")'""")
print(result.stdout)
```
```bash CLI theme={null}
sandbox_id=$(hx --json vm create node | jq -r '.data.id')
hx vm exec "$sandbox_id" 'node -e "console.log(\"hello world\")"'
hx vm stop "$sandbox_id"
```
## Start Here
Create your first sandbox with Node, Python or CLI.
Choose from a list of base images and get started.
Build and upload your own Docker based runtime image with the CLI.
Learn how to access your sandbox's local filesystem and stream data.
Create persistent volumes and mount them into sandboxes at launch.
# Sandbox Lifecycle
Source: https://hyperbrowser.ai/docs/sandboxes/lifecycle
Sandboxes instantly launch with the configuration you need. When a sandbox is created, it is always running until you stop it or it times out.
By default, this is based on your team’s default Session Timeout setting which you can change on the Settings page.
You can also configure the timeout per sandbox during creation. `timeoutMinutes`
is the sandbox's total lifetime beginning when its VM starts, not a timeout
applied separately to each SDK operation.
For example, set a custom session timeout when creating the sandbox:
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
timeoutMinutes: 30,
});
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "python",
"timeout_minutes": 30,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
timeout_minutes=30,
)
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "node",
"timeoutMinutes": 30
}'
```
## Getting a Sandbox
Fetch a detailed sandbox handle by ID:
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.get(
"550e8400-e29b-41d4-a716-446655440000",
);
const detail = sandbox.toJSON();
// Selected fields from detail:
// {
// "id": "550e8400-e29b-41d4-a716-446655440000",
// "teamId": "team-id",
// "status": "active",
// "endTime": null,
// "startTime": 1775822400000,
// "createdAt": "2026-04-10T12:00:00.000Z",
// "updatedAt": "2026-04-10T12:00:00.000Z",
// "region": "us-west",
// "sessionUrl": "https://app.hyperbrowser.ai/sandboxes/550e8400-e29b-41d4-a716-446655440000",
// "duration": 0,
// "proxyBytesUsed": 0,
// "cpu": 2,
// "memoryMiB": 2048,
// "diskMiB": 8192,
// "timeoutMinutes": 30,
// "runtime": {
// "transport": "regional_proxy",
// "host": "https://",
// "baseUrl": "https:///sandbox/"
// },
// "exposedPorts": [
// {
// "port": 3000,
// "auth": true,
// "url": "https:///",
// "browserUrl": "https:///_hb/auth?grant=&next=%2F",
// "browserUrlExpiresAt": "2026-04-11T12:00:00.000Z"
// }
// ],
// "token": "sandbox-runtime-token",
// "tokenExpiresAt": "2026-04-11T12:00:00.000Z"
// }
```
```python Python theme={null}
sandbox = client.sandboxes.get("550e8400-e29b-41d4-a716-446655440000")
detail = sandbox.to_dict()
# Selected fields from detail:
# {
# "id": "550e8400-e29b-41d4-a716-446655440000",
# "team_id": "team-id",
# "status": "active",
# "end_time": None,
# "start_time": 1775822400000,
# "created_at": "",
# "updated_at": "",
# "region": "us-west",
# "session_url": "https://app.hyperbrowser.ai/sandboxes/550e8400-e29b-41d4-a716-446655440000",
# "duration": 0,
# "proxy_bytes_used": 0,
# "cpu": 2,
# "memory_mib": 2048,
# "disk_mib": 8192,
# "timeout_minutes": 30,
# "runtime": {
# "transport": "regional_proxy",
# "host": "https://",
# "base_url": "https:///sandbox/"
# },
# "exposed_ports": [
# {
# "port": 3000,
# "auth": True,
# "url": "https:///",
# "browser_url": "https:///_hb/auth?grant=&next=%2F",
# "browser_url_expires_at": ""
# }
# ],
# "token": "sandbox-runtime-token",
# "token_expires_at": ""
# }
```
```bash cURL theme={null}
curl -X GET https://api.hyperbrowser.ai/api/sandbox/550e8400-e29b-41d4-a716-446655440000 \
-H "x-api-key: YOUR_API_KEY"
# Abbreviated REST response:
# {
# "id": "550e8400-e29b-41d4-a716-446655440000",
# "teamId": "team-id",
# "status": "active",
# "endTime": null,
# "startTime": 1775822400000,
# "createdAt": "2026-04-10T12:00:00.000Z",
# "updatedAt": "2026-04-10T12:00:00.000Z",
# "region": "us-west",
# "sessionUrl": "https://app.hyperbrowser.ai/sandboxes/550e8400-e29b-41d4-a716-446655440000",
# "duration": 0,
# "proxyBytesUsed": 0,
# "vcpus": 2,
# "memMiB": 2048,
# "diskSizeMiB": 8192,
# "timeoutMinutes": 30,
# "runtime": {
# "transport": "regional_proxy",
# "host": "https://",
# "baseUrl": "https:///sandbox/"
# },
# "exposedPorts": [
# {
# "port": 3000,
# "auth": true,
# "url": "https:///",
# "browserUrl": "https:///_hb/auth?grant=&next=%2F",
# "browserUrlExpiresAt": "2026-04-11T12:00:00.000Z"
# }
# ],
# "token": "sandbox-runtime-token",
# "tokenExpiresAt": "2026-04-11T12:00:00.000Z"
# }
```
The REST API returns VM sizing fields as `vcpus`, `memMiB`, and
`diskSizeMiB`. The Node SDK exposes the same values as `cpu`, `memoryMiB`, and
`diskMiB`; the Python SDK exposes them as `cpu`, `memory_mib`, and `disk_mib`.
`duration` is the elapsed runtime in milliseconds after a sandbox ends, so it
is `0` while the sandbox is active. `timeoutMinutes` is the configured total
lifetime.
## Listing Sandboxes
List your sandboxes with optional filtering:
```typescript Node.js theme={null}
const response = await client.sandboxes.list({
status: "active",
page: 1,
limit: 20,
});
// Selected response fields:
// {
// "totalCount": 1,
// "page": 1,
// "perPage": 20,
// "sandboxes": [
// {
// "id": "sandbox-id",
// "status": "active",
// "region": "us-west"
// }
// ]
// }
```
```python Python 1.0+ theme={null}
response = client.sandboxes.list(
{
"status": "active",
"page": 1,
"limit": 20,
}
)
# Selected fields from response.model_dump():
# {
# "total_count": 1,
# "page": 1,
# "per_page": 20,
# "sandboxes": [
# {
# "id": "sandbox-id",
# "status": "active",
# "region": "us-west"
# }
# ]
# }
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxListParams
response = client.sandboxes.list(
SandboxListParams(
status="active",
page=1,
limit=20,
)
)
# Selected fields from response.model_dump():
# {
# "total_count": 1,
# "page": 1,
# "per_page": 20,
# "sandboxes": [
# {
# "id": "sandbox-id",
# "status": "active",
# "region": "us-west"
# }
# ]
# }
```
```bash cURL theme={null}
curl -X GET "https://api.hyperbrowser.ai/api/sandboxes?status=active&page=1&limit=20" \
-H "x-api-key: YOUR_API_KEY"
# Abbreviated REST response:
# {
# "totalCount": 1,
# "page": 1,
# "perPage": 20,
# "sandboxes": [
# {
# "id": "550e8400-e29b-41d4-a716-446655440000",
# "status": "active",
# "region": "us-west"
# }
# ]
# }
```
## Refreshing and Connecting
Connect to an existing sandbox or refresh the runtime token.
Create and get responses for an active sandbox include a runtime token that
is valid for 24 hours. The Node and Python SDKs automatically obtain a fresh
token shortly before expiry and retry a replayable runtime HTTP request once
with a fresh token after a `401`. Call `refresh()` when you also want to
refresh the handle's cached sandbox metadata.
For a more detailed guide to runtime tokens, see [Sandbox Runtime URLs](/docs/sandboxes/runtime).
```typescript Node.js theme={null}
// Refresh an existing handle
await sandbox.refresh();
// Reconnect from an ID
const reattached = await client.sandboxes.connect(
"550e8400-e29b-41d4-a716-446655440000",
);
// Selected fields from reattached.toJSON():
// {
// "id": "550e8400-e29b-41d4-a716-446655440000",
// "runtime": {
// "baseUrl": "https:///sandbox/"
// },
// "tokenExpiresAt": "<24 hours after the reconnect request>"
// }
```
```python Python theme={null}
# Refresh an existing handle
sandbox.refresh()
# Reconnect from an ID
reattached = client.sandboxes.connect(
"550e8400-e29b-41d4-a716-446655440000"
)
# Selected fields from reattached.to_dict():
# {
# "id": "550e8400-e29b-41d4-a716-446655440000",
# "runtime": {
# "base_url": "https:///sandbox/"
# },
# "token_expires_at": ""
# }
```
## Stopping a Sandbox
Always stop a sandbox when you are done with it:
```typescript Node.js theme={null}
const response = await sandbox.stop();
// {
// "success": true
// }
```
```python Python theme={null}
response = sandbox.stop()
# response.model_dump()
# {
# "success": True
# }
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/sandbox/SANDBOX_ID/stop \
-H "x-api-key: YOUR_API_KEY"
# {
# "success": true
# }
```
Stopping a sandbox is safe to call more than once.
## Exposing Ports
Expose a port when you need a custom process to be accessible from outside from the sandbox.
This provides a custom runtime URL for that exposed port.
```typescript Node.js theme={null}
const exposure = await sandbox.expose({
port: 3000,
auth: true,
});
// {
// "port": 3000,
// "auth": true,
// "url": "https:///",
// "browserUrl": "https:///_hb/auth?grant=&next=%2F",
// "browserUrlExpiresAt": "2026-04-11T12:00:00Z"
// }
// Helper for deriving the URL for an exposed port:
// sandbox.getExposedUrl(3000)
```
```python Python 1.0+ theme={null}
exposure = sandbox.expose(
{
"port": 3000,
"auth": True,
}
)
# exposure.model_dump(mode="json")
# {
# "port": 3000,
# "auth": True,
# "url": "https:///",
# "browser_url": "https:///_hb/auth?grant=&next=%2F",
# "browser_url_expires_at": "2026-04-11T12:00:00Z"
# }
# Helper for deriving the URL for an exposed port:
# sandbox.get_exposed_url(3000)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxExposeParams
exposure = sandbox.expose(
SandboxExposeParams(
port=3000,
auth=True,
)
)
# exposure.model_dump(mode="json")
# {
# "port": 3000,
# "auth": True,
# "url": "https:///",
# "browser_url": "https:///_hb/auth?grant=&next=%2F",
# "browser_url_expires_at": "2026-04-11T12:00:00Z"
# }
# Helper for deriving the URL for an exposed port:
# sandbox.get_exposed_url(3000)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox/SANDBOX_ID/expose \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"port": 3000,
"auth": true
}'
# {
# "port": 3000,
# "auth": true,
# "url": "https:///",
# "browserUrl": "https:///_hb/auth?grant=&next=%2F",
# "browserUrlExpiresAt": "2026-04-11T12:00:00Z"
# }
```
If `auth` is enabled, send the sandbox bearer token when calling the exposed URL:
```typescript Node.js theme={null}
const detail = await sandbox.info();
const healthUrl = new URL("health", exposure.url);
const response = await fetch(healthUrl, {
headers: {
Authorization: `Bearer ${detail.token}`,
},
});
const body = await response.text();
console.log(response.status); // 200
console.log(body); //
```
```python Python theme={null}
import requests
detail = sandbox.info()
response = requests.get(
f"{exposure.url.rstrip('/')}/health",
headers={"Authorization": f"Bearer {detail.token}"},
timeout=30,
)
response.raise_for_status()
body = response.text
print(response.status_code) # 200
print(body) #
```
```bash cURL theme={null}
curl "${EXPOSED_URL%/}/health" \
-H "Authorization: Bearer $SANDBOX_TOKEN"
```
For a dedicated guide to runtime URLs, exposed service URLs, browser auth
links, and end-to-end examples, see [Sandbox Runtime URLs](/docs/sandboxes/runtime).
Port `4001` is reserved for the sandbox runtime API url and cannot be exposed.
# Sandbox Networking
Source: https://hyperbrowser.ai/docs/sandboxes/networking
Control outbound network access from a sandbox
Sandbox network policies control connections initiated by workloads inside a
sandbox. Outbound internet access is allowed by default. Add a policy at
creation time or update a running sandbox when you need to restrict its access.
Network policies affect outbound traffic only. They do not change inbound
access through [exposed ports](/docs/sandboxes/runtime).
## Policy Fields
Controls outbound internet access for destinations not covered by
`allowOut` or `denyOut`. Set it to `false` for a default-deny policy.
Outbound destinations to allow. Entries can be IPv4 addresses, IPv4 CIDR
ranges, exact domains, or wildcard domains such as `*.example.com`.
Outbound destinations to deny. Entries can be IPv4 addresses or IPv4 CIDR
ranges. Domain entries are not supported in `denyOut`.
An empty policy is equivalent to:
```json theme={null}
{
"allowInternetAccess": true,
"allowOut": [],
"denyOut": []
}
```
### Rule Precedence
When `allowOut` and `denyOut` both match the same destination, `allowOut` takes
precedence.
For example:
```json theme={null}
{
"allowInternetAccess": true,
"allowOut": ["203.0.113.10"],
"denyOut": ["203.0.113.0/24"]
}
```
* `203.0.113.10` is allowed because it matches both rules and `allowOut` wins.
* Other addresses in `203.0.113.0/24` are blocked.
* Destinations outside that range are allowed because
`allowInternetAccess` is `true`.
A domain in `allowOut` requires a default-deny boundary: set
`allowInternetAccess` to `false` or include `0.0.0.0/0` in `denyOut`.
Hyperbrowser rejects the policy otherwise.
IPv4 addresses are normalized to `/32` CIDRs, CIDRs are normalized to their
network address, domains are lowercased, and duplicate entries are removed.
IPv4 and CIDR rules match the destination across all ports and protocols.
IPv6 policy targets are not currently supported.
## Create With a Network Policy
The following policy blocks outbound internet access except for the exact
domain `api.github.com` and the IPv4 address `1.1.1.1`.
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(api_key="YOUR_API_KEY")
sandbox = client.sandboxes.create(
{
"image_name": "python",
"allow_internet_access": False,
"allow_out": ["api.github.com", "1.1.1.1"],
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSandboxParams
client = Hyperbrowser(api_key="YOUR_API_KEY")
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
allow_internet_access=False,
allow_out=["api.github.com", "1.1.1.1"],
)
)
```
```bash CLI theme={null}
hx vm create python \
--allow-internet=false \
--allow-out api.github.com \
--allow-out 1.1.1.1
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "python",
"allowInternetAccess": false,
"allowOut": ["api.github.com", "1.1.1.1"]
}'
```
To allow internet access by default while blocking selected networks, leave
`allowInternetAccess` as `true` and add IPv4 or CIDR entries to `denyOut`.
## Update a Running Sandbox
Update the policy without restarting the sandbox:
```python Python 1.0+ theme={null}
result = sandbox.update_network(
{
"allow_internet_access": False,
"allow_out": ["api.github.com"],
"deny_out": [],
}
)
print(result.network)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxNetworkPolicy
result = sandbox.update_network(
SandboxNetworkPolicy(
allow_internet_access=False,
allow_out=["api.github.com"],
deny_out=[],
)
)
print(result.network)
```
```bash CLI theme={null}
hx vm network set "$sandbox_id" \
--allow-internet=false \
--allow-out api.github.com
```
```bash cURL theme={null}
curl -X PUT \
"https://api.hyperbrowser.ai/api/sandbox/$sandbox_id/network" \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"allowInternetAccess": false,
"allowOut": ["api.github.com"],
"denyOut": []
}'
```
The update endpoint uses patch semantics:
* Omitted fields keep their current values.
* A supplied `allowOut` or `denyOut` array replaces that entire list.
* Supply an empty array to clear one list.
The Python SDK sends only fields supplied in the request.
The CLI sends only the flags provided to `hx vm network set`.
Policy changes apply to new connections. Established connections may
continue until they close.
## Clear a Network Policy
Restore the default outbound policy and remove all allow and deny entries.
```python Python theme={null}
sandbox.clear_network()
```
```bash CLI theme={null}
hx vm network clear "$sandbox_id"
```
```bash cURL theme={null}
curl -X PUT \
"https://api.hyperbrowser.ai/api/sandbox/$sandbox_id/network" \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"allowInternetAccess": true,
"allowOut": [],
"denyOut": []
}'
```
Sandbox create, get, and network-update responses include the effective policy
under the `network` field.
## Domain Rules
Domain entries are supported only in `allowOut`, and a policy containing one
must have a default-deny boundary. Set `allowInternetAccess` to `false` or add
`0.0.0.0/0` to `denyOut`.
* `example.com` matches only `example.com`, not its subdomains.
* `*.example.com` matches subdomains such as `api.example.com`, but not the
apex `example.com`. Add both entries when you need both forms.
* Domain rules apply to HTTP on port `80` and HTTPS on port `443`. HTTPS
matching uses the TLS server name. For other ports or protocols, allow the
destination with an IPv4 or CIDR entry.
* While domain rules are active, DNS permits `A`, `AAAA`, `SVCB`, and `HTTPS`
lookups for allowed names and refuses lookups for other names.
* Domain-based connections reject hostnames that resolve to loopback, private,
carrier-grade NAT, or link-local addresses.
Domain rules require the application to send a hostname in the HTTP `Host`
header or HTTPS TLS server name. Connections that do not expose a matching
hostname need an IPv4 or CIDR allow entry.
## Common Patterns
### Block All Outbound Access
```json theme={null}
{
"allowInternetAccess": false,
"allowOut": [],
"denyOut": []
}
```
### Allow Only Selected Web Domains
```json theme={null}
{
"allowInternetAccess": false,
"allowOut": ["example.com", "*.example.com"],
"denyOut": []
}
```
### Block a Network With One Exception
Because allow rules take precedence, a narrow allow can override a broader
deny:
```json theme={null}
{
"allowInternetAccess": true,
"allowOut": ["203.0.113.10"],
"denyOut": ["203.0.113.0/24"]
}
```
# OpenClaw + WhatsApp
Source: https://hyperbrowser.ai/docs/sandboxes/openclaw-whatsapp
Connect OpenClaw to WhatsApp in a Hyperbrowser sandbox
Run [OpenClaw](https://docs.openclaw.ai) in a Hyperbrowser sandbox and connect it
to WhatsApp. Your agent receives and responds to WhatsApp messages through the
built-in WhatsApp Web / Baileys channel.
## How It Works
1. Create a sandbox from the `openclaw` base image.
2. Configure OpenClaw with your model provider and the WhatsApp channel.
3. Link a WhatsApp account by scanning a QR code.
4. Start the OpenClaw gateway and message your agent.
## Prerequisites
* A [Hyperbrowser API key](https://app.hyperbrowser.ai)
* An OpenAI API key (or another [supported model provider](https://docs.openclaw.ai/start/openclaw))
* A WhatsApp account to link to the agent
* The Hyperbrowser [Node SDK](/docs/sdks/node) or [Python SDK](/docs/sdks/python)
## Quick Start
Create a sandbox, configure the WhatsApp channel with an allowlisted phone number,
and start the gateway:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
import { StringDecoder } from "node:string_decoder";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
// 1. Create a sandbox from the openclaw image
const sandbox = await client.sandboxes.create({
imageName: "openclaw",
timeoutMinutes: 60,
});
console.log("Sandbox ID:", sandbox.id);
console.log("Session URL:", sandbox.sessionUrl);
// 2. Set the default model
await sandbox.exec(
'openclaw config set agents.defaults.model.primary "openai/gpt-5.2"'
);
// 3. Install the WhatsApp plugin
await sandbox.exec("openclaw plugins install @openclaw/whatsapp");
const gatewayToken = process.env.OPENCLAW_APP_TOKEN ?? crypto.randomUUID();
const gatewayPort = Number(process.env.GATEWAY_PORT ?? 18789);
await sandbox.exec("openclaw config set gateway.mode local");
// 4. Configure the WhatsApp channel (dedicated-number allowlist)
const ownerNumber = "+15551234567"; // replace with the phone number allowed to DM the bot
await sandbox.exec(
[
"openclaw config set channels.whatsapp.dmPolicy allowlist",
`openclaw config set channels.whatsapp.allowFrom '["${ownerNumber}"]' --strict-json`,
"openclaw config set channels.whatsapp.groupPolicy disabled",
].join(" && ")
);
// 5. Link WhatsApp in a PTY terminal so the QR renders correctly
console.log("\nScan the QR code below with your WhatsApp app:\n");
const login = await sandbox.terminal.create({
command: "bash",
args: ["-lc", "openclaw channels login --channel whatsapp"],
rows: 40,
cols: 120,
});
const loginConnection = await login.attach();
const loginDecoder = new StringDecoder("utf8");
for await (const event of loginConnection.events()) {
if (event.type === "output") {
process.stdout.write(loginDecoder.write(event.raw));
continue;
}
break;
}
process.stdout.write(loginDecoder.end());
await loginConnection.close();
console.log("\nWhatsApp linked successfully.");
// 6. Start the gateway
const gateway = await sandbox.processes.start(
`openclaw gateway --bind lan --port ${gatewayPort} --auth token --token "${gatewayToken}"`,
{ env: { OPENAI_API_KEY: process.env.OPENAI_API_KEY! } }
);
// 7. Wait for the gateway to become ready
for (let i = 0; i < 15; i++) {
const check = await sandbox.exec("openclaw gateway status --require-rpc");
if (check.exitCode === 0) break;
await new Promise((r) => setTimeout(r, 2000));
}
console.log("Gateway started. You can now message your agent on WhatsApp.");
```
```python Python 1.0+ theme={null}
import codecs
import os
import time
import uuid
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
# 1. Create a sandbox from the openclaw image
sandbox = client.sandboxes.create(
{
"image_name": "openclaw",
"timeout_minutes": 60,
}
)
print(f"Sandbox ID: {sandbox.id}")
print(f"Session URL: {sandbox.session_url}")
# 2. Set the default model
sandbox.exec('openclaw config set agents.defaults.model.primary "openai/gpt-5.2"')
# 3. Install the WhatsApp plugin
sandbox.exec("openclaw plugins install @openclaw/whatsapp")
gateway_token = os.environ.get("OPENCLAW_APP_TOKEN", str(uuid.uuid4()))
gateway_port = int(os.environ.get("GATEWAY_PORT", "18789"))
sandbox.exec("openclaw config set gateway.mode local")
# 4. Configure the WhatsApp channel (dedicated-number allowlist)
owner_number = "+15551234567" # replace with the phone number allowed to DM the bot
allow_json = f'["{owner_number}"]'
sandbox.exec(
" && ".join(
[
"openclaw config set channels.whatsapp.dmPolicy allowlist",
f"openclaw config set channels.whatsapp.allowFrom '{allow_json}' --strict-json",
"openclaw config set channels.whatsapp.groupPolicy disabled",
]
)
)
# 5. Link WhatsApp in a PTY terminal so the QR renders correctly
print("\nScan the QR code below with your WhatsApp app:\n")
login = sandbox.terminal.create(
{
"command": "bash",
"args": ["-lc", "openclaw channels login --channel whatsapp"],
"rows": 40,
"cols": 120,
}
)
login_connection = login.attach()
login_decoder = codecs.getincrementaldecoder("utf-8")()
for event in login_connection.events():
if event.type == "output":
print(login_decoder.decode(event.raw), end="")
continue
break
print(login_decoder.decode(b"", final=True), end="")
login_connection.close()
print("\nWhatsApp linked successfully.")
# 6. Start the gateway
gateway = sandbox.processes.start(
f'openclaw gateway --bind lan --port {gateway_port} --auth token --token "{gateway_token}"',
env={"OPENAI_API_KEY": os.environ["OPENAI_API_KEY"]},
)
# 7. Wait for the gateway to become ready
for _ in range(15):
check = sandbox.exec("openclaw gateway status --require-rpc")
if check.exit_code == 0:
break
time.sleep(2)
print("Gateway started. You can now message your agent on WhatsApp.")
```
```python Python (legacy) theme={null}
import codecs
import os
import time
import uuid
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import (
CreateSandboxParams,
SandboxTerminalCreateParams,
)
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
# 1. Create a sandbox from the openclaw image
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="openclaw",
timeout_minutes=60,
)
)
print(f"Sandbox ID: {sandbox.id}")
print(f"Session URL: {sandbox.session_url}")
# 2. Set the default model
sandbox.exec('openclaw config set agents.defaults.model.primary "openai/gpt-5.2"')
# 3. Install the WhatsApp plugin
sandbox.exec("openclaw plugins install @openclaw/whatsapp")
gateway_token = os.environ.get("OPENCLAW_APP_TOKEN", str(uuid.uuid4()))
gateway_port = int(os.environ.get("GATEWAY_PORT", "18789"))
sandbox.exec("openclaw config set gateway.mode local")
# 4. Configure the WhatsApp channel (dedicated-number allowlist)
owner_number = "+15551234567" # replace with the phone number allowed to DM the bot
allow_json = f'["{owner_number}"]'
sandbox.exec(
" && ".join(
[
"openclaw config set channels.whatsapp.dmPolicy allowlist",
f"openclaw config set channels.whatsapp.allowFrom '{allow_json}' --strict-json",
"openclaw config set channels.whatsapp.groupPolicy disabled",
]
)
)
# 5. Link WhatsApp in a PTY terminal so the QR renders correctly
print("\nScan the QR code below with your WhatsApp app:\n")
login = sandbox.terminal.create(
SandboxTerminalCreateParams(
command="bash",
args=["-lc", "openclaw channels login --channel whatsapp"],
rows=40,
cols=120,
)
)
login_connection = login.attach()
login_decoder = codecs.getincrementaldecoder("utf-8")()
for event in login_connection.events():
if event.type == "output":
print(login_decoder.decode(event.raw), end="")
continue
break
print(login_decoder.decode(b"", final=True), end="")
login_connection.close()
print("\nWhatsApp linked successfully.")
# 6. Start the gateway
gateway = sandbox.processes.start(
f'openclaw gateway --bind lan --port {gateway_port} --auth token --token "{gateway_token}"',
env={"OPENAI_API_KEY": os.environ["OPENAI_API_KEY"]},
)
# 7. Wait for the gateway to become ready
for _ in range(15):
check = sandbox.exec("openclaw gateway status --require-rpc")
if check.exit_code == 0:
break
time.sleep(2)
print("Gateway started. You can now message your agent on WhatsApp.")
```
## Verify the Setup
After linking, verify everything is running from the sandbox terminal or via the SDK:
```typescript Node.js theme={null}
const status = await sandbox.exec("openclaw gateway status --require-rpc");
console.log(status.stdout);
const channels = await sandbox.exec("openclaw channels list");
console.log(channels.stdout);
const probe = await sandbox.exec("openclaw channels status --probe");
console.log(probe.stdout);
const deep = await sandbox.exec("openclaw status --deep");
console.log(deep.stdout);
```
```python Python theme={null}
status = sandbox.exec("openclaw gateway status --require-rpc")
print(status.stdout)
channels = sandbox.exec("openclaw channels list")
print(channels.stdout)
probe = sandbox.exec("openclaw channels status --probe")
print(probe.stdout)
deep = sandbox.exec("openclaw status --deep")
print(deep.stdout)
```
Send a WhatsApp message from the allowed phone number to the linked WhatsApp
number. With `dmPolicy=allowlist`, no pairing approval step is needed.
## Personal Number Setup
If you want to use your own WhatsApp number instead of a dedicated one, enable
`selfChatMode`. Replace step 4 in the quick start with:
```typescript Node.js theme={null}
const myNumber = "+15551234567"; // your personal WhatsApp number
await sandbox.exec(
[
"openclaw config set channels.whatsapp.dmPolicy allowlist",
`openclaw config set channels.whatsapp.allowFrom '["${myNumber}"]' --strict-json`,
"openclaw config set channels.whatsapp.selfChatMode true --strict-json",
].join(" && ")
);
```
```python Python theme={null}
my_number = "+15551234567" # your personal WhatsApp number
allow_json = f'["{my_number}"]'
sandbox.exec(
" && ".join([
"openclaw config set channels.whatsapp.dmPolicy allowlist",
f"openclaw config set channels.whatsapp.allowFrom '{allow_json}' --strict-json",
"openclaw config set channels.whatsapp.selfChatMode true --strict-json",
])
)
```
With `selfChatMode` enabled, send a message in your own WhatsApp chat
("Message yourself") and OpenClaw responds there.
## Adding Group Chats
To let your agent respond in group chats when mentioned, add this configuration
after the initial setup:
```typescript Node.js theme={null}
const ownerNumber = "+15551234567";
await sandbox.exec(
[
"openclaw config set channels.whatsapp.groupPolicy allowlist",
`openclaw config set channels.whatsapp.groupAllowFrom '["${ownerNumber}"]' --strict-json`,
`openclaw config set channels.whatsapp.groups '{"*":{"requireMention":true}}' --strict-json`,
"openclaw config validate",
].join(" && ")
);
```
```python Python theme={null}
owner_number = "+15551234567"
group_allow_json = f'["{owner_number}"]'
sandbox.exec(
" && ".join([
"openclaw config set channels.whatsapp.groupPolicy allowlist",
f"openclaw config set channels.whatsapp.groupAllowFrom '{group_allow_json}' --strict-json",
'openclaw config set channels.whatsapp.groups \'{"*":{"requireMention":true}}\' --strict-json',
"openclaw config validate",
])
)
```
This means:
* Only allowlisted senders can control the agent in groups.
* The agent only responds when mentioned.
## Pairing Mode
If you want new DM senders to require explicit approval instead of an allowlist,
switch to pairing mode:
```typescript Node.js theme={null}
await sandbox.exec(
[
"openclaw config set channels.whatsapp.dmPolicy pairing",
"openclaw config unset channels.whatsapp.allowFrom",
"openclaw config validate",
].join(" && ")
);
// After an unknown sender messages the bot:
const pairings = await sandbox.exec("openclaw pairing list whatsapp");
console.log(pairings.stdout);
// Approve a specific pairing code
await sandbox.exec("openclaw pairing approve whatsapp CODE_HERE");
```
```python Python theme={null}
sandbox.exec(
" && ".join([
"openclaw config set channels.whatsapp.dmPolicy pairing",
"openclaw config unset channels.whatsapp.allowFrom",
"openclaw config validate",
])
)
# After an unknown sender messages the bot:
pairings = sandbox.exec("openclaw pairing list whatsapp")
print(pairings.stdout)
# Approve a specific pairing code
sandbox.exec("openclaw pairing approve whatsapp CODE_HERE")
```
If you keep `allowFrom` populated alongside `dmPolicy=pairing`, allowlisted
numbers bypass the pairing step.
## Troubleshooting
* **WhatsApp link lost after restart** -- The WhatsApp session lives in
`~/.openclaw`. Use a [volume](/docs/sandboxes/volumes) to persist it across sandbox
restarts.
* **Messages not delivered** -- Verify the channel is connected with
`openclaw channels status --probe`. Check that `allowFrom` includes the
correct E.164 phone number (for example `+15551234567`).
* **QR code expired** -- Run `openclaw channels login --channel whatsapp` again
to generate a new QR code.
* **Gateway bind errors** -- If you only need access from inside the sandbox
(not from the host), set `gateway.bind` to `loopback` instead of `lan`.
## Next Steps
Run and manage processes inside your sandbox.
Save sandbox state and restore it later.
Persist WhatsApp sessions and data across restarts.
See all available base images.
# Sandbox Processes
Source: https://hyperbrowser.ai/docs/sandboxes/processes
## Run a Command
`sandbox.exec(...)` to run a command in the sandbox.
```typescript Node.js theme={null}
const result = await sandbox.exec("node -v");
console.log("Exit code:", result.exitCode);
console.log("Stdout:", result.stdout.trim());
console.log("Stderr:", result.stderr.trim());
```
```python Python theme={null}
result = sandbox.exec("node -v")
print(f"Exit code: {result.exit_code}")
print(f"Stdout: {result.stdout.strip()}")
print(f"Stderr: {result.stderr.strip()}")
```
You can also pass request-scoped options on the same call:
```typescript Node.js theme={null}
const result = await sandbox.exec("pwd && echo $FOO && whoami", {
cwd: "/tmp",
env: {
FOO: "bar",
},
timeoutMs: 5_000,
runAs: "root",
});
```
```python Python theme={null}
result = sandbox.exec(
"pwd && echo $FOO && whoami",
cwd="/tmp",
env={"FOO": "bar"},
timeout_ms=5000,
run_as="root",
)
```
## Start a Background Process
Use `sandbox.processes.start(...)` when you need to keep a process running:
```typescript Node.js theme={null}
const process = await sandbox.processes.start(
"read line; echo stdout:$line; echo stderr:$line 1>&2",
{
runAs: "root",
}
);
await process.writeStdin({
data: "hello\n",
eof: true,
});
const result = await process.wait();
console.log(result.stdout);
console.log(result.stderr);
```
```python Python theme={null}
process = sandbox.processes.start(
"read line; echo stdout:$line; echo stderr:$line 1>&2",
run_as="root",
)
process.write_stdin(data="hello\n", eof=True)
result = process.wait()
print(result.stdout)
print(result.stderr)
```
## Get, List, and Signal Processes
Inspect running processes or control them later by ID:
```typescript Node.js theme={null}
const process = await sandbox.processes.start("sleep 30", {
runAs: "root",
});
const fetched = await sandbox.getProcess(process.id);
const listing = await sandbox.processes.list({
status: "running",
limit: 20,
});
await fetched.signal("TERM");
const result = await fetched.wait();
console.log(listing.data.map((entry) => entry.id));
console.log(result.status);
```
```python Python theme={null}
process = sandbox.processes.start("sleep 30", run_as="root")
fetched = sandbox.get_process(process.id)
listing = sandbox.processes.list(status="running", limit=20)
fetched.signal("TERM")
result = fetched.wait()
print([entry.id for entry in listing.data])
print(result.status)
```
`result()` is a convenience alias for `wait()` on a process handle.
## Stream Process Output
Stream stdout and stderr as server-sent events:
```typescript Node.js theme={null}
const process = await sandbox.processes.start("echo stream-out; echo stream-err 1>&2");
for await (const event of process.stream()) {
if (event.type === "exit") {
console.log("Exited with:", event.result.exitCode);
break;
}
console.log(event.type, event.data);
}
```
```python Python theme={null}
process = sandbox.processes.start("echo stream-out; echo stream-err 1>&2")
for event in process.stream():
if event.type == "exit":
print(f"Exited with: {event.result.exit_code}")
break
print(event.type, event.data)
```
## Common Parameters
Command to run inside the sandbox.
Working directory for the process.
Environment variables to inject into the process.
Maximum runtime in milliseconds.
Run the command as a specific sandbox user, for example `root`.
## Process Status Values
* `queued`
* `running`
* `exited`
* `failed`
* `killed`
* `timed_out`
# Sandbox Runtime URLs
Source: https://hyperbrowser.ai/docs/sandboxes/runtime
Understand runtime URLs, exposed ports, and authenticated browser access
Sandboxes have real URL endpoints when you start it. No extra configuration required. Pick your favorite framework and get started.
Sandboxes have two types of URLs:
* `runtime` - This is the URL you use to access the sandbox's runtime API. It is reserved on port 4001.
* `custom` - This is any other port you wish to expose within the sandbox that is not reserved. You can set up any process you want on the sandbox on the specified port.
## Runtime URL vs Custom Port URL
Every running sandbox includes a runtime base URL for API calls.
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
});
console.log("Runtime URL:", sandbox.runtime.baseUrl);
// The SDK sends this to sandbox.runtime.baseUrl + /exec for you.
const result = await sandbox.exec(`node -e 'console.log("hello from runtime")'`);
console.log(result.stdout.trim());
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "python",
}
)
print(f"Runtime URL: {sandbox.runtime.base_url}")
# The SDK sends this to sandbox.runtime.base_url + /exec for you.
result = sandbox.exec("""python3 -c 'print("hello from runtime")'""")
print(result.stdout.strip())
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
)
)
print(f"Runtime URL: {sandbox.runtime.base_url}")
# The SDK sends this to sandbox.runtime.base_url + /exec for you.
result = sandbox.exec("""python3 -c 'print("hello from runtime")'""")
print(result.stdout.strip())
```
```bash cURL theme={null}
create_response=$(
curl -sS https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "node"
}'
)
runtime_base_url=$(jq -r '.runtime.baseUrl' <<<"$create_response")
runtime_token=$(jq -r '.token' <<<"$create_response")
echo "Runtime URL: $runtime_base_url"
curl -sS \
"$runtime_base_url/exec" \
-H "Authorization: Bearer $runtime_token" \
-H "Content-Type: application/json" \
-d '{"command":"printf \"%s\\n\" \"hello from runtime\""}' \
| jq -r '.result.stdout'
```
The SDK handles the sandbox runtime layer for you when you use the runtime functions. See [Node SDK
sandboxes](/docs/sdks/node#sandboxes) and [Python SDK sandboxes](/docs/sdks/python#sandboxes).
To make a service reachable, start it inside the sandbox and expose its port.
For example:
* `runtime.host` looks like `https://`
* `runtime.baseUrl` looks like `https:///sandbox/`
* exposed port `3000` uses a separate, sandbox-specific URL
The SDK can derive that URL locally:
```typescript Node.js theme={null}
sandbox.getExposedUrl(3000);
```
```python Python theme={null}
sandbox.get_exposed_url(3000)
```
## Example: Expose a Public HTTP Server
The example below provides a public URL which you can use for running your own services inside the sandbox.
For sensitive workloads, treat this as any other publicly accessible endpoint and authenticate your requests properly.
Start a simple HTTP server inside the sandbox, expose port `3000`, and call it from outside the sandbox.
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
});
const server = await sandbox.processes.start(`node -e "const http = require('http');
http
.createServer((_, res) => {
res.writeHead(200, { 'content-type': 'text/plain' });
res.end('hello from sandbox');
})
.listen(3000, '0.0.0.0')"`);
await new Promise((resolve) => setTimeout(resolve, 1000));
const exposure = await sandbox.expose({
port: 3000,
auth: false,
});
console.log("Service URL:", exposure.url);
const response = await fetch(exposure.url);
console.log(await response.text());
```
```python Python 1.0+ theme={null}
import time
from urllib.request import urlopen
sandbox = client.sandboxes.create(
{
"image_name": "python",
}
)
server = sandbox.processes.start(
'''python3 -c "from http.server import BaseHTTPRequestHandler, HTTPServer
class Handler(BaseHTTPRequestHandler):
def do_GET(self):
body = b'hello from sandbox'
self.send_response(200)
self.send_header('Content-Type', 'text/plain')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
HTTPServer(('0.0.0.0', 3000), Handler).serve_forever()"'''
)
time.sleep(1)
exposure = sandbox.expose(
{
"port": 3000,
"auth": False,
}
)
print(f"Service URL: {exposure.url}")
with urlopen(exposure.url) as response:
print(response.read().decode())
```
```python Python (legacy) theme={null}
import time
from urllib.request import urlopen
from hyperbrowser.models import CreateSandboxParams, SandboxExposeParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
)
)
server = sandbox.processes.start(
'''python3 -c "from http.server import BaseHTTPRequestHandler, HTTPServer
class Handler(BaseHTTPRequestHandler):
def do_GET(self):
body = b'hello from sandbox'
self.send_response(200)
self.send_header('Content-Type', 'text/plain')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
HTTPServer(('0.0.0.0', 3000), Handler).serve_forever()"'''
)
time.sleep(1)
exposure = sandbox.expose(
SandboxExposeParams(
port=3000,
auth=False,
)
)
print(f"Service URL: {exposure.url}")
with urlopen(exposure.url) as response:
print(response.read().decode())
```
## Auth Enabled (`auth: true`)
When `auth` is `true`, the exposed custom URL now requires the auth token to be passed on each request.
Call `exposure.url` programmatically and pass the sandbox bearer token in
the `Authorization` header.
Open `exposure.browserUrl` in a browser when you need a user-facing flow
for an auth-protected endpoint.
`browserUrl` performs auth bootstrap first and then redirects to the path in
`next`.
Use `auth: false` when you want open browser access and keep auth/authorization
logic inside your own application.
### Backend Workflow (Bearer Token)
This flow is best for backend automation and API calls. Call the protected exposed URL with the sandbox bearer token.
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
});
await sandbox.processes.start(`node -e "const http = require('http');
http
.createServer((req, res) => {
if (req.url === '/api/status') {
res.writeHead(200, { 'content-type': 'application/json' });
res.end(JSON.stringify({ ok: true }));
return;
}
res.writeHead(404);
res.end('not found');
})
.listen(3000, '0.0.0.0')"`);
await new Promise((resolve) => setTimeout(resolve, 1000));
const exposure = await sandbox.expose({
port: 3000,
auth: true,
});
const detail = await sandbox.info();
const apiUrl = new URL("/api/status", exposure.url);
const response = await fetch(apiUrl, {
headers: {
Authorization: `Bearer ${detail.token}`,
},
});
console.log(await response.text());
```
```python Python 1.0+ theme={null}
import time
from urllib.parse import urljoin
from urllib.request import Request, urlopen
sandbox = client.sandboxes.create(
{
"image_name": "python",
}
)
sandbox.processes.start(
'''python3 -c "from http.server import BaseHTTPRequestHandler, HTTPServer
class Handler(BaseHTTPRequestHandler):
def do_GET(self):
if self.path == '/api/status':
body = b'{"ok": true}'
self.send_response(200)
self.send_header('Content-Type', 'application/json')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
return
self.send_response(404)
self.end_headers()
def log_message(self, format, *args):
pass
HTTPServer(('0.0.0.0', 3000), Handler).serve_forever()"'''
)
time.sleep(1)
exposure = sandbox.expose(
{
"port": 3000,
"auth": True,
}
)
detail = sandbox.info()
api_url = urljoin(exposure.url, "/api/status")
request = Request(
api_url,
headers={"Authorization": f"Bearer {detail.token}"},
)
with urlopen(request) as response:
print(response.read().decode())
```
```python Python (legacy) theme={null}
import time
from urllib.parse import urljoin
from urllib.request import Request, urlopen
from hyperbrowser.models import CreateSandboxParams, SandboxExposeParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="python",
)
)
sandbox.processes.start(
'''python3 -c "from http.server import BaseHTTPRequestHandler, HTTPServer
class Handler(BaseHTTPRequestHandler):
def do_GET(self):
if self.path == '/api/status':
body = b'{"ok": true}'
self.send_response(200)
self.send_header('Content-Type', 'application/json')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
return
self.send_response(404)
self.end_headers()
def log_message(self, format, *args):
pass
HTTPServer(('0.0.0.0', 3000), Handler).serve_forever()"'''
)
time.sleep(1)
exposure = sandbox.expose(
SandboxExposeParams(
port=3000,
auth=True,
)
)
detail = sandbox.info()
api_url = urljoin(exposure.url, "/api/status")
request = Request(
api_url,
headers={"Authorization": f"Bearer {detail.token}"},
)
with urlopen(request) as response:
print(response.read().decode())
```
```bash cURL theme={null}
curl "$EXPOSED_URL/api/status" \
-H "Authorization: Bearer $SANDBOX_TOKEN"
```
### Browser Workflow (Preview URL Redirect)
Use the auth-enabled endpoint to generate a protected preview URL for browser access.
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
});
await sandbox.processes.start(`node -e "const http = require('http');
http
.createServer((req, res) => {
if (req.url === '/preview') {
res.writeHead(200, { 'content-type': 'text/html' });
res.end('
Sandbox Preview
Protected by sandbox auth.
');
return;
}
res.writeHead(404);
res.end('not found');
})
.listen(3000, '0.0.0.0')"`);
await new Promise((resolve) => setTimeout(resolve, 1000));
const exposure = await sandbox.expose({
port: 3000,
auth: true,
});
const browserUrl = new URL(exposure.browserUrl);
browserUrl.searchParams.set("next", "/preview");
console.log("Open preview URL:", browserUrl.toString());
```
```python Python 1.0+ theme={null}
import time
from urllib.parse import parse_qsl, urlencode, urlparse, urlunparse
sandbox = client.sandboxes.create(
{
"image_name": "python",
}
)
sandbox.processes.start(
'''python3 -c "from http.server import BaseHTTPRequestHandler, HTTPServer
class Handler(BaseHTTPRequestHandler):
def do_GET(self):
if self.path == '/preview':
body = b'
'
self.send_response(200)
self.send_header('Content-Type', 'text/html')
self.send_header('Content-Length', str(len(body)))
self.end_headers()
self.wfile.write(body)
return
self.send_response(404)
self.end_headers()
def log_message(self, format, *args):
pass
HTTPServer(('0.0.0.0', 3000), Handler).serve_forever()"'''
)
time.sleep(1)
exposure = sandbox.expose(
SandboxExposeParams(
port=3000,
auth=True,
)
)
parsed = urlparse(exposure.browser_url)
query = dict(parse_qsl(parsed.query))
query["next"] = "/preview"
browser_url = urlunparse(parsed._replace(query=urlencode(query)))
print(f"Open preview URL: {browser_url}")
```
`browserUrl` is not the raw port API URL. It is a bootstrap URL that sets the
auth cookie and then redirects to `next`. Use `url` for programmatic clients,
and use `browserUrl` when you want to land a browser on a protected page.
# Sandbox Snapshots
Source: https://hyperbrowser.ai/docs/sandboxes/snapshots
Create memory snapshots and start sandboxes from snapshots
Memory snapshots let you capture a running sandbox and start new sandboxes from that saved state.
When you take a snapshot, you are freezing the sandbox's memory, processes, and filesystem at that point in time.
Starting a sandbox from a snapshot is instant and you can start a new sandbox from the same snapshot multiple times.
## Create a Memory Snapshot
Create a snapshot from a running sandbox.
```typescript Node.js theme={null}
const snapshot = await sandbox.createMemorySnapshot({
snapshotName: "node-after-setup",
});
console.log("Snapshot name:", snapshot.snapshotName);
console.log("Snapshot ID:", snapshot.snapshotId);
console.log("Status:", snapshot.status);
```
```python Python 1.0+ theme={null}
snapshot = sandbox.create_memory_snapshot(
{
"snapshot_name": "python-after-setup",
}
)
print(f"Snapshot name: {snapshot.snapshot_name}")
print(f"Snapshot ID: {snapshot.snapshot_id}")
print(f"Status: {snapshot.status}")
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxMemorySnapshotParams
snapshot = sandbox.create_memory_snapshot(
SandboxMemorySnapshotParams(
snapshot_name="python-after-setup",
)
)
print(f"Snapshot name: {snapshot.snapshot_name}")
print(f"Snapshot ID: {snapshot.snapshot_id}")
print(f"Status: {snapshot.status}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox/SANDBOX_ID/snapshot \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"snapshotName": "node-after-setup"
}'
```
## Start a Sandbox From a Snapshot
Use the snapshot name to start a new sandbox:
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
snapshotName: "node-after-setup",
snapshotId: "snapshot-id", // if omitted uses the latest snapshot
});
console.log(sandbox.id);
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"snapshot_name": "python-after-setup",
"snapshot_id": "snapshot-id", # if omitted uses the latest snapshot
}
)
print(sandbox.id)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
snapshot_name="python-after-setup",
snapshot_id="snapshot-id", # if omitted uses the latest snapshot
)
)
print(sandbox.id)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"snapshotName": "node-after-setup",
"snapshotId": "snapshot-id"
}'
```
## List Images and Snapshots
List the available sandbox images and snapshots:
```typescript Node.js theme={null}
const { images } = await client.sandboxes.listImages();
const { snapshots } = await client.sandboxes.listSnapshots({
status: "created",
limit: 100,
});
console.log(images.map((image) => image.imageName));
console.log(snapshots.map((snapshot) => snapshot.snapshotName));
```
```python Python 1.0+ theme={null}
images = client.sandboxes.list_images()
snapshots = client.sandboxes.list_snapshots(
{
"status": "created",
"limit": 100,
}
)
print([image.image_name for image in images.images])
print([snapshot.snapshot_name for snapshot in snapshots.snapshots])
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxSnapshotListParams
images = client.sandboxes.list_images()
snapshots = client.sandboxes.list_snapshots(
SandboxSnapshotListParams(
status="created",
limit=100,
)
)
print([image.image_name for image in images.images])
print([snapshot.snapshot_name for snapshot in snapshots.snapshots])
```
```bash cURL theme={null}
curl -X GET https://api.hyperbrowser.ai/api/images \
-H "x-api-key: YOUR_API_KEY"
curl -X GET "https://api.hyperbrowser.ai/api/snapshots?status=created&limit=100" \
-H "x-api-key: YOUR_API_KEY"
```
## Snapshot Response
The snapshot create API returns metadata about both the snapshot and its backing image:
Snapshot name.
Unique snapshot identifier.
The team that owns the snapshot.
Snapshot creation status.
Backing image name for the snapshot. What the snapshot was created from.
# Sandbox Terminal
Source: https://hyperbrowser.ai/docs/sandboxes/terminal
Open interactive terminal sessions inside a sandbox
Sandboxes have PTY session support for interactive terminal sessions.
SDK is available for Node.js and Python to build your own terminal client.
This page documents how to build a terminal client, to start a terminal session, see [CLI Terminal](/docs/sandboxes/cli#open-an-interactive-terminal).
## Create and Attach to a Terminal
Create a terminal and attach to its websocket stream:
```typescript Node.js theme={null}
const terminal = await sandbox.terminal.create({
command: "bash",
args: ["-l"],
rows: 24,
cols: 80,
});
const connection = await terminal.attach();
for await (const event of connection.events()) {
if (event.type === "output") {
process.stdout.write(event.data);
continue;
}
console.log("Exit code:", event.status.exitCode);
break;
}
```
```python Python 1.0+ theme={null}
terminal = sandbox.terminal.create(
{
"command": "bash",
"args": ["-l"],
"rows": 24,
"cols": 80,
}
)
connection = terminal.attach()
for event in connection.events():
if event.type == "output":
print(event.data, end="")
continue
print(f"Exit code: {event.status.exit_code}")
break
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxTerminalCreateParams
terminal = sandbox.terminal.create(
SandboxTerminalCreateParams(
command="bash",
args=["-l"],
rows=24,
cols=80,
)
)
connection = terminal.attach()
for event in connection.events():
if event.type == "output":
print(event.data, end="")
continue
print(f"Exit code: {event.status.exit_code}")
break
```
## Write Input and Resize the PTY
Send input through the attached connection and resize the terminal:
```typescript Node.js theme={null}
const terminal = await sandbox.terminal.create({
command: "bash",
args: ["-l"],
});
const connection = await terminal.attach();
await connection.resize(32, 110);
await connection.write("pwd\n");
await connection.write("echo terminal-ok\n");
await connection.write("exit\n");
await connection.close();
```
```python Python 1.0+ theme={null}
terminal = sandbox.terminal.create(
{
"command": "bash",
"args": ["-l"],
}
)
connection = terminal.attach()
connection.resize(32, 110)
connection.write("pwd\n")
connection.write("echo terminal-ok\n")
connection.write("exit\n")
connection.close()
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxTerminalCreateParams
terminal = sandbox.terminal.create(
SandboxTerminalCreateParams(
command="bash",
args=["-l"],
)
)
connection = terminal.attach()
connection.resize(32, 110)
connection.write("pwd\n")
connection.write("echo terminal-ok\n")
connection.write("exit\n")
connection.close()
```
## Get, Refresh, and Wait
Fetch an existing terminal or wait for it to complete:
```typescript Node.js theme={null}
const terminal = await sandbox.terminal.create({
command: "bash",
args: ["-lc", "echo hello-from-terminal"],
});
const fetched = await sandbox.terminal.get(terminal.id, true);
console.log(fetched.current.output?.map((chunk) => chunk.data).join(""));
const status = await terminal.wait({
timeoutMs: 2_000,
includeOutput: true,
});
console.log(status.running, status.exitCode);
```
```python Python 1.0+ theme={null}
terminal = sandbox.terminal.create(
{
"command": "bash",
"args": ["-lc", "echo hello-from-terminal"],
}
)
fetched = sandbox.terminal.get(terminal.id, include_output=True)
print("".join(chunk.data for chunk in fetched.current.output or []))
status = terminal.wait(
timeout_ms=2000,
include_output=True,
)
print(status.running, status.exit_code)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxTerminalCreateParams
terminal = sandbox.terminal.create(
SandboxTerminalCreateParams(
command="bash",
args=["-lc", "echo hello-from-terminal"],
)
)
fetched = sandbox.terminal.get(terminal.id, include_output=True)
print("".join(chunk.data for chunk in fetched.current.output or []))
status = terminal.wait(
timeout_ms=2000,
include_output=True,
)
print(status.running, status.exit_code)
```
## Signal and Kill
You can signal or kill a terminal-backed process:
```typescript Node.js theme={null}
const terminal = await sandbox.terminal.create({
command: "bash",
args: ["-lc", "sleep 30"],
});
await terminal.signal("TERM");
const status = await terminal.wait({ timeoutMs: 5_000 });
console.log("Still running:", status.running);
console.log("Exit code:", status.exitCode);
```
```python Python 1.0+ theme={null}
terminal = sandbox.terminal.create(
{
"command": "bash",
"args": ["-lc", "sleep 30"],
}
)
terminal.signal("TERM")
status = terminal.wait(timeout_ms=5000)
print("Still running:", status.running)
print("Exit code:", status.exit_code)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxTerminalCreateParams
terminal = sandbox.terminal.create(
SandboxTerminalCreateParams(
command="bash",
args=["-lc", "sleep 30"],
)
)
terminal.signal("TERM")
status = terminal.wait(timeout_ms=5000)
print("Still running:", status.running)
print("Exit code:", status.exit_code)
```
`status.running` tells you whether the terminal session is still alive after the signal.
After `signal("TERM")` and a successful `wait(...)`, you would usually expect
`status.running` to be `false`.
## Common Terminal Parameters
Command to launch inside the terminal session.
Optional command arguments.
Working directory for the terminal process.
Environment variables to set for the terminal process.
Initial terminal row count.
Initial terminal column count.
Optional PTY runtime limit in milliseconds.
`sandbox.pty` is an alias for `sandbox.terminal`.
## Create Interactive Terminal
Run an interactive sandbox shell from your backend process and bridge local terminal input/output to the sandbox PTY stream.
```typescript Node.js theme={null}
const terminal = await sandbox.terminal.create({
command: "bash",
args: ["-l"],
rows: process.stdout.rows ?? 24,
cols: process.stdout.columns ?? 80,
});
console.log("Sandbox PTY ID:", terminal.id);
const connection = await terminal.attach();
const onStdinData = (chunk: Buffer) => {
void connection.write(chunk);
};
const onResize = () => {
void connection.resize(process.stdout.rows ?? 24, process.stdout.columns ?? 80);
};
if (process.stdin.isTTY) {
process.stdin.setRawMode(true);
}
process.stdin.resume();
process.stdin.on("data", onStdinData);
if (process.stdout.isTTY) {
process.stdout.on("resize", onResize);
}
try {
for await (const event of connection.events()) {
if (event.type === "output") {
process.stdout.write(event.data);
continue;
}
process.stdout.write(`\n[remote exited: ${event.status.exitCode ?? "unknown"}]\n`);
break;
}
} finally {
process.stdin.off("data", onStdinData);
process.stdin.pause();
process.stdout.off("resize", onResize);
if (process.stdin.isTTY) {
process.stdin.setRawMode(false);
}
await connection.close();
await sandbox.stop();
}
```
```python Python 1.0+ theme={null}
import os
import sys
import termios
import threading
import tty
terminal = sandbox.terminal.create(
{
"command": "bash",
"args": ["-l"],
"rows": os.get_terminal_size().lines if sys.stdout.isatty() else 24,
"cols": os.get_terminal_size().columns if sys.stdout.isatty() else 80,
}
)
print(f"Sandbox PTY ID: {terminal.id}")
connection = terminal.attach()
stop_input = threading.Event()
def stdin_pump():
while not stop_input.is_set():
chunk = os.read(sys.stdin.fileno(), 1024)
if not chunk:
break
connection.write(chunk)
input_thread = threading.Thread(target=stdin_pump, daemon=True)
input_thread.start()
original_tty = None
if sys.stdin.isatty():
original_tty = termios.tcgetattr(sys.stdin.fileno())
tty.setraw(sys.stdin.fileno())
try:
for event in connection.events():
if event.type == "output":
sys.stdout.write(event.data)
sys.stdout.flush()
continue
print(f"\n[remote exited: {event.status.exit_code}]")
break
finally:
stop_input.set()
connection.close()
if original_tty is not None:
termios.tcsetattr(sys.stdin.fileno(), termios.TCSADRAIN, original_tty)
sandbox.stop()
```
```python Python (legacy) theme={null}
import os
import sys
import termios
import threading
import tty
from hyperbrowser.models import SandboxTerminalCreateParams
terminal = sandbox.terminal.create(
SandboxTerminalCreateParams(
command="bash",
args=["-l"],
rows=os.get_terminal_size().lines if sys.stdout.isatty() else 24,
cols=os.get_terminal_size().columns if sys.stdout.isatty() else 80,
)
)
print(f"Sandbox PTY ID: {terminal.id}")
connection = terminal.attach()
stop_input = threading.Event()
def stdin_pump():
while not stop_input.is_set():
chunk = os.read(sys.stdin.fileno(), 1024)
if not chunk:
break
connection.write(chunk)
input_thread = threading.Thread(target=stdin_pump, daemon=True)
input_thread.start()
original_tty = None
if sys.stdin.isatty():
original_tty = termios.tcgetattr(sys.stdin.fileno())
tty.setraw(sys.stdin.fileno())
try:
for event in connection.events():
if event.type == "output":
sys.stdout.write(event.data)
sys.stdout.flush()
continue
print(f"\n[remote exited: {event.status.exit_code}]")
break
finally:
stop_input.set()
connection.close()
if original_tty is not None:
termios.tcsetattr(sys.stdin.fileno(), termios.TCSADRAIN, original_tty)
sandbox.stop()
```
The Python raw-mode bridge uses `termios` and `tty`, so it is intended for POSIX terminals (Linux/macOS).
# Overview
Source: https://hyperbrowser.ai/docs/sandboxes/volumes
Persistent storage that can be mounted into sandboxes
Volumes are persistent, team scoped filesystems that you can attach to a sandbox at launch.
You can attach a volume to multiple sandboxes, and a volume is synced across writes.
A volume persists even after all sandboxes are stopped, so you can completely isolate your data needs from your sandbox compute.
## Quick Start
1. Create a volume.
2. Mount it when creating a sandbox.
3. Read and write files at the mount path.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
// 1) Create volume
const volume = await client.volumes.create({ name: "my-workspace" });
// 2) Create sandbox with mounted volume
const sandbox = await client.sandboxes.create({
imageName: "node",
mounts: {
"/mnt/workspace": {
id: volume.id,
type: "rw",
},
},
});
// 3) Write/read through mounted path
await sandbox.files.writeText("/mnt/workspace/hello.txt", "hello from volume");
const text = await sandbox.files.readText("/mnt/workspace/hello.txt");
console.log(text);
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(api_key="YOUR_API_KEY")
# 1) Create volume
volume = client.volumes.create({"name": "my-workspace"})
# 2) Create sandbox with mounted volume
sandbox = client.sandboxes.create(
{
"image_name": "node",
"mounts": {
"/mnt/workspace": {
"id": volume.id,
"type": "rw",
}
},
}
)
# 3) Write/read through mounted path
sandbox.files.write_text("/mnt/workspace/hello.txt", "hello from volume")
text = sandbox.files.read_text("/mnt/workspace/hello.txt")
print(text)
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import (
CreateSandboxParams,
CreateVolumeParams,
SandboxVolumeMount,
)
client = Hyperbrowser(api_key="YOUR_API_KEY")
# 1) Create volume
volume = client.volumes.create(CreateVolumeParams(name="my-workspace"))
# 2) Create sandbox with mounted volume
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="node",
mounts={
"/mnt/workspace": SandboxVolumeMount(
id=volume.id,
type="rw",
)
},
)
)
# 3) Write/read through mounted path
sandbox.files.write_text("/mnt/workspace/hello.txt", "hello from volume")
text = sandbox.files.read_text("/mnt/workspace/hello.txt")
print(text)
```
```bash CLI theme={null}
# 1) Create volume
volume_id=$(hx --json vm volumes create my-workspace | jq -r '.id')
# 2) Create sandbox with mounted volume
sandbox_id=$(hx --json vm create node --mount "$volume_id:/mnt/workspace" | jq -r '.data.id')
# 3) Write/read through mounted path
hx file write "$sandbox_id" /mnt/workspace/hello.txt --data 'hello from volume'
hx file cat "$sandbox_id" /mnt/workspace/hello.txt
```
```bash cURL theme={null}
# 1) Create volume
volume_response=$(curl -sS -X POST https://api.hyperbrowser.ai/api/volume \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-H "Content-Type: application/json" \
-d '{"name":"my-workspace"}')
volume_id=$(jq -r '.id' <<<"$volume_response")
# 2) Create sandbox with mounted volume
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-H "Content-Type: application/json" \
-d "{
\"imageName\": \"node\",
\"mounts\": {
\"/mnt/workspace\": {
\"id\": \"$volume_id\",
\"type\": \"rw\"
}
}
}"
```
## Why Volumes
Use volumes when you need data to outlive a sandbox and be reused later:
* Persistent workspaces for agents.
* Reusable caches across runs.
* Shared read only data mounted into multiple sandboxes over time.
## Volume Lifecycle
* Create volumes via SDKs, volume endpoints, or `hx vm volumes`.
* Mount volumes during sandbox creation
* Access files through the mounted path with normal sandbox filesystem operations.
* Delete unused volumes with `client.volumes.delete`, `DELETE /api/volume/:key`, or `hx vm volumes delete`. Active mounts and ambiguous names return `409`.
## In This Section
* [Managing Volumes](/docs/sandboxes/volumes/managing)
* [Mounting Volumes](/docs/sandboxes/volumes/mounting)
* [Read and Write Mounted Volume Data](/docs/sandboxes/volumes/data-access)
# Read and Write Mounted Volume Data
Source: https://hyperbrowser.ai/docs/sandboxes/volumes/data-access
Work with volume files through mounted paths
Once a volume is mounted, use the standard sandbox filesystem APIs against the mount path.
## Read and Write
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.get("sandbox-id");
await sandbox.files.writeText("/mnt/workspace/notes.txt", "persisted data");
const text = await sandbox.files.readText("/mnt/workspace/notes.txt");
console.log(text);
```
```python Python theme={null}
sandbox = client.sandboxes.get("sandbox-id")
sandbox.files.write_text("/mnt/workspace/notes.txt", "persisted data")
text = sandbox.files.read_text("/mnt/workspace/notes.txt")
print(text)
```
```bash CLI theme={null}
hx file write /mnt/workspace/notes.txt --data 'persisted data'
hx file cat /mnt/workspace/notes.txt
```
## List and Inspect
```typescript Node.js theme={null}
const entries = await sandbox.files.list("/mnt/workspace", { depth: 1 });
const info = await sandbox.files.getInfo("/mnt/workspace/notes.txt");
console.log(entries.map((entry) => entry.path));
console.log(info.size, info.permissions);
```
```python Python theme={null}
entries = sandbox.files.list("/mnt/workspace", depth=1)
info = sandbox.files.get_info("/mnt/workspace/notes.txt")
print([entry.path for entry in entries])
print(info.size, info.permissions)
```
```bash CLI theme={null}
hx file ls /mnt/workspace --depth 1
hx file stat /mnt/workspace/notes.txt
```
## Upload and Download
You can also use standard filesystem transfer operations against mounted paths.
```bash theme={null}
# Upload local file into mounted volume path
hx vm cp ./local.txt :/mnt/workspace/local.txt
# Download from mounted volume path
hx vm cp :/mnt/workspace/local.txt ./downloaded.txt
```
## Access Mode
* `rw` mounts allow writes.
* `ro` mounts reject write operations.
# Managing Volumes
Source: https://hyperbrowser.ai/docs/sandboxes/volumes/managing
## Create a Volume
```typescript Node.js theme={null}
const created = await client.volumes.create({ name: "my-workspace" });
console.log(created.id, created.name);
```
```python Python 1.0+ theme={null}
created = client.volumes.create({"name": "my-workspace"})
print(created.id, created.name)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateVolumeParams
created = client.volumes.create(CreateVolumeParams(name="my-workspace"))
print(created.id, created.name)
```
```bash CLI theme={null}
hx vm volumes create my-workspace
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/volume \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "my-workspace"
}'
```
### Create Response
```json theme={null}
{
"id": "550e8400-e29b-41d4-a716-446655440000",
"name": "my-workspace",
"size": 0,
"transferAmount": 0
}
```
## List Volumes
```typescript Node.js theme={null}
const listed = await client.volumes.list();
console.log(listed.volumes.length);
```
```python Python theme={null}
listed = client.volumes.list()
print(len(listed.volumes))
```
```bash CLI theme={null}
hx vm volumes list
# alias: hx vm volumes ls
```
```bash cURL theme={null}
curl -X GET https://api.hyperbrowser.ai/api/volume \
-H "x-api-key: YOUR_API_KEY"
```
### List Response
```json theme={null}
{
"volumes": [
{
"id": "550e8400-e29b-41d4-a716-446655440000",
"name": "my-workspace",
"size": 0,
"transferAmount": 0
}
],
"totalCount": 1,
"page": 1,
"perPage": 20
}
```
## Get a Volume
```typescript Node.js theme={null}
const volume = await client.volumes.get("550e8400-e29b-41d4-a716-446655440000");
console.log(volume.id, volume.name);
```
```python Python theme={null}
volume = client.volumes.get("550e8400-e29b-41d4-a716-446655440000")
print(volume.id, volume.name)
```
```bash CLI theme={null}
hx vm volumes get
```
```bash cURL theme={null}
curl -X GET https://api.hyperbrowser.ai/api/volume/VOLUME_ID \
-H "x-api-key: YOUR_API_KEY"
```
### Get Response
```json theme={null}
{
"id": "550e8400-e29b-41d4-a716-446655440000",
"name": "my-workspace"
}
```
## Delete a Volume
Delete a volume by ID or name. Name lookup is case-insensitive. If more than one volume shares that name, the request returns `409` and you must pass an id. Active sandboxes that still mount the volume also return `409`.
```typescript Node.js theme={null}
const deleted = await client.volumes.delete("550e8400-e29b-41d4-a716-446655440000");
console.log(deleted.id, deleted.name);
```
```python Python theme={null}
deleted = client.volumes.delete("550e8400-e29b-41d4-a716-446655440000")
print(deleted.id, deleted.name)
```
```bash CLI theme={null}
hx vm volumes delete my-workspace --force
# alias: hx vm volumes rm my-workspace --force
```
```bash cURL theme={null}
curl -X DELETE https://api.hyperbrowser.ai/api/volume/VOLUME_ID_OR_NAME \
-H "x-api-key: YOUR_API_KEY"
```
### Delete Response
```json theme={null}
{
"deleted": true,
"id": "550e8400-e29b-41d4-a716-446655440000",
"name": "my-workspace"
}
```
## Next Step
* Continue to [Mounting Volumes](/docs/sandboxes/volumes/mounting) to attach volumes to a sandbox.
# Mounting Volumes
Source: https://hyperbrowser.ai/docs/sandboxes/volumes/mounting
Attach volumes to sandboxes at launch
Attach existing volumes when you create a sandbox.
## Mount at Sandbox Creation
Use the SDK, CLI or REST API to mount volumes at sandbox creation.
```typescript Node.js theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node",
mounts: {
"/mnt/workspace": {
id: "550e8400-e29b-41d4-a716-446655440000",
type: "rw",
},
"/mnt/cache": {
id: "660e8400-e29b-41d4-a716-446655440000",
type: "ro",
},
},
});
```
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "node",
"mounts": {
"/mnt/workspace": {
"id": "550e8400-e29b-41d4-a716-446655440000",
"type": "rw",
},
"/mnt/cache": {
"id": "660e8400-e29b-41d4-a716-446655440000",
"type": "ro",
},
},
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams, SandboxVolumeMount
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="node",
mounts={
"/mnt/workspace": SandboxVolumeMount(
id="550e8400-e29b-41d4-a716-446655440000",
type="rw",
),
"/mnt/cache": SandboxVolumeMount(
id="660e8400-e29b-41d4-a716-446655440000",
type="ro",
),
},
)
)
```
```bash CLI theme={null}
hx vm create node \
--mount 550e8400-e29b-41d4-a716-446655440000:/mnt/workspace \
--mount 660e8400-e29b-41d4-a716-446655440000:/mnt/cache:ro
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/sandbox \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"imageName": "node",
"mounts": {
"/mnt/workspace": {
"id": "550e8400-e29b-41d4-a716-446655440000",
"type": "rw"
},
"/mnt/cache": {
"id": "660e8400-e29b-41d4-a716-446655440000",
"type": "ro"
}
}
}'
```
Raw `mounts` payload:
```json theme={null}
{
"imageName": "node",
"mounts": {
"/mnt/workspace": {
"id": "550e8400-e29b-41d4-a716-446655440000",
"type": "rw"
},
"/mnt/cache": {
"id": "660e8400-e29b-41d4-a716-446655440000",
"type": "ro"
}
}
}
```
## CLI `--mount` Formats
Use repeated `--mount` flags.
```bash theme={null}
:/abs/path[:ro|rw]
```
Examples:
```bash theme={null}
hx vm create node --mount :/mnt/workspace
hx vm create node --mount :/mnt/cache:ro
```
```bash theme={null}
source=,target=/abs/path[,readonly|mode=ro|mode=rw]
```
Examples:
```bash theme={null}
hx vm create node --mount source=,target=/mnt/workspace
hx vm create node --mount source=,target=/mnt/cache,readonly
```
## Mount Object Fields
Volume ID (UUID).
Access mode: `rw` or `ro`.
## Naming Rules
* Mount path must be absolute and below `/`.
* Mount path cannot be `/`.
* Mount path must already be normalized.
* Mount path cannot target protected system locations such as `/etc`, `/usr`, `/proc`, `/dev`, `/root`, `/run`, or `/sys`.
* Duplicate mount targets are rejected.
* Volume must exist and belong to your team.
## Launch Behavior
At launch, Hyperbrowser resolves each mount to a team-owned volume, then provisions runtime mount instructions with scoped access.
If mount resolution fails, sandbox creation fails.
Mounts are currently configured at sandbox launch. Updating mounts on an
already running sandbox is not supported
## Next Step
* Continue to [Read and Write Mounted Volume Data](/docs/sandboxes/volumes/data-access).
# Introduction
Source: https://hyperbrowser.ai/docs/sdks/introduction
Official SDKs for integrating Hyperbrowser into your applications
## Choose Your SDK
## Quick Start
Get up and running in minutes with our SDK quick start guides:
```typescript Node.js theme={null}
import Hyperbrowser from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
// Create a browser session
const session = await client.sessions.create({
acceptCookies: true,
});
try {
// Connect with Playwright
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
// Navigate and interact
await page.goto("https://example.com");
const pageTitle = await page.title();
console.log(`Page title: ${pageTitle}`);
} catch (err) {
console.error(`Encountered error: ${err}`);
} finally {
// Clean up
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
from dotenv import load_dotenv
import os
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create({"accept_cookies": True})
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
page.goto("https://example.com")
page_title = page.title()
print(f"Page title: {page_title}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
from dotenv import load_dotenv
import os
from hyperbrowser.models import CreateSessionParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
session = client.sessions.create(CreateSessionParams(accept_cookies=True))
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
page.goto("https://example.com")
page_title = page.title()
print(f"Page title: {page_title}")
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
## Need Help?
Browse our comprehensive guides and API reference
Join our Discord community for support and discussions
View source code, report issues, and contribute
# Node SDK
Source: https://hyperbrowser.ai/docs/sdks/node
Complete guide to the Hyperbrowser Node.js SDK
View on GitHub
View on NPM
## Installation
Install the Hyperbrowser SDK via npm or yarn:
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
## Quick Start
Initialize the client with your API key:
```typescript theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
```
### Configuration Options
```typescript theme={null}
interface HyperbrowserConfig {
apiKey?: string; // API key (can also use HYPERBROWSER_API_KEY env var)
baseUrl?: string; // Base API URL (default: "https://api.hyperbrowser.ai")
timeout?: number; // Request timeout in milliseconds (default: 30000)
runtimeProxyOverride?: string; // Optional host override for sandbox runtime traffic
}
```
## TypeScript Support
The SDK is fully typed with TypeScript. Import types from `@hyperbrowser/sdk/types`:
```typescript theme={null}
import {
SessionDetail,
CreateSessionParams,
CreateSandboxParams,
ScrapeJobResponse,
CrawlJobResponse,
ExtractJobResponse,
BrowserUseTaskResponse,
// ... and many more
} from "@hyperbrowser/sdk/types";
```
## Sandboxes
Sandbox APIs work in two layers.
Sandbox control methods use your API key to create, inspect, connect to, and stop sandboxes. When you create or connect to a sandbox, the SDK also retrieves a sandbox-scoped runtime token and returns an authenticated `SandboxHandle`.
Sandbox VM operations run inside that started sandbox and use the runtime token on the handle automatically. The SDK handles runtime token refreshes for you, so once you have a running `SandboxHandle`, you can call files, processes, terminal, networking, and snapshot methods without managing runtime auth yourself.
## Sandbox Control
Use these methods to inspect existing sandboxes and discover reusable sandbox
resources.
### Details
#### List Sandboxes
```typescript theme={null}
const response = await client.sandboxes.list({
status: "active", // optional: sandbox status filter
search: "sdk", // optional: search term
page: 1, // optional: page number
limit: 20, // optional: results per page
});
console.log(response.totalCount);
console.log(response.sandboxes.map((entry) => entry.id));
```
#### List Images
```typescript theme={null}
const { images } = await client.sandboxes.listImages();
console.log(images.map((image) => image.imageName));
```
#### List Snapshots
```typescript theme={null}
const { snapshots } = await client.sandboxes.listSnapshots({
imageName: "node", // optional: only snapshots for this image
status: "created", // optional: snapshot status filter
limit: 10, // optional: max results
});
console.log(snapshots.map((snapshot) => snapshot.snapshotName));
```
#### Get Sandbox Info
```typescript theme={null}
const detail = await sandbox.info();
console.log(detail.runtime.baseUrl);
```
### Lifecycle
Use these methods to create, connect to, and stop sandbox instances.
#### Create Sandbox
Use `client.sandboxes.create(...)` to start a sandbox from an image.
```typescript theme={null}
const sandbox = await client.sandboxes.create({
imageName: "node", // required unless restoring from a snapshot
region: "us-west", // optional: sandbox region
timeoutMinutes: 30, // optional: max sandbox lifetime
enableRecording: true, // optional: record the sandbox
exposedPorts: [{ port: 3000, auth: true }], // optional: pre-expose ports
});
```
#### Start From A Snapshot
Start a new sandbox from a memory checkpoint.
```typescript theme={null}
const sandbox = await client.sandboxes.create({
snapshotName: "node-after-setup", // required: snapshot name
snapshotId: "snapshot-id", // optional: pin a specific snapshot version
});
```
#### Connect To A Running Sandbox
Use `connect(...)` to re-authenticate a running sandbox, refresh its runtime token, and return a `SandboxHandle` for runtime operations. If you already have a handle, call `sandbox.connect()` to re-authenticate it in place.
```typescript theme={null}
const sandbox = await client.sandboxes.connect(
"sandbox-id" // running sandbox ID
);
await sandbox.connect(); // re-authenticate an existing handle in place
```
#### Stop Sandbox
```typescript theme={null}
await sandbox.stop(); // stop the sandbox when finished
```
## Sandbox VM Operations
These methods operate on the sandbox VM itself when the sandbox is running.
### Networking
#### Expose Port
```typescript theme={null}
const exposure = await sandbox.expose({
port: 3000, // required: port inside the sandbox
auth: true, // optional: require the sandbox bearer token
});
console.log(exposure.url);
```
#### Unexpose Port
```typescript theme={null}
await sandbox.unexpose(
3000 // exposed port to remove
);
```
#### Get Exposed URL
```typescript theme={null}
const url = sandbox.getExposedUrl(
3000 // exposed port
);
console.log(url);
```
### Processes
#### Run A Command
Use `sandbox.exec(...)` for one-shot commands. You can also pass a plain string
like `await sandbox.exec("node -v")`. Commands are executed through `/bin/sh -lc`.
```typescript theme={null}
const result = await sandbox.exec("pwd && echo $FOO && whoami", {
cwd: "/tmp", // optional: working directory
env: { FOO: "bar" }, // optional: environment variables
timeoutMs: 5_000, // optional: runtime limit in milliseconds
runAs: "root", // optional: run as a specific sandbox user
});
console.log(result.stdout.trim());
```
If stdout contains JSON, parse it explicitly:
```typescript theme={null}
const result = await sandbox.exec(
`python3 -c 'import json; print(json.dumps({"user": "root", "ok": True}))'`,
{ runAs: "root" }
);
const data = JSON.parse(result.stdout.trim());
console.log(data.user);
```
#### Start A Process
Use `sandbox.processes.start(...)` for long-running processes.
```typescript theme={null}
const process = await sandbox.processes.start("sleep 30", {
cwd: "/tmp", // optional: working directory
env: { FOO: "bar" }, // optional: environment variables
runAs: "root", // optional: run as a specific sandbox user
});
console.log(process.id);
```
#### Get A Process
```typescript theme={null}
const process = await sandbox.getProcess(
"process-id" // process ID
);
```
#### List Processes
```typescript theme={null}
const response = await sandbox.processes.list({
status: ["queued", "running"], // optional: one status or multiple statuses
limit: 20, // optional: max results
cursor: "next-cursor", // optional: pagination cursor
createdAfter: Date.now() - 60_000, // optional: lower timestamp bound
createdBefore: Date.now(), // optional: upper timestamp bound
});
console.log(response.data.map((entry) => entry.id));
```
#### Write Process Stdin
```typescript theme={null}
await process.writeStdin({
data: "hello\n", // optional: stdin payload
encoding: "utf8", // optional: "utf8" or "base64"
eof: true, // optional: close stdin after this write
});
```
### Files
#### Read Text File
```typescript theme={null}
const text = await sandbox.files.readText(
"/tmp/hello.txt", // path inside the sandbox
{
offset: 0, // optional: byte offset
length: 128, // optional: max bytes to read
}
);
console.log(text);
```
#### Write Text File
```typescript theme={null}
await sandbox.files.writeText(
"/tmp/hello.txt", // path inside the sandbox
"hello from sandbox", // file contents
{
append: true, // optional: append instead of overwrite
mode: "0640", // optional: chmod-style mode string
}
);
```
#### List Files
```typescript theme={null}
const entries = await sandbox.files.list(
"/tmp", // directory path
{
depth: 2, // optional: traversal depth, minimum 1
}
);
console.log(entries.map((entry) => entry.path));
```
#### Watch A Directory
```typescript theme={null}
const handle = await sandbox.files.watchDir(
"/tmp/watch", // directory to watch
(event) => {
console.log(event.type, event.name);
}, // callback for file events
{
recursive: true, // optional: watch nested directories
timeoutMs: 30_000, // optional: auto-stop after this many ms
}
);
await handle.stop();
```
#### Create Upload URL
```typescript theme={null}
const upload = await sandbox.files.uploadUrl(
"/tmp/upload.txt", // target path inside the sandbox
{
oneTime: true, // optional: invalidate the URL after one use
expiresInSeconds: 60, // optional: URL lifetime
}
);
console.log(upload.method, upload.url);
```
#### Create Download URL
```typescript theme={null}
const download = await sandbox.files.downloadUrl(
"/tmp/hello.txt", // source path inside the sandbox
{
oneTime: true, // optional: invalidate the URL after one use
expiresInSeconds: 60, // optional: URL lifetime
}
);
console.log(download.method, download.url);
```
### Terminal
#### Create A Terminal
Use `sandbox.terminal.create(...)` or the alias `sandbox.pty.create(...)`.
```typescript theme={null}
const terminal = await sandbox.terminal.create({
command: "bash", // required: command to launch
args: ["-l"], // optional: command arguments
cwd: "/tmp", // optional: working directory
env: { FOO: "bar" }, // optional: environment variables
rows: 24, // optional: terminal rows
cols: 80, // optional: terminal columns
timeoutMs: 60_000, // optional: PTY timeout in milliseconds
});
console.log(terminal.id);
```
#### Get A Terminal
```typescript theme={null}
const terminal = await sandbox.terminal.get(
"terminal-id", // terminal ID
true // optional: include buffered output
);
console.log(terminal.current.output?.length ?? 0);
```
### Snapshots
#### Create A Memory Snapshot
```typescript theme={null}
const snapshot = await sandbox.createMemorySnapshot({
snapshotName: "node-after-setup", // optional: custom snapshot name
});
console.log(snapshot.snapshotId);
```
Sandbox guides:
* [Creating Sandboxes](/docs/sandboxes/create)
* [Sandbox Lifecycle](/docs/sandboxes/lifecycle)
* [Sandbox Processes](/docs/sandboxes/processes)
* [Local Filesystem](/docs/sandboxes/filesystem/overview)
* [Sandbox Terminal](/docs/sandboxes/terminal)
* [Sandbox Snapshots](/docs/sandboxes/snapshots)
## Session Runtime Updates
Use session update helpers to change supported settings on an active browser without recreating it.
### CAPTCHA Solving
```typescript theme={null}
await client.sessions.startCaptchaSolving("session-id", {
solverType: "visual",
});
await client.sessions.stopCaptchaSolving("session-id");
```
`solverType: "visual"` enables the visual reCAPTCHA solver. Omit `solverType` to use the default automatic CAPTCHA solver configuration.
### Manual CAPTCHA Evaluation
```typescript theme={null}
const result = await client.sessions.evaluateCaptcha("session-id", {
captchaType: "recaptcha",
iterations: 2,
});
```
Supported `captcha` and `captchaType` values are `turnstile`, `cloudflare-challenge`, `aliexpress`, `recaptcha`, and `amazon`.
## Integration Examples
```typescript Playwright theme={null}
import { chromium } from "playwright-core";
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const session = await client.sessions.create({
acceptCookies: true,
});
try {
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
await page.goto("https://example.com");
console.log(await page.title());
} catch (err) {
console.error(`Encountered error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```typescript Puppeteer theme={null}
import { connect } from "puppeteer-core";
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const session = await client.sessions.create({
acceptCookies: true,
});
try {
const browser = await connect({
browserWSEndpoint: session.wsEndpoint,
defaultViewport: null,
});
const defaultContext = browser.defaultBrowserContext();
const page = (await defaultContext.pages())[0];
await page.goto("https://example.com");
console.log(await page.title());
} catch (err) {
console.error(`Encountered error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```typescript Selenium theme={null}
import dotenv from "dotenv";
import https from "https";
import { Builder, WebDriver } from "selenium-webdriver";
import fs from "fs";
import { Options } from "selenium-webdriver/chrome";
import { Hyperbrowser } from "@hyperbrowser/sdk";
// Load environment variables from .env file
dotenv.config();
const client = new Hyperbrowser({ apiKey: process.env.HYPERBROWSER_API_KEY });
async function main() {
const session = await client.sessions.create();
if (!session.webdriverEndpoint) {
await client.sessions.stop(session.id);
throw new Error("No webdriver endpoint found");
}
const customHttpsAgent = new https.Agent({});
(customHttpsAgent as any).addRequest = (req: any, options: any) => {
req.setHeader("x-hyperbrowser-token", session.token);
(https.Agent.prototype as any).addRequest.call(
customHttpsAgent,
req,
options
);
};
const driver: WebDriver = await new Builder()
.forBrowser("chrome")
.usingHttpAgent(customHttpsAgent)
.usingServer(session.webdriverEndpoint)
.setChromeOptions(new Options())
.build();
try {
// Navigate to a URL
await driver.get("https://www.google.com");
console.log("Navigated to Google");
// Search
const searchBox = await driver.findElement({ name: "q" });
await searchBox.sendKeys("Selenium WebDriver");
await searchBox.submit();
console.log("Performed search");
// Screenshot
await driver.takeScreenshot().then((data) => {
fs.writeFileSync("search_results.png", data, "base64");
});
console.log("Screenshot saved");
} finally {
await driver.quit();
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
## Computer Actions
Programmatically control the browser with low-level actions.
For the `session-id` parameter, you can also just pass in the detailed session object itself, and it is actually recommended to do so.
### Click
```typescript theme={null}
const response = await client.computerAction.click(
"session-id", // or session object
500, // x coordinate
300, // y coordinate
"left", // button: "left" | "right" | "middle" | "back" | "forward" | "wheel"
1, // number of clicks
false // do not return screenshot (default: false)
);
console.log(response.success);
console.log(response.screenshot); // base64 if requested
```
### Type Text
```typescript theme={null}
const response = await client.computerAction.typeText(
"session-id",
"Hello, World!",
false // do not return screenshot (default: false)
);
```
### Press Keys
Uses the xdotool format for keys: [https://github.com/sickcodes/xdotool-gui/blob/master/key\_list.csv](https://github.com/sickcodes/xdotool-gui/blob/master/key_list.csv)
```typescript theme={null}
const response = await client.computerAction.pressKeys(
"session-id",
["Control_L", "a"], // Key combination
false // do not return screenshot (default: false)
);
```
### Move Mouse
```typescript theme={null}
const response = await client.computerAction.moveMouse(
"session-id",
500, // x
300, // y
false // do not return screenshot (default: false)
);
```
### Drag
```typescript theme={null}
const response = await client.computerAction.drag(
"session-id",
[
{ x: 100, y: 100 },
{ x: 200, y: 200 },
{ x: 300, y: 300 },
],
false // do not return screenshot (default: false)
);
```
### Scroll
```typescript theme={null}
const response = await client.computerAction.scroll(
"session-id",
500, // x position
300, // y position
0, // scroll x delta
100, // scroll y delta
false // do not return screenshot (default: false)
);
```
### Screenshot
```typescript theme={null}
const response = await client.computerAction.screenshot("session-id");
console.log(response.screenshot); // base64
```
## Support
* **GitHub Issues**: [https://github.com/hyperbrowserai/node-sdk/issues](https://github.com/hyperbrowserai/node-sdk/issues)
* **Email**: [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai)
# Python SDK
Source: https://hyperbrowser.ai/docs/sdks/python
Complete guide to the Hyperbrowser Python SDK
View on GitHub
View on PyPI
## Installation
Install the Hyperbrowser SDK:
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Quick Start
The Hyperbrowser Python SDK supports both synchronous and asynchronous clients.
### Synchronous Client
```python theme={null}
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Create a session
session = client.sessions.create()
print(session.ws_endpoint)
```
### Asynchronous Client
```python theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
session = await client.sessions.create()
print(session.ws_endpoint)
asyncio.run(main())
```
### Configuration Options
Both clients accept the same configuration parameters:
```python theme={null}
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(
api_key="your-api-key", # Can also use HYPERBROWSER_API_KEY env var
base_url="https://api.hyperbrowser.ai", # Optional, default shown
timeout=30, # Request timeout in seconds
runtime_proxy_override="regional-proxy.internal" # Optional sandbox runtime override
)
```
## Typing Support
The SDK is fully typed end-to-end. In Hyperbrowser 1.0, request parameters are
described with `TypedDict`, so editors autocomplete keys directly inside plain
dictionary literals.
Import a request type from `hyperbrowser.types` when you want to annotate a
variable:
```python Python 1.0+ theme={null}
from hyperbrowser.types import CreateSessionParams
params: CreateSessionParams = {"use_stealth": True}
session = client.sessions.create(params)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
params: CreateSessionParams = CreateSessionParams(use_stealth=True)
session = client.sessions.create(params)
```
* **Request parameters**: Most methods accept a single typed dictionary. Pass only the fields you need.
* **Field names and aliases**: Use Pythonic snake-case keys; the SDK serializes them to API field names automatically (for example, `use_ultra_stealth` becomes `useUltraStealth`).
* **JSON Schema and open mappings**: User-owned dictionaries, including JSON Schema objects, agent action payloads, environment variables, and storage state values, are preserved rather than recursively renaming their keys. Use the SDK's snake-case key only for the field that contains the object, such as `"output_model_schema"` or `"schema"`.
* **Pydantic-generated schemas**: Schema fields also accept a Pydantic model class where supported; the SDK generates and dereferences its JSON Schema without mutating your model or input dictionary.
* **Responses**: Methods return typed Pydantic models.
* **Backwards compatibility**: Existing Pydantic request classes imported from `hyperbrowser.models` remain supported in 1.0, including their constructors and validation behavior.
Upgrading an existing application? See [Migrating to Python SDK
1.0](/docs/sdks/python-1-0-migration) for side-by-side examples and a compatibility
checklist.
### Example: Create a session
```python Python 1.0+ theme={null}
session = client.sessions.create(
{
"accept_cookies": True,
"screen": {"width": 1920, "height": 1080},
}
)
print("session created", session.id)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams, ScreenConfig
session = client.sessions.create(
CreateSessionParams(
accept_cookies=True,
screen=ScreenConfig(width=1920, height=1080),
)
)
print("session created", session.id)
```
## Sandboxes
Sandbox APIs work in two layers.
Sandbox control methods use your API key to create, inspect, connect to, and stop sandboxes. When you create or connect to a sandbox, the SDK also retrieves a sandbox-scoped runtime token and returns an authenticated `SandboxHandle`.
Sandbox VM operations run inside that started sandbox and use the runtime token on the handle automatically. The SDK handles runtime token refreshes for you, so once you have a running `SandboxHandle`, you can call files, processes, terminal, networking, and snapshot methods without managing runtime auth yourself.
## Sandbox Control
Use these methods to inspect existing sandboxes and discover reusable sandbox
resources.
### Details
#### List Sandboxes
```python Python 1.0+ theme={null}
response = client.sandboxes.list(
{
"status": "active", # optional: sandbox status filter
"search": "sdk", # optional: search term
"page": 1, # optional: page number
"limit": 20, # optional: results per page
}
)
print(response.total_count)
print([entry.id for entry in response.sandboxes])
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxListParams
response = client.sandboxes.list(
SandboxListParams(
status="active", # optional: sandbox status filter
search="sdk", # optional: search term
page=1, # optional: page number
limit=20, # optional: results per page
)
)
print(response.total_count)
print([entry.id for entry in response.sandboxes])
```
#### List Images
```python theme={null}
images = client.sandboxes.list_images()
print([image.image_name for image in images.images])
```
#### List Snapshots
```python Python 1.0+ theme={null}
snapshots = client.sandboxes.list_snapshots(
{
"image_name": "node", # optional: only snapshots for this image
"status": "created", # optional: snapshot status filter
"limit": 10, # optional: max results
}
)
print([snapshot.snapshot_name for snapshot in snapshots.snapshots])
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxSnapshotListParams
snapshots = client.sandboxes.list_snapshots(
SandboxSnapshotListParams(
image_name="node", # optional: only snapshots for this image
status="created", # optional: snapshot status filter
limit=10, # optional: max results
)
)
print([snapshot.snapshot_name for snapshot in snapshots.snapshots])
```
#### Get Sandbox Info
```python theme={null}
detail = sandbox.info()
print(detail.runtime.base_url)
```
### Lifecycle
Use these methods to create, connect to, and stop sandbox instances.
#### Create Sandbox
Use `client.sandboxes.create(...)` to start a sandbox from an image.
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"image_name": "node", # required unless restoring from a snapshot
"region": "us-west", # optional: sandbox region
"timeout_minutes": 30, # optional: max sandbox lifetime
"enable_recording": True, # optional: record the sandbox
"exposed_ports": [{"port": 3000, "auth": True}], # optional
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams, SandboxExposeParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
image_name="node", # required unless restoring from a snapshot
region="us-west", # optional: sandbox region
timeout_minutes=30, # optional: max sandbox lifetime
enable_recording=True, # optional: record the sandbox
exposed_ports=[SandboxExposeParams(port=3000, auth=True)], # optional
)
)
```
#### Start From A Snapshot
Start a new sandbox from a memory checkpoint.
```python Python 1.0+ theme={null}
sandbox = client.sandboxes.create(
{
"snapshot_name": "node-after-setup", # required: snapshot name
"snapshot_id": "snapshot-id", # optional: pin a specific snapshot version
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSandboxParams
sandbox = client.sandboxes.create(
CreateSandboxParams(
snapshot_name="node-after-setup", # required: snapshot name
snapshot_id="snapshot-id", # optional: pin a specific snapshot version
)
)
```
#### Connect To A Running Sandbox
Use `connect(...)` to re-authenticate a running sandbox, refresh its runtime token, and return a `SandboxHandle` for runtime operations. If you already have a handle, call `sandbox.connect()` to re-authenticate it in place.
```python theme={null}
sandbox = client.sandboxes.connect(
"sandbox-id", # running sandbox ID
)
sandbox.connect() # re-authenticate an existing handle in place
```
#### Stop Sandbox
```python theme={null}
sandbox.stop() # stop the sandbox when finished
```
## Sandbox VM Operations
These methods operate on the sandbox VM itself when the sandbox is running.
### Networking
#### Expose Port
```python Python 1.0+ theme={null}
exposure = sandbox.expose(
{
"port": 3000, # required: port inside the sandbox
"auth": True, # optional: require the sandbox bearer token
}
)
print(exposure.url)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxExposeParams
exposure = sandbox.expose(
SandboxExposeParams(
port=3000, # required: port inside the sandbox
auth=True, # optional: require the sandbox bearer token
)
)
print(exposure.url)
```
#### Unexpose Port
```python theme={null}
sandbox.unexpose(
3000, # exposed port to remove
)
```
#### Get Exposed URL
```python theme={null}
url = sandbox.get_exposed_url(
3000, # exposed port
)
print(url)
```
### Processes
#### Run A Command
Use `sandbox.exec(...)` for one-shot commands. You can also pass a plain string
like `sandbox.exec("node -v")`. Commands are executed through `/bin/sh -lc`.
```python theme={null}
result = sandbox.exec(
"pwd && echo $FOO && whoami",
cwd="/tmp", # optional: working directory
env={"FOO": "bar"}, # optional: environment variables
timeout_ms=5000, # optional: runtime limit in milliseconds
run_as="root", # optional: run as a specific sandbox user
)
print(result.stdout.strip())
```
If stdout contains JSON, parse it explicitly:
```python theme={null}
import json
result = sandbox.exec(
"""python3 -c 'import json; print(json.dumps({"user": "root", "ok": True}))'""",
run_as="root",
)
data = json.loads(result.stdout.strip())
print(data["user"])
```
#### Start A Process
Use `sandbox.processes.start(...)` for long-running processes.
```python theme={null}
process = sandbox.processes.start(
"sleep 30",
cwd="/tmp", # optional: working directory
env={"FOO": "bar"}, # optional: environment variables
run_as="root", # optional: run as a specific sandbox user
)
print(process.id)
```
#### Get A Process
```python theme={null}
process = sandbox.get_process(
"process-id", # process ID
)
```
#### List Processes
```python theme={null}
response = sandbox.processes.list(
status=["queued", "running"], # optional: one status or multiple statuses
limit=20, # optional: max results
cursor="next-cursor", # optional: pagination cursor
created_after=1711929600000, # optional: lower timestamp bound
created_before=1712016000000, # optional: upper timestamp bound
)
print([entry.id for entry in response.data])
```
#### Write Process Stdin
```python theme={null}
process.write_stdin(
data="hello\n", # optional: stdin payload
encoding="utf8", # optional: "utf8" or "base64"
eof=True, # optional: close stdin after this write
)
```
### Files
#### Read Text File
```python theme={null}
text = sandbox.files.read_text(
"/tmp/hello.txt", # path inside the sandbox
offset=0, # optional: byte offset
length=128, # optional: max bytes to read
)
print(text)
```
#### Write Text File
```python theme={null}
sandbox.files.write_text(
"/tmp/hello.txt", # path inside the sandbox
"hello from sandbox", # file contents
append=True, # optional: append instead of overwrite
mode="0640", # optional: chmod-style mode string
)
```
#### List Files
```python theme={null}
entries = sandbox.files.list(
"/tmp", # directory path
depth=2, # optional: traversal depth, minimum 1
)
print([entry.path for entry in entries])
```
#### Watch A Directory
```python theme={null}
def on_event(event):
print(event.type, event.name)
watch = sandbox.files.watch_dir(
"/tmp/watch", # directory to watch
on_event, # callback for file events
recursive=True, # optional: watch nested directories
timeout_ms=30000, # optional: auto-stop after this many ms
)
watch.stop()
```
#### Create Upload URL
```python theme={null}
upload = sandbox.files.upload_url(
"/tmp/upload.txt", # target path inside the sandbox
one_time=True, # optional: invalidate the URL after one use
expires_in_seconds=60, # optional: URL lifetime
)
print(upload.method, upload.url)
```
#### Create Download URL
```python theme={null}
download = sandbox.files.download_url(
"/tmp/hello.txt", # source path inside the sandbox
one_time=True, # optional: invalidate the URL after one use
expires_in_seconds=60, # optional: URL lifetime
)
print(download.method, download.url)
```
### Terminal
#### Create A Terminal
Use `sandbox.terminal.create(...)` or the alias `sandbox.pty.create(...)`.
```python Python 1.0+ theme={null}
terminal = sandbox.terminal.create(
{
"command": "bash", # required: command to launch
"args": ["-l"], # optional: command arguments
"cwd": "/tmp", # optional: working directory
"env": {"FOO": "bar"}, # optional: environment variables
"rows": 24, # optional: terminal rows
"cols": 80, # optional: terminal columns
"timeout_ms": 60000, # optional: PTY timeout in milliseconds
}
)
print(terminal.id)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxTerminalCreateParams
terminal = sandbox.terminal.create(
SandboxTerminalCreateParams(
command="bash", # required: command to launch
args=["-l"], # optional: command arguments
cwd="/tmp", # optional: working directory
env={"FOO": "bar"}, # optional: environment variables
rows=24, # optional: terminal rows
cols=80, # optional: terminal columns
timeout_ms=60000, # optional: PTY timeout in milliseconds
)
)
print(terminal.id)
```
#### Get A Terminal
```python theme={null}
terminal = sandbox.terminal.get(
"terminal-id", # terminal ID
include_output=True, # optional: include buffered output
)
print(len(terminal.current.output or []))
```
### Snapshots
#### Create A Memory Snapshot
```python Python 1.0+ theme={null}
snapshot = sandbox.create_memory_snapshot(
{
"snapshot_name": "node-after-setup", # optional: custom snapshot name
}
)
print(snapshot.snapshot_id)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SandboxMemorySnapshotParams
snapshot = sandbox.create_memory_snapshot(
SandboxMemorySnapshotParams(
snapshot_name="node-after-setup", # optional: custom snapshot name
)
)
print(snapshot.snapshot_id)
```
Sandbox guides:
* [Creating Sandboxes](/docs/sandboxes/create)
* [Sandbox Lifecycle](/docs/sandboxes/lifecycle)
* [Sandbox Processes](/docs/sandboxes/processes)
* [Local Filesystem](/docs/sandboxes/filesystem/overview)
* [Sandbox Terminal](/docs/sandboxes/terminal)
* [Sandbox Snapshots](/docs/sandboxes/snapshots)
## Session Runtime Updates
Use session update helpers to change supported settings on an active browser without recreating it.
### CAPTCHA Solving
```python Python 1.0+ theme={null}
client.sessions.start_captcha_solving(
"session-id",
{"solver_type": "visual"},
)
client.sessions.stop_captcha_solving("session-id")
```
```python Python (legacy) theme={null}
from hyperbrowser.models import UpdateSessionSolveCaptchasParams
client.sessions.start_captcha_solving(
"session-id",
UpdateSessionSolveCaptchasParams(solver_type="visual"),
)
client.sessions.stop_captcha_solving("session-id")
```
`solver_type="visual"` enables the visual reCAPTCHA solver. Omit `solver_type` to use the default automatic CAPTCHA solver configuration.
### Manual CAPTCHA Evaluation
```python Python 1.0+ theme={null}
result = client.sessions.evaluate_captcha(
"session-id",
{
"captcha_type": "recaptcha",
"iterations": 2,
},
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CaptchaEvaluationParams
result = client.sessions.evaluate_captcha(
"session-id",
CaptchaEvaluationParams(
captcha_type="recaptcha",
iterations=2,
),
)
```
Supported `captcha` and `captcha_type` values are `turnstile`, `cloudflare-challenge`, `aliexpress`, `recaptcha`, and `amazon`.
## Integration Examples
```python Playwright (Sync, Python 1.0+) theme={null}
from playwright.sync_api import sync_playwright
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
# Create session
session = client.sessions.create({"accept_cookies": True})
try:
# Connect with Playwright
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
page.goto("https://example.com")
print(f"Page title: {page.title()}")
except Exception as e:
print(f"Error: {e}")
finally:
# Stop session
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
```python Playwright (Sync, legacy) theme={null}
from playwright.sync_api import sync_playwright
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
import os
from hyperbrowser.models import CreateSessionParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
# Create session
session = client.sessions.create(params=CreateSessionParams(accept_cookies=True))
try:
# Connect with Playwright
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
page.goto("https://example.com")
print(f"Page title: {page.title()}")
except Exception as e:
print(f"Error: {e}")
finally:
# Stop session
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
```python Playwright (Async, Python 1.0+) theme={null}
import asyncio
from playwright.async_api import async_playwright
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
# Create session
session = await client.sessions.create({"accept_cookies": True})
try:
# Connect with Playwright
async with async_playwright() as p:
browser = await p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
await page.goto("https://example.com")
print(f"Page title: {await page.title()}")
except Exception as e:
print(f"Error: {e}")
finally:
# Stop session
await client.sessions.stop(session.id)
if __name__ == "__main__":
asyncio.run(main())
```
```python Playwright (Async, legacy) theme={null}
import asyncio
from playwright.async_api import async_playwright
from hyperbrowser import AsyncHyperbrowser
from dotenv import load_dotenv
import os
from hyperbrowser.models import CreateSessionParams
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
# Create session
session = await client.sessions.create(
params=CreateSessionParams(accept_cookies=True)
)
try:
# Connect with Playwright
async with async_playwright() as p:
browser = await p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
await page.goto("https://example.com")
print(f"Page title: {await page.title()}")
except Exception as e:
print(f"Error: {e}")
finally:
# Stop session
await client.sessions.stop(session.id)
if __name__ == "__main__":
asyncio.run(main())
```
```python Selenium (Python 1.0+) theme={null}
import os
from dotenv import load_dotenv
from selenium import webdriver
from selenium.webdriver.remote.client_config import ClientConfig
from selenium.webdriver.remote.remote_connection import RemoteConnection
from selenium.webdriver.chrome.options import Options
from hyperbrowser import Hyperbrowser
# Load environment variables from .env file
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class CustomRC(RemoteConnection):
_signing_key = None
def __init__(self, server: str, token: str):
super().__init__(
client_config=ClientConfig(
remote_server_addr=server,
extra_headers={"x-hyperbrowser-token": token},
)
)
def main():
session = client.sessions.create({"accept_cookies": True})
driver = None
try:
custom_conn = CustomRC(session.webdriver_endpoint, session.token)
driver = webdriver.Remote(custom_conn, options=Options())
# Navigate to a URL
driver.get("https://www.google.com")
print("Navigated to Google")
# Search
search_box = driver.find_element("name", "q")
search_box.send_keys("Selenium WebDriver")
search_box.submit()
print("Performed search")
# Screenshot
driver.save_screenshot("search_results.png")
print("Screenshot saved")
except Exception as e:
print(f"Error: {e}")
finally:
if driver is not None:
driver.quit()
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
```python Selenium (legacy) theme={null}
import os
from dotenv import load_dotenv
from selenium import webdriver
from selenium.webdriver.remote.client_config import ClientConfig
from selenium.webdriver.remote.remote_connection import RemoteConnection
from selenium.webdriver.chrome.options import Options
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
# Load environment variables from .env file
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class CustomRC(RemoteConnection):
_signing_key = None
def __init__(self, server: str, token: str):
super().__init__(
client_config=ClientConfig(
remote_server_addr=server,
extra_headers={"x-hyperbrowser-token": token},
)
)
def main():
session = client.sessions.create(params=CreateSessionParams(accept_cookies=True))
driver = None
try:
custom_conn = CustomRC(session.webdriver_endpoint, session.token)
driver = webdriver.Remote(custom_conn, options=Options())
# Navigate to a URL
driver.get("https://www.google.com")
print("Navigated to Google")
# Search
search_box = driver.find_element("name", "q")
search_box.send_keys("Selenium WebDriver")
search_box.submit()
print("Performed search")
# Screenshot
driver.save_screenshot("search_results.png")
print("Screenshot saved")
except Exception as e:
print(f"Error: {e}")
finally:
if driver is not None:
driver.quit()
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
## Computer Actions
Programmatically control the browser with low-level actions.
For the `session-id` parameter, you can also just pass in the detailed session object itself, and it is actually recommended to do so.
### Click
```python Sync theme={null}
response = client.computer_action.click(
"session-id", # or session object
x=500,
y=300,
button="left", # "left" | "right" | "middle" | "back" | "forward" | "wheel"
num_clicks=1,
return_screenshot=False # do not return screenshot (default: False)
)
print(response.success)
print(response.screenshot) # base64 if requested
```
```python Async theme={null}
response = await client.computer_action.click(
"session-id",
x=500,
y=300,
button="left",
return_screenshot=False # do not return screenshot (default: False)
)
```
### Type Text
```python Sync theme={null}
response = client.computer_action.type_text(
"session-id",
text="Hello, World!",
return_screenshot=False # do not return screenshot (default: False)
)
```
```python Async theme={null}
response = await client.computer_action.type_text(
"session-id",
text="Hello, World!"
)
```
### Press Keys
Uses the xdotool format for keys: [https://github.com/sickcodes/xdotool-gui/blob/master/key\_list.csv](https://github.com/sickcodes/xdotool-gui/blob/master/key_list.csv)
```python Sync theme={null}
response = client.computer_action.press_keys(
"session-id",
keys=["Control_L", "a"], # Key combination
return_screenshot=False # do not return screenshot (default: False)
)
```
```python Async theme={null}
response = await client.computer_action.press_keys(
"session-id",
keys=["Control_L", "a"]
)
```
### Move Mouse
```python Sync theme={null}
response = client.computer_action.move_mouse(
"session-id",
x=500,
y=300,
return_screenshot=False # do not return screenshot (default: False)
)
```
```python Async theme={null}
response = await client.computer_action.move_mouse(
"session-id",
x=500,
y=300
)
```
### Drag
```python Python 1.0+ (sync) theme={null}
response = client.computer_action.drag(
"session-id",
path=[
{"x": 100, "y": 100},
{"x": 200, "y": 200},
{"x": 300, "y": 300},
],
return_screenshot=False, # do not return screenshot (default: False)
)
```
```python Python (legacy, sync) theme={null}
from hyperbrowser.models import Coordinate
response = client.computer_action.drag(
"session-id",
path=[
Coordinate(x=100, y=100),
Coordinate(x=200, y=200),
Coordinate(x=300, y=300),
],
return_screenshot=False, # do not return screenshot (default: False)
)
```
```python Python 1.0+ (async) theme={null}
response = await client.computer_action.drag(
"session-id",
path=[
{"x": 100, "y": 100},
{"x": 200, "y": 200},
],
)
```
```python Python (legacy, async) theme={null}
from hyperbrowser.models import Coordinate
response = await client.computer_action.drag(
"session-id",
path=[
Coordinate(x=100, y=100),
Coordinate(x=200, y=200),
],
)
```
### Scroll
```python Sync theme={null}
response = client.computer_action.scroll(
"session-id",
x=500,
y=300,
scroll_x=0,
scroll_y=100,
return_screenshot=False # do not return screenshot (default: False)
)
```
```python Async theme={null}
response = await client.computer_action.scroll(
"session-id",
x=500,
y=300,
scroll_x=0,
scroll_y=100
)
```
### Screenshot
```python Sync theme={null}
response = client.computer_action.screenshot("session-id")
print(response.screenshot) # base64
```
```python Async theme={null}
response = await client.computer_action.screenshot("session-id")
```
## Support
* **GitHub Issues**: [https://github.com/hyperbrowserai/python-sdk/issues](https://github.com/hyperbrowserai/python-sdk/issues)
* **Email**: [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai)
# Migrating to Python SDK 1.0
Source: https://hyperbrowser.ai/docs/sdks/python-1-0-migration
Adopt typed request dictionaries while keeping existing Pydantic request code working
Python SDK 1.0 makes plain dictionaries the preferred request format. Method
signatures use `TypedDict`, so editors can autocomplete and validate top-level
and nested keys directly in dictionary literals.
Existing Pydantic request objects remain supported. You can upgrade first, then
migrate call sites incrementally.
## Upgrade
```bash theme={null}
pip install --upgrade "hyperbrowser>=1.0,<2"
```
## Before and after
New and updated code should pass a dictionary:
```python theme={null}
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(api_key="your-api-key")
session = client.sessions.create(
{
"use_stealth": True,
"screen": {"width": 1920, "height": 1080},
}
)
```
The equivalent Pydantic request from earlier SDK versions remains valid:
```python theme={null}
from hyperbrowser.models import CreateSessionParams, ScreenConfig
session = client.sessions.create(
CreateSessionParams(
use_stealth=True,
screen=ScreenConfig(width=1920, height=1080),
)
)
```
Both forms can be used in the same application.
## Named request annotations
Import request `TypedDict` definitions from `hyperbrowser.types` when a named
variable is useful:
```python theme={null}
from hyperbrowser.types import CreateSessionParams
params: CreateSessionParams = {
"use_stealth": True,
"screen": {"width": 1920, "height": 1080},
}
session = client.sessions.create(params)
```
The same name under `hyperbrowser.models` refers to the legacy Pydantic request
class. Alias one of the imports if you need both:
```python theme={null}
from hyperbrowser.models import CreateSessionParams as LegacyCreateSessionParams
from hyperbrowser.types import CreateSessionParams
```
## What stays the same
* Use Pythonic `snake_case` request keys. The SDK still translates its own
fields to the API's wire names.
* Sync and async clients accept the same request shapes.
* Responses remain Pydantic models with methods such as `model_dump()` and
`model_dump_json()`.
* Legacy Pydantic requests retain their constructors and validation behavior.
If your application deliberately validates a request before making an SDK
call, keep constructing the legacy class. Dictionaries are validated when
the SDK method is called.
## JSON Schema and open mappings
Fields such as `schema` and `output_model_schema` accept raw JSON Schema
dictionaries:
```python theme={null}
result = client.extract.start_and_wait(
{
"urls": ["https://example.com"],
"prompt": "Extract the product name and price.",
"schema": {
"type": "object",
"properties": {
"product_name": {"type": "string"},
"price_in_usd": {"type": "number"},
},
"required": ["product_name", "price_in_usd"],
},
}
)
```
The SDK treats schema content as user-owned data. It does not rename `$defs`,
`$ref`, property names such as `product_name`, or custom keywords. The same
preservation rule applies to environment variables, storage state, sensitive
data, and agent action payloads.
Endpoints that support boolean JSON Schemas also accept `True` and `False`
directly.
Where documented, you can also provide a Pydantic model class and the SDK will
generate its JSON Schema.
## Migration checklist
1. Upgrade and run your current test suite. Existing Pydantic request objects
should continue to work.
2. Prefer dictionaries in new and frequently edited request code.
3. Import named request annotations from `hyperbrowser.types`.
4. Continue importing responses and intentionally retained legacy request
classes from `hyperbrowser.models`.
5. Run your type checker to catch invalid or misspelled nested keys.
No response-model migration is required.
# Ad Blocking
Source: https://hyperbrowser.ai/docs/sessions/ad-blocking
Learn how to block ads and trackers in your browser sessions
Hyperbrowser's browser instances can automatically block ads and trackers. This improves page load times and reduces detection risk. You can also block trackers and other annoyances like cookie notices.
## Enable Ad Blocking
Set the appropriate options when creating a session:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const session = await client.sessions.create({
adblock: true,
trackers: true,
annoyances: true,
// You must have trackers set to true to enable blocking annoyances and adblock set to true to enable blocking trackers.
});
try {
console.log("Session Live URL:", session.liveUrl);
console.log("WS Endpoint:", session.wsEndpoint);
// ... connect with Playwright/Puppeteer and automate
} finally {
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```python Python 1.0+ theme={null}
import asyncio
import os
from dotenv import load_dotenv
from hyperbrowser import AsyncHyperbrowser
# Load environment variables from .env file
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
# Create a session and connect to it using Playwright
session = await client.sessions.create(
params={
"adblock": True,
"trackers": True,
"annoyances": True,
}
)
try:
print(f"Session Live URL: {session.live_url}")
print(f"WS Endpoint: {session.ws_endpoint}")
# ... connect with Playwright/Puppeteer and automate
except Exception as e:
print(f"Error: {e}")
finally:
await client.sessions.stop(session.id)
# Run the async main function
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
import os
from dotenv import load_dotenv
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import CreateSessionParams
# Load environment variables from .env file
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
# Create a session and connect to it using Playwright
session = await client.sessions.create(
params=CreateSessionParams(
adblock=True,
trackers=True,
annoyances=True,
)
)
try:
print(f"Session Live URL: {session.live_url}")
print(f"WS Endpoint: {session.ws_endpoint}")
# ... connect with Playwright/Puppeteer and automate
except Exception as e:
print(f"Error: {e}")
finally:
await client.sessions.stop(session.id)
# Run the async main function
if __name__ == "__main__":
asyncio.run(main())
```
## Trackers and Annoyances
When enabled at session creation, Hyperbrowser can block many trackers and filter common annoyances such as cookie prompts and popups.
```typescript Node.js theme={null}
await client.sessions.create({
adblock: true,
trackers: true,
annoyances: true,
});
```
```python Python 1.0+ theme={null}
await client.sessions.create(
params={
"adblock": True,
"trackers": True,
"annoyances": True,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
await client.sessions.create(
params=CreateSessionParams(
adblock=True,
trackers=True,
annoyances=True,
)
)
```
To enable trackers blocking, adblock must be enabled. To enable annoyance blocking, adblock and trackers must both be enabled.
## Automatically Accept Cookies
Some site prompt users for cookies in a particularly intrusive way for scraping. If the `acceptCookies` param is set, then Hyperbrowser will automatically accept cookies on the browsers behalf.
```typescript Node.js theme={null}
await client.sessions.create({
acceptCookies: true,
});
```
```python Python 1.0+ theme={null}
await client.sessions.create(
params={
"accept_cookies": True,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
await client.sessions.create(
params=CreateSessionParams(
accept_cookies=True,
)
)
```
## Next Steps
Manage session lifecycle
Persist cookies and local storage
Use dedicated IP addresses
Record and replay sessions
# CAPTCHA Solving
Source: https://hyperbrowser.ai/docs/sessions/captcha-solving
Learn how to solve CAPTCHAs in your browser sessions
Hyperbrowser can automatically detect and solve CAPTCHAs when you enable it during session creation.
CAPTCHA solving requires being on a paid plan.
## Enable CAPTCHA Solving
Set the session creation parameter to enable solving:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const session = await client.sessions.create({
solveCaptchas: true,
});
try {
console.log("Session Live URL:", session.liveUrl);
console.log("WS Endpoint:", session.wsEndpoint);
// ... connect with Playwright/Puppeteer and automate
} finally {
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```python Python 1.0+ theme={null}
import asyncio
import os
from dotenv import load_dotenv
from hyperbrowser import AsyncHyperbrowser
# Load environment variables from .env file
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
# Create a session and connect with your preferred automation library
session = await client.sessions.create(
params={
"solve_captchas": True,
}
)
try:
print(f"Session Live URL: {session.live_url}")
print(f"WS Endpoint: {session.ws_endpoint}")
# ... connect with Playwright/Puppeteer and automate
except Exception as e:
print(f"Error: {e}")
finally:
await client.sessions.stop(session.id)
# Run the async main function
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
import os
from dotenv import load_dotenv
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import CreateSessionParams
# Load environment variables from .env file
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
# Create a session and connect with your preferred automation library
session = await client.sessions.create(
params=CreateSessionParams(
solve_captchas=True,
)
)
try:
print(f"Session Live URL: {session.live_url}")
print(f"WS Endpoint: {session.ws_endpoint}")
# ... connect with Playwright/Puppeteer and automate
except Exception as e:
print(f"Error: {e}")
finally:
await client.sessions.stop(session.id)
# Run the async main function
if __name__ == "__main__":
asyncio.run(main())
```
Some sites require proxies to be enabled for CAPTCHA solving to work reliably.
## Enable or Disable During a Session
You can start or stop automatic CAPTCHA solving on an active session without recreating the browser. This is useful when you only want solving enabled around specific pages or workflows.
```typescript Node.js theme={null}
await client.sessions.startCaptchaSolving(session.id, {
solverType: "visual",
});
// ... navigate to pages that may contain CAPTCHAs
await client.sessions.stopCaptchaSolving(session.id);
```
```python Python 1.0+ theme={null}
await client.sessions.start_captcha_solving(
session.id,
{"solver_type": "visual"},
)
# ... navigate to pages that may contain CAPTCHAs
await client.sessions.stop_captcha_solving(session.id)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import UpdateSessionSolveCaptchasParams
await client.sessions.start_captcha_solving(
session.id,
UpdateSessionSolveCaptchasParams(solver_type="visual"),
)
# ... navigate to pages that may contain CAPTCHAs
await client.sessions.stop_captcha_solving(session.id)
```
```bash cURL theme={null}
curl -X PUT "https://api.hyperbrowser.ai/api/session/$SESSION_ID/update" \
-H "Content-Type: application/json" \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-d '{
"type": "solveCaptchas",
"params": {
"enabled": true,
"solverType": "visual"
}
}'
curl -X PUT "https://api.hyperbrowser.ai/api/session/$SESSION_ID/update" \
-H "Content-Type: application/json" \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-d '{
"type": "solveCaptchas",
"params": {
"enabled": false
}
}'
```
Set `solverType` to `"visual"` to use the visual reCAPTCHA solver. If omitted, the session uses the default automatic CAPTCHA solver configuration.
## Run Manual CAPTCHA Evaluation
You can also trigger a bounded CAPTCHA evaluation on an active session. This runs once against the current browser pages and returns the evaluation result.
```typescript Node.js theme={null}
const result = await client.sessions.evaluateCaptcha(session.id, {
captchaType: "recaptcha",
iterations: 2,
});
```
```python Python 1.0+ theme={null}
result = await client.sessions.evaluate_captcha(
session.id,
{
"captcha_type": "recaptcha",
"iterations": 2,
},
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CaptchaEvaluationParams
result = await client.sessions.evaluate_captcha(
session.id,
CaptchaEvaluationParams(
captcha_type="recaptcha",
iterations=2,
),
)
```
```bash cURL theme={null}
curl -X POST "https://api.hyperbrowser.ai/api/session/$SESSION_ID/captcha/evaluate" \
-H "Content-Type: application/json" \
-H "x-api-key: $HYPERBROWSER_API_KEY" \
-d '{
"captchaType": "recaptcha",
"iterations": 2
}'
```
Supported manual CAPTCHA targets are `turnstile`, `cloudflare-challenge`, `aliexpress`, `recaptcha`, and `amazon`.
## Waiting for CAPTCHA Solve
Solving can take time. When navigating to pages that might contain CAPTCHAs, add appropriate waits:
```typescript Node.js theme={null}
const sleep = (ms: number) => new Promise((res) => setTimeout(res, ms));
await page.goto("https://news.ycombinator.com/", { waitUntil: "networkidle0" });
await sleep(20_000);
```
A better approach is to wait on session events so you know exactly when a CAPTCHA is detected and when it is solved. Below is an example using Playwright, but you can adapt the same logic for Puppeteer or other libraries.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
import { chromium } from "playwright";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms));
const MAX_DETECTION_TRIES = 10;
const MAX_SOLVED_TRIES = 30;
const waitForCaptcha = async (sessionId: string, startTimestamp: number) => {
let solved = false;
let detected = false;
let detectionTries = 0;
let solvedTries = 0;
console.log("Waiting for captcha detection...");
while (!detected && detectionTries < MAX_DETECTION_TRIES) {
const resp = await client.sessions.eventLogs.list(sessionId, {
startTimestamp,
types: ["captcha_detected"],
});
const data = resp.data;
detected = data.length > 0;
detectionTries++;
await sleep(1000);
}
if (detected) {
console.log("Captcha detected!");
} else {
console.log("Captcha not detected!");
return;
}
console.log("Waiting for captcha solving...");
while (!solved && solvedTries < MAX_SOLVED_TRIES) {
const resp = await client.sessions.eventLogs.list(sessionId, {
startTimestamp,
types: ["captcha_solved"],
});
const data = resp.data;
solved = data.length > 0;
solvedTries++;
await sleep(1000);
}
if (solved) {
console.log("Captcha solved!");
} else {
console.log("Captcha not solved!");
}
};
const main = async () => {
const session = await client.sessions.create({
solveCaptchas: true,
});
console.log("Session Live URL:", session.liveUrl);
try {
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const context = browser.contexts()[0];
const page = context.pages()[0];
const startTimestamp = Date.now();
await page.goto("https://2captcha.com/demo/cloudflare-turnstile");
await waitForCaptcha(session.id, startTimestamp);
await sleep(5_000);
} catch (err) {
console.error(`Error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
};
main();
```
## Next Steps
Manage session lifecycle
Block ads and trackers
Persist cookies and local storage
Load browser extensions
# Computer Actions
Source: https://hyperbrowser.ai/docs/sessions/computer-actions
Control sessions with low-level mouse and keyboard input
Computer Actions let you drive a live session using screen-level primitives like click, type, drag, scroll, and screenshot. Use them when DOM automation is unreliable or impossible (canvas apps, remote desktops, non-standard controls, or stubborn overlays).
## When To Use Computer Actions
* You need full-screen interaction, not just DOM selectors.
* The page uses canvas or custom rendering where selectors are unreliable.
* You want a robust fallback when Playwright/Puppeteer actions fail.
## How It Works
Every session includes a `computerActionEndpoint`. The SDKs wrap this for you with `client.computerAction.*` (Node.js) and `client.computer_action.*` (Python). You can pass either a session ID or the full session object, and passing the session object is recommended.
Coordinates are pixels relative to the top-left of the session screen. The default screen size is 1280x720. If you change `screen` in session creation, adjust your coordinates accordingly.
## Quickstart
This example navigates using keyboard input, scrolls, and captures a screenshot.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const session = await client.sessions.create();
try {
// Focus address bar and navigate
await client.computerAction.pressKeys(session, ["Control_L", "l"]);
await client.computerAction.typeText(session, "https://example.com");
await client.computerAction.pressKeys(session, ["Return"]);
const shot = await client.computerAction.screenshot(session);
console.log(shot.screenshot); // base64
} finally {
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```python Python theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create()
try:
# Focus address bar and navigate
client.computer_action.press_keys(session, keys=["Control_L", "l"])
client.computer_action.type_text(session, text="https://example.com")
client.computer_action.press_keys(session, keys=["Return"])
shot = client.computer_action.screenshot(session)
print(shot.screenshot) # base64
finally:
client.sessions.stop(session.id)
```
## Action Examples
### Click
```typescript Node.js theme={null}
const response = await client.computerAction.click(
"session-id", // or session object
500, // x coordinate
300, // y coordinate
"left", // button: "left" | "right" | "middle" | "back" | "forward" | "wheel"
1, // number of clicks
false // do not return screenshot (default: false)
);
console.log(response.success);
console.log(response.screenshot); // base64 if requested
```
```python Python theme={null}
response = client.computer_action.click(
"session-id", # or session object
x=500,
y=300,
button="left", # "left" | "right" | "middle" | "back" | "forward" | "wheel"
num_clicks=1,
return_screenshot=False # do not return screenshot (default: False)
)
print(response.success)
print(response.screenshot) # base64 if requested
```
### Type Text
```typescript Node.js theme={null}
const response = await client.computerAction.typeText(
"session-id",
"Hello, World!",
false // do not return screenshot (default: false)
);
```
```python Python theme={null}
response = client.computer_action.type_text(
"session-id",
text="Hello, World!",
return_screenshot=False # do not return screenshot (default: False)
)
```
### Press Keys
Uses the xdotool format for keys: [https://github.com/sickcodes/xdotool-gui/blob/master/key\_list.csv](https://github.com/sickcodes/xdotool-gui/blob/master/key_list.csv)
```typescript Node.js theme={null}
const response = await client.computerAction.pressKeys(
"session-id",
["Control_L", "a"], // Key combination
false // do not return screenshot (default: false)
);
```
```python Python theme={null}
response = client.computer_action.press_keys(
"session-id",
keys=["Control_L", "a"], # Key combination
return_screenshot=False # do not return screenshot (default: False)
)
```
### Move Mouse
```typescript Node.js theme={null}
const response = await client.computerAction.moveMouse(
"session-id",
500, // x
300, // y
false // do not return screenshot (default: false)
);
```
```python Python theme={null}
response = client.computer_action.move_mouse(
"session-id",
x=500,
y=300,
return_screenshot=False # do not return screenshot (default: False)
)
```
### Drag
```typescript Node.js theme={null}
const response = await client.computerAction.drag(
"session-id",
[
{ x: 100, y: 100 },
{ x: 200, y: 200 },
{ x: 300, y: 300 },
],
false // do not return screenshot (default: false)
);
```
```python Python 1.0+ theme={null}
response = client.computer_action.drag(
"session-id",
path=[
{"x": 100, "y": 100},
{"x": 200, "y": 200},
{"x": 300, "y": 300},
],
return_screenshot=False, # do not return screenshot (default: False)
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import Coordinate
response = client.computer_action.drag(
"session-id",
path=[
Coordinate(x=100, y=100),
Coordinate(x=200, y=200),
Coordinate(x=300, y=300),
],
return_screenshot=False, # do not return screenshot (default: False)
)
```
### Scroll
```typescript Node.js theme={null}
const response = await client.computerAction.scroll(
"session-id",
500, // x position
300, // y position
0, // scroll x delta
100, // scroll y delta
false // do not return screenshot (default: false)
);
```
```python Python theme={null}
response = client.computer_action.scroll(
"session-id",
x=500,
y=300,
scroll_x=0,
scroll_y=100,
return_screenshot=False # do not return screenshot (default: False)
)
```
### Screenshot
```typescript Node.js theme={null}
const response = await client.computerAction.screenshot("session-id");
console.log(response.screenshot); // base64
```
```python Python theme={null}
response = client.computer_action.screenshot("session-id")
print(response.screenshot) # base64
```
## Related
* [Create a Session](/docs/sessions/create)
* [Session Parameters](/docs/sessions/parameters)
* [OpenAI CUA Agent](/docs/agents/openai-cua)
* [Claude Computer Use Agent](/docs/agents/claude-computer-use)
* [Meta Computer Use Agent](/docs/agents/meta-computer-use)
* [Grok Computer Use Agent](/docs/agents/grok-computer-use)
# Configuring Sessions
Source: https://hyperbrowser.ai/docs/sessions/create
Learn how to create and configure Hyperbrowser sessions
Sessions are isolated browser instances running in the cloud that you can control programmatically. Each session gives you a WebSocket endpoint to connect with Playwright, Puppeteer, or any CDP-compatible tool.
## Quick Start
Create a session with default settings:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
const session = await client.sessions.create();
console.log("Session ID:", session.id);
console.log("WebSocket:", session.wsEndpoint);
console.log("Live Url:", session.liveUrl);
}
main();
```
```python Python theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create()
print(f"Session ID: {session.id}")
print(f"WebSocket: {session.ws_endpoint}")
print(f"Live Url: {session.live_url}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{}'
```
You can view the live session at the live url link provided in the response.
Active sessions can be also viewed and managed in the [Sessions](https://app.hyperbrowser.ai/features/sessions) page in the dashboard.
Remember to stop the sessions when are you are done with them.
See the [Session Lifecycle](/docs/sessions/lifecycle) guide for more information on managing sessions through code.
## Configuration Options
Customize your session with these options:
```typescript Node.js theme={null}
const session = await client.sessions.create({
acceptCookies: true,
useStealth: true,
useUltraStealth: false,
useProxy: true,
screen: {
width: 1920,
height: 1080,
},
timeoutMinutes: 30,
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
{
"accept_cookies": True,
"use_stealth": True,
"use_ultra_stealth": False,
"use_proxy": True,
"screen": {"width": 1920, "height": 1080},
"timeout_minutes": 30,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams, ScreenConfig
session = client.sessions.create(
params=CreateSessionParams(
accept_cookies=True,
use_stealth=True,
use_ultra_stealth=False,
use_proxy=True,
screen=ScreenConfig(width=1920, height=1080),
timeout_minutes=30,
)
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"acceptCookies": true,
"useStealth": true,
"useUltraStealth": false,
"useProxy": true,
"screen": {
"width": 1920,
"height": 1080
},
"timeoutMinutes": 30
}'
```
### Common Configuration Options
Automatically accept cookies on pages that are visited in the session
Enable standard stealth mode to evade basic bot detection
Enable advanced stealth mode for maximum bot detection evasion (reach out to us at [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) to get access)
Route traffic through a proxy (see [Proxy Guide](/docs/sessions/proxy) for configuration)
Screen dimensions with `width` and `height` (default: 1280x720)
Session timeout in minutes (min: 1, max: 720)
This shows the most commonly used options. For the complete list of configuration parameters including proxy settings, CAPTCHA solving, recordings, profiles, extensions, and more, see:
* [Python SDK Documentation](/docs/sdks/python#create-session) - Full parameter reference
* [Node.js SDK Documentation](/docs/sdks/node#create-session) - Full parameter reference
* [API Reference](/docs/api-reference/create-new-session) - Complete REST API documentation
## Session Response
The API returns a session object with these properties:
```json theme={null}
{
"id": "550e8400-e29b-41d4-a716-446655440000",
"sessionUrl": "https://...",
"wsEndpoint": "wss://...",
"liveUrl": "https://...",
"computerActionEndpoint": "https://...",
"status": "active",
"createdAt": "2024-01-15T10:30:00Z"
}
```
To see the complete list of properties, see the [API Reference](/docs/api-reference/create-new-session).
Unique session identifier
WebSocket URL for browser automation
URL to view the session in real-time
URL to view the session page in the dashboard
URL to send computer actions to the session
Current status (`active`, `closed`, or `error`)
ISO 8601 timestamp
The parameters that were used to launch the session.
See the [API Reference](/docs/api-reference/create-new-session#response-launch-state) for more details.
Credits used by the session
## Stealth Mode
Bypass bot detection with stealth features:
```typescript Node.js theme={null}
// Standard stealth (recommended for most sites)
const session = await client.sessions.create({
useStealth: true,
});
// Ultra stealth (for challenging sites)
const session = await client.sessions.create({
useUltraStealth: true,
});
```
```python Python 1.0+ theme={null}
# Standard stealth (recommended for most sites)
session = client.sessions.create(params={"use_stealth": True})
# Ultra stealth (for challenging sites)
session = client.sessions.create(params={"use_ultra_stealth": True})
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
# Standard stealth (recommended for most sites)
session = client.sessions.create(params=CreateSessionParams(use_stealth=True))
# Ultra stealth (for challenging sites)
session = client.sessions.create(params=CreateSessionParams(use_ultra_stealth=True))
```
```bash cURL theme={null}
# Standard stealth
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{"useStealth": true}'
# Ultra stealth
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{"useUltraStealth": true}'
```
Ultra stealth is only available on enterprise plans. Reach out to us at [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) to learn more.
## Proxies
Route sessions through proxies for geo-targeting or additional anonymity:
```typescript Node.js theme={null}
const session = await client.sessions.create({
useProxy: true,
proxyCountry: "US",
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(params={"use_proxy": True, "proxy_country": "US"})
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
session = client.sessions.create(
params=CreateSessionParams(use_proxy=True, proxy_country="US")
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"proxyCountry": "US"
}'
```
See the [Proxy Guide](/docs/sessions/proxy) for advanced proxy configuration including states, cities, and custom proxy servers.
## Session Timeout
Sessions automatically close after a certain amount of time based on your team's default timeout setting. You can change the default timeout on the [Settings page](https://app.hyperbrowser.ai/settings). You can also configure the timeout per session during session creation:
```typescript Node.js theme={null}
const session = await client.sessions.create({
timeoutMinutes: 60, // 1 hour
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(params={"timeout_minutes": 60})
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
session = client.sessions.create(params=CreateSessionParams(timeout_minutes=60))
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{"timeoutMinutes": 60}'
```
## Custom Screen Size
Set custom screen dimensions for specific testing scenarios:
```typescript Node.js theme={null}
const session = await client.sessions.create({
screen: {
width: 1920,
height: 1080,
},
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(params={"screen": {"width": 1920, "height": 1080}})
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams, ScreenConfig
session = client.sessions.create(
params=CreateSessionParams(screen=ScreenConfig(width=1920, height=1080))
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"screen": {
"width": 1920,
"height": 1080
}
}'
```
## Next Steps
Stop, monitor, and manage active sessions
Use Playwright to control your session
Use Puppeteer to control your session
Deep dive into anti-detection features
# File Downloads
Source: https://hyperbrowser.ai/docs/sessions/downloads
Save and retrieve files downloaded during browser sessions
Hyperbrowser makes it easy to download files during your browser sessions. The files you download get stored securely in our cloud infrastructure as a zip file. You can then retrieve them using a simple API call.
**Self-Hosted Hyperbrowser**: For some instances of self-hosted Hyperbrowser, downloads will be available at `/tmp//downloads` on the host machine.
## How to Download Files
1. Create a new Hyperbrowser session and set the `saveDownloads` param to `true`
2. Connect to the session using your preferred automation framework (like Puppeteer or Playwright)
3. Set the download location in your code
4. Download files
5. Retrieve the download zip URL from Sessions API
## Retrieving Downloads in Sessions
```typescript Node.js theme={null}
import { chromium } from "playwright-core";
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function sleep(ms) {
return new Promise((resolve) => setTimeout(resolve, ms));
}
async function waitForDownload(sessionId, timeout = 15_000) {
const maxRetries = timeout / 1000;
let retries = 0;
while (retries < maxRetries) {
console.log(
`Waiting for download zip to be ready... (${retries + 1}/${maxRetries})`
);
const downloadsResponse = await client.sessions.getDownloadsURL(sessionId);
if (
downloadsResponse.status === "completed" ||
downloadsResponse.status === "failed"
) {
return downloadsResponse;
}
await sleep(1000);
retries++;
}
throw new Error(`Download zip not ready after ${timeout}ms`);
}
async function main() {
const session = await client.sessions.create({
saveDownloads: true,
});
console.log("Session created:", session.id);
try {
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
const cdp = await browser.newBrowserCDPSession();
await cdp.send("Browser.setDownloadBehavior", {
behavior: "allow",
downloadPath: "/tmp/downloads",
eventsEnabled: true,
});
await page.goto("https://browser-tests-alpha.vercel.app/api/download-test");
// Download file from the page
const downloadPromise = page.waitForEvent("download");
await page.getByRole("link", { name: "Download File" }).click();
const download = await downloadPromise;
// Wait for the download to complete
await download.path();
console.log("Downloaded file:", download.suggestedFilename());
await client.sessions.stop(session.id);
await sleep(3000);
// Wait for the zipped downloads to be uploaded to our storage
const downloadsResponse = await waitForDownload(session.id);
console.log("downloadsResponse", downloadsResponse);
} catch (err) {
console.error(`Encountered error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from playwright.async_api import async_playwright
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def waitForDownload(sessionId, timeout=15_000):
maxRetries = timeout // 1000
retries = 0
while retries < maxRetries:
print(f"Waiting for download zip to be ready... ({retries + 1}/{maxRetries})")
downloadsResponse = await client.sessions.get_downloads_url(sessionId)
if (
downloadsResponse.status == "completed"
or downloadsResponse.status == "failed"
):
return downloadsResponse
await asyncio.sleep(1)
retries += 1
raise Exception(f"Download zip not ready after {timeout}ms")
async def main():
session = await client.sessions.create({"save_downloads": True})
print(f"Session created: {session.id}")
try:
async with async_playwright() as p:
browser = await p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
cdp = await browser.new_browser_cdp_session()
await cdp.send(
"Browser.setDownloadBehavior",
{
"behavior": "allow",
"downloadPath": "/tmp/downloads",
"eventsEnabled": True,
},
)
# Navigate to a website
await page.goto("https://browser-tests-alpha.vercel.app/api/download-test")
# Start listening for download events before clicking
async with page.expect_download() as download_info:
await page.get_by_role("link", name="Download File").click()
download = await download_info.value
# Wait for the download to complete
await download.path()
print("Downloaded file:", download.suggested_filename)
# Stop the session and wait a few seconds to check downloads status
await client.sessions.stop(session.id)
print("Session stopped, waiting a few seconds to check downloads status.")
await asyncio.sleep(3)
downloads_response = await waitForDownload(session.id)
print("downloads_response\n", downloads_response.model_dump_json(indent=2))
except Exception as e:
print(f"Error: {e}")
finally:
await client.sessions.stop(session.id)
if __name__ == "__main__":
asyncio.run(main())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import CreateSessionParams
from playwright.async_api import async_playwright
from dotenv import load_dotenv
import os
load_dotenv()
client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def waitForDownload(sessionId, timeout=15_000):
maxRetries = timeout // 1000
retries = 0
while retries < maxRetries:
print(f"Waiting for download zip to be ready... ({retries + 1}/{maxRetries})")
downloadsResponse = await client.sessions.get_downloads_url(sessionId)
if (
downloadsResponse.status == "completed"
or downloadsResponse.status == "failed"
):
return downloadsResponse
await asyncio.sleep(1)
retries += 1
raise Exception(f"Download zip not ready after {timeout}ms")
async def main():
session = await client.sessions.create(CreateSessionParams(save_downloads=True))
print(f"Session created: {session.id}")
try:
async with async_playwright() as p:
browser = await p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
cdp = await browser.new_browser_cdp_session()
await cdp.send(
"Browser.setDownloadBehavior",
{
"behavior": "allow",
"downloadPath": "/tmp/downloads",
"eventsEnabled": True,
},
)
# Navigate to a website
await page.goto("https://browser-tests-alpha.vercel.app/api/download-test")
# Start listening for download events before clicking
async with page.expect_download() as download_info:
await page.get_by_role("link", name="Download File").click()
download = await download_info.value
# Wait for the download to complete
await download.path()
print("Downloaded file:", download.suggested_filename)
# Stop the session and wait a few seconds to check downloads status
await client.sessions.stop(session.id)
print("Session stopped, waiting a few seconds to check downloads status.")
await asyncio.sleep(3)
downloads_response = await waitForDownload(session.id)
print("downloads_response\n", downloads_response.model_dump_json(indent=2))
except Exception as e:
print(f"Error: {e}")
finally:
await client.sessions.stop(session.id)
if __name__ == "__main__":
asyncio.run(main())
```
```typescript Node.js theme={null}
import { connect } from "puppeteer-core";
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function sleep(ms) {
return new Promise((resolve) => setTimeout(resolve, ms));
}
async function waitForDownload(sessionId, timeout = 15_000) {
const maxRetries = timeout / 1000;
let retries = 0;
while (retries < maxRetries) {
console.log(
`Waiting for download zip to be ready... (${retries + 1}/${maxRetries})`
);
const downloadsResponse = await client.sessions.getDownloadsURL(sessionId);
if (
downloadsResponse.status === "completed" ||
downloadsResponse.status === "failed"
) {
return downloadsResponse;
}
await sleep(1000);
retries++;
}
throw new Error(`Download zip not ready after ${timeout}ms`);
}
async function main() {
const session = await client.sessions.create({
saveDownloads: true,
});
console.log("Session created:", session.id);
try {
const browser = await connect({
browserWSEndpoint: session.wsEndpoint,
defaultViewport: null,
});
const defaultContext = browser.defaultBrowserContext();
const page = (await defaultContext.pages())[0];
const cdp = await browser.target().createCDPSession();
await cdp.send("Browser.setDownloadBehavior", {
behavior: "allow",
downloadPath: "/tmp/downloads",
eventsEnabled: true,
});
await page.goto("https://browser-tests-alpha.vercel.app/api/download-test");
// Download file from the page
// Set up download listener
cdp.on("Browser.downloadWillBegin", (event) => {
console.log("Download started:", event.suggestedFilename);
});
// Create a promise that resolves when the download is complete
const downloadPromise = new Promise((resolve) => {
cdp.on("Browser.downloadProgress", (event) => {
if (event.state === "completed" || event.state === "canceled") {
resolve("Done");
}
});
});
// Start the download and wait for it to complete
await Promise.all([downloadPromise, page.locator("#download").click()]);
await client.sessions.stop(session.id);
await sleep(3000);
// Wait for the zipped downloads to be uploaded to our storage
const downloadsResponse = await waitForDownload(session.id);
console.log("downloadsResponse", downloadsResponse);
} catch (err) {
console.error(`Encountered error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
The downloads zip is synced in real-time to files that are downloaded during the session so you can retrieve it before the session stops as well.
By default, most PDF urls opened in the browser will open in a PDF viewer in the browser rather being downloaded. To force all PDFs to be downloaded, enable the `enableAlwaysOpenPdfExternally`/`enable_always_open_pdf_externally` parameter when creating the session.
## Retrieving Downloads with API
After your session completes and files are downloaded, you can retrieve them using the Downloads API:
1. **Get the session ID** - Note the `id` of the session that downloaded files
2. **Call the API** - Use the [Get Session Downloads URL](/docs/api-reference/get-session-downloads-url) endpoint or SDK method `getDownloadsURL()`
3. **Poll for completion** - Check the status until it's `completed`
4. **Download the zip** - Use the `downloadsUrl` to download your files
### Response Format
The API returns a response with the following structure:
```typescript theme={null}
{
status: "not_enabled" | "pending" | "in_progress" | "completed" | "failed"
downloadsUrl: string | null
error?: string | null
}
```
### Status Values
| Status | Description |
| - | - |
| `not_enabled` | The `saveDownloads` parameter was not set to `true` when creating the session |
| `pending` | Files are queued for processing or no files have been downloaded yet in the session |
| `in_progress` | Zip file is being created and uploaded to storage |
| `completed` | Download zip is ready - `downloadsUrl` contains the download link |
| `failed` | Processing failed - check `error` field for details |
The download zip URL is temporary and will expire according to your plan's data retention policy. Download the file promptly after retrieval.
## Using With AI Agents
Similarly as above, you can get the downloads zip of files that were downloaded during the session used by an AI Agent. We just need to create a session with `saveDownloads` set to `true` and then pass in that session ID. Here is an example of doing it with OpenAI CUA.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function sleep(ms: number) {
return new Promise((resolve) => setTimeout(resolve, ms));
}
async function waitForDownload(sessionId, timeout = 15_000) {
const maxRetries = timeout / 1000;
let retries = 0;
while (retries < maxRetries) {
console.log(
`Waiting for download zip to be ready... (${retries + 1}/${maxRetries})`
);
const downloadsResponse = await client.sessions.getDownloadsURL(sessionId);
if (
downloadsResponse.status === "completed" ||
downloadsResponse.status === "failed"
) {
return downloadsResponse;
}
await sleep(1000);
retries++;
}
throw new Error(`Download zip not ready after ${timeout}ms`);
}
async function main() {
const session = await client.sessions.create({
saveDownloads: true,
});
console.log("Session created:", session.id);
try {
const resp = await client.agents.cua.startAndWait({
task: "1. Go to this site https://browser-tests-alpha.vercel.app/api/download-test. 2. Click on the Download File link once, then end the task. Do not wait or double check the download, just end the task.",
sessionId: session.id,
});
console.log("Status:", resp.status);
console.log("Final result:", resp.data?.finalResult);
await sleep(3000);
// Wait for the zipped downloads to be uploaded to our storage
const downloadsResponse = await waitForDownload(session.id);
console.log("downloadsResponse", downloadsResponse);
} catch (err) {
console.error(`Encountered error: ${err}`);
} finally {
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```python Python 1.0+ theme={null}
from time import sleep
from hyperbrowser import Hyperbrowser
from dotenv import load_dotenv
import os
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def waitForDownload(sessionId, timeout=15_000):
maxRetries = timeout // 1000
retries = 0
while retries < maxRetries:
print(f"Waiting for download zip to be ready... ({retries + 1}/{maxRetries})")
downloadsResponse = client.sessions.get_downloads_url(sessionId)
if (
downloadsResponse.status == "completed"
or downloadsResponse.status == "failed"
):
return downloadsResponse
sleep(1)
retries += 1
raise Exception(f"Download zip not ready after {timeout}ms")
def main():
session = client.sessions.create({"save_downloads": True})
print(f"Session created: {session.id}")
try:
resp = client.agents.cua.start_and_wait(
{
"task": "1. Go to this site https://browser-tests-alpha.vercel.app/api/download-test. 2. Click on the Download File link once, then end the task. Do not wait or double check the download, just end the task.",
"session_id": session.id,
}
)
print("Status:", resp.status)
print("Final Result:", resp.data.final_result)
sleep(3)
downloads_response = waitForDownload(session.id)
print("downloads_response", downloads_response.model_dump_json(indent=2))
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
```python Python (legacy) theme={null}
from time import sleep
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams, StartCuaTaskParams
from dotenv import load_dotenv
import os
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def waitForDownload(sessionId, timeout=15_000):
maxRetries = timeout // 1000
retries = 0
while retries < maxRetries:
print(f"Waiting for download zip to be ready... ({retries + 1}/{maxRetries})")
downloadsResponse = client.sessions.get_downloads_url(sessionId)
if (
downloadsResponse.status == "completed"
or downloadsResponse.status == "failed"
):
return downloadsResponse
sleep(1)
retries += 1
raise Exception(f"Download zip not ready after {timeout}ms")
def main():
session = client.sessions.create(CreateSessionParams(save_downloads=True))
print(f"Session created: {session.id}")
try:
resp = client.agents.cua.start_and_wait(
StartCuaTaskParams(
task="1. Go to this site https://browser-tests-alpha.vercel.app/api/download-test. 2. Click on the Download File link once, then end the task. Do not wait or double check the download, just end the task.",
session_id=session.id,
)
)
print("Status:", resp.status)
print("Final Result:", resp.data.final_result)
sleep(3)
downloads_response = waitForDownload(session.id)
print("downloads_response", downloads_response.model_dump_json(indent=2))
except Exception as e:
print(f"Error: {e}")
finally:
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
Check out the [Agents Docs](/docs/agents/openai-cua) for more information on how to use AI Agents with Hyperbrowser.
## Storage and Retention
Downloaded zip files are stored securely in Hyperbrowser's cloud infrastructure and are retained according to your plan's data retention policy.
## Next Steps
Now that you've configured your sessions, start using them to automate tasks:
Extract data from websites
Use AI to automate browser tasks
# Browser Extensions
Source: https://hyperbrowser.ai/docs/sessions/extensions
Load custom Chrome extensions into your browser sessions
Load your own custom Chrome extensions into Hyperbrowser sessions to add custom functionality, developer tools, or automation capabilities.
Hyperbrowser officially supports custom chrome extensions. Extensions from the Chrome webstore may work, but our ability to provide support regarding those would be limited. You must first upload your extension as a `.zip` file before using it in sessions.
## Upload Your Extension
Before using an extension in a session, you must upload it to Hyperbrowser.
### Step 1: Package Your Extension
Package your Chrome extension as a `.zip` file with the standard Chrome extension structure:
```
my-extension/
├── manifest.json
├── background.js
├── content.js
└── icons/
└── icon.png
```
**Example manifest.json:**
```json theme={null}
{
"manifest_version": 3,
"name": "My Custom Extension",
"version": "1.0",
"description": "Custom extension for automation",
"permissions": ["storage", "tabs"],
"background": {
"service_worker": "background.js"
},
"content_scripts": [
{
"matches": [""],
"js": ["content.js"]
}
]
}
```
Ensure your `manifest.json` is at the root level of the `.zip` file, not inside a subdirectory. You should zip the contents of the extension directory, not the directory itself.
### Step 2: Upload to Hyperbrowser
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
// Upload extension (.zip file)
const extension = await client.extensions.create({
filePath: "/path/to/extension.zip",
name: "My Custom Extension", // optional
});
console.log("Extension uploaded:", extension.id);
console.log("Extension name:", extension.name);
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Upload extension (.zip file)
extension = client.extensions.create(
{
"file_path": "/path/to/extension.zip",
"name": "My Custom Extension", # optional
}
)
print(f"Extension uploaded: {extension.id}")
print(f"Extension name: {extension.name}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateExtensionParams
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Upload extension (.zip file)
extension = client.extensions.create(
CreateExtensionParams(
file_path="/path/to/extension.zip",
name="My Custom Extension", # optional
)
)
print(f"Extension uploaded: {extension.id}")
print(f"Extension name: {extension.name}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/extensions/add \
-H "x-api-key: YOUR_API_KEY" \
-F "file=@/path/to/extension.zip" \
-F "name=My Custom Extension"
```
## Use Extensions in Sessions
Once uploaded, load your extension(s) when creating a session:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
// Create session with your uploaded extension
const session = await client.sessions.create({
extensionIds: [
"extension-id-from-upload-step",
],
});
console.log("Session created:", session.id);
console.log("Extension loaded successfully");
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Create session with your uploaded extension
session = client.sessions.create(
params={"extension_ids": ["extension-id-from-upload-step"]}
)
print(f"Session created: {session.id}")
print(f"Extension loaded successfully")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
import os
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Create session with your uploaded extension
session = client.sessions.create(
params=CreateSessionParams(extension_ids=["extension-id-from-upload-step"])
)
print(f"Session created: {session.id}")
print(f"Extension loaded successfully")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"extensionIds": ["extension-id-from-upload-step"]
}'
```
## Managing Extensions
### List Your Extensions
View all uploaded extensions:
```typescript Node.js theme={null}
const extensions = await client.extensions.list();
console.log(extensions)
```
```python Python theme={null}
extensions = client.extensions.list()
print(extensions)
```
```bash cURL theme={null}
curl -X GET https://api.hyperbrowser.ai/api/extensions/list \
-H "x-api-key: YOUR_API_KEY"
```
## Best Practices
Always test your extension in a local Chrome browser before uploading to Hyperbrowser. This helps catch manifest errors and permission issues early.
Keep extensions lightweight. Remove unnecessary files and compress images to reduce upload time and session startup overhead.
Use Manifest V3 for better compatibility and performance. Chrome is phasing out Manifest V2.
Only load extensions you actually need. Each extension adds memory overhead and startup time.
## Troubleshooting
### Extension Not Loading
If your extension doesn't load:
1. **Verify the .zip structure** - Ensure `manifest.json` is at the root level, not in a subdirectory
2. **Check manifest validity** - Validate your `manifest.json` against Chrome extension standards
3. **Test locally first** - Load the extension in Chrome (chrome://extensions) to verify it works
4. **Check file size** - Very large extensions may fail to upload
5. **Review permissions** - Ensure your manifest includes all necessary permissions
### Extension Not Working
If the extension loads but doesn't function:
1. **Check console logs** - Use Live View to see browser console errors
2. **Verify content script matching** - Ensure your `matches` patterns are correct in manifest
3. **Test permissions** - Some APIs require specific permissions in manifest
4. **Check timing** - Extension scripts may need time to initialize before your automation runs
### Upload Failures
If upload fails:
1. **Verify file format** - Must be a `.zip` file
2. **Check file size** - Keep extensions under 8MB
3. **Ensure valid manifest** - Invalid `manifest.json` will cause upload to fail
4. **Remove unnecessary files** - Delete source maps, tests, or development files
## Limitations
**Current Limitations:**
* Extensions must be uploaded as `.zip` files
* Extensions must be compatible with Chrome/Chromium
## Next Steps
Handle file downloads in sessions
Persist extension settings across sessions
Watch extensions in action
Combine with anti-detection features
# Session Lifecycle
Source: https://hyperbrowser.ai/docs/sessions/lifecycle
Manage the lifecycle of your Hyperbrowser sessions
Every Hyperbrowser session follows a predictable lifecycle from creation to termination. Understanding this lifecycle helps you build reliable automation workflows and manage resources effectively.
## Session States
Sessions transition through these states:
* **`active`** - Session is running and ready to accept connections
* **`closed`** - Session has been terminated normally
* **`error`** - Session encountered an error and terminated unexpectedly
## Creating a Session
Create a new session with optional configuration.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
// Create a new session
const session = await client.sessions.create({
screen: {
width: 1920,
height: 1080,
}
});
console.log(`Session ${session.id} is ${session.status}`);
console.log(`WebSocket endpoint: ${session.wsEndpoint}`);
console.log(`Live view: ${session.liveUrl}`);
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(api_key="your-api-key")
# Create a new session
session = client.sessions.create(
{
"screen": {
"width": 1920,
"height": 1080,
}
}
)
print(f"Session {session.id} is {session.status}")
print(f"WebSocket endpoint: {session.ws_endpoint}")
print(f"Live view: {session.live_url}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams, ScreenConfig
client = Hyperbrowser(api_key="your-api-key")
# Create a new session
session = client.sessions.create(
params=CreateSessionParams(screen=ScreenConfig(width=1920, height=1080))
)
print(f"Session {session.id} is {session.status}")
print(f"WebSocket endpoint: {session.ws_endpoint}")
print(f"Live view: {session.live_url}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"timeoutMinutes": 30,
"useUltraStealth": true
}'
```
## Getting Session Details
Retrieve current information about a specific session:
```typescript Node.js theme={null}
const session = await client.sessions.get("session-id");
console.log("Session details:", {
id: session.id,
status: session.status,
createdAt: session.createdAt,
wsEndpoint: session.wsEndpoint,
liveUrl: session.liveUrl,
});
```
```python Python theme={null}
session = client.sessions.get("session-id")
print(f"Session details: {session.id} - {session.status}")
print(f"Created at: {session.created_at}")
print(f"WebSocket endpoint: {session.ws_endpoint}")
print(f"Live view: {session.live_url}")
```
```bash cURL theme={null}
curl -X GET https://api.hyperbrowser.ai/api/session/SESSION_ID \
-H "x-api-key: YOUR_API_KEY"
```
## Listing Sessions
Query all your sessions with optional filtering by status:
```typescript Node.js theme={null}
// List all sessions
const response = await client.sessions.list({
status: "active",
page: 1,
});
console.log(`Total sessions: ${response.totalCount}`);
console.log(`Showing page ${response.page} of ${Math.ceil(response.totalCount / response.perPage)}`);
response.sessions.forEach((session) => {
console.log(`- ${session.id} (${session.status})`);
});
```
```python Python 1.0+ theme={null}
# List active sessions
response = client.sessions.list(
{
"status": "active",
"page": 1,
}
)
print(f"Total sessions: {response.total_count}")
print(f"Showing page {response.page}")
for session in response.sessions:
print(f"- {session.id} ({session.status})")
```
```python Python (legacy) theme={null}
from hyperbrowser.models import SessionListParams
# List active sessions
response = client.sessions.list(
params=SessionListParams(
status="active",
page=1,
)
)
print(f"Total sessions: {response.total_count}")
print(f"Showing page {response.page}")
for session in response.sessions:
print(f"- {session.id} ({session.status})")
```
```bash cURL theme={null}
# List active sessions
curl -X GET "https://api.hyperbrowser.ai/api/sessions?status=active&page=1" \
-H "x-api-key: YOUR_API_KEY"
```
## Stopping a Session
Always stop sessions when you're done to free up resources:
```typescript Node.js theme={null}
// Stop a specific session
await client.sessions.stop("session-id");
console.log("Session stopped successfully");
```
```python Python theme={null}
# Stop a specific session
client.sessions.stop("session-id")
print("Session stopped successfully")
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/session/SESSION_ID/stop \
-H "x-api-key: YOUR_API_KEY"
```
Stopping a session is idempotent - you can safely call it multiple times without errors.
## Complete Lifecycle Example
Here's a complete example demonstrating best practices for session management:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms));
async function runAutomation() {
// 1. Create session with configuration
console.log("Creating session...");
const session = await client.sessions.create({
acceptCookies: true,
});
console.log(`Session created: ${session.id}`);
console.log(`Watch live: ${session.liveUrl}`);
try {
// 2. Connect with Playwright
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const context = browser.contexts()[0];
const page = context.pages()[0];
// 3. Use the session
console.log("Waiting 5 seconds before navigating to example.com");
await sleep(5000);
await page.goto("https://example.com");
const title = await page.title();
console.log(`Page title: ${title}`);
console.log("Waiting 5 seconds before stopping the session");
await sleep(5000);
} catch (error) {
console.error("Automation failed:", error);
} finally {
// 4. Always stop the session
console.log("Stopping session...");
await client.sessions.stop(session.id);
console.log("Session stopped");
}
}
runAutomation().catch(console.error);
```
```python Python 1.0+ theme={null}
import os
import time
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def run_automation():
# 1. Create session with configuration
print("Creating session...")
session = client.sessions.create(
{
"accept_cookies": True,
}
)
print(f"Session created: {session.id}")
print(f"Watch live: {session.live_url}")
try:
# 2. Connect with Playwright
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
context = browser.contexts[0]
page = context.pages[0]
# 3. Use the session
print("Waiting 5 seconds before navigating to example.com")
time.sleep(5)
page.goto("https://example.com")
title = page.title()
print(f"Page title: {title}")
print("Waiting 5 seconds before stopping the session")
time.sleep(5)
except Exception as error:
print(f"Automation failed: {error}")
finally:
# 4. Always stop the session
print("Stopping session...")
client.sessions.stop(session.id)
print("Session stopped")
if __name__ == "__main__":
run_automation()
```
```python Python (legacy) theme={null}
import os
import time
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
from playwright.sync_api import sync_playwright
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def run_automation():
# 1. Create session with configuration
print("Creating session...")
session = client.sessions.create(
params=CreateSessionParams(
accept_cookies=True,
)
)
print(f"Session created: {session.id}")
print(f"Watch live: {session.live_url}")
try:
# 2. Connect with Playwright
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
context = browser.contexts[0]
page = context.pages[0]
# 3. Use the session
print("Waiting 5 seconds before navigating to example.com")
time.sleep(5)
page.goto("https://example.com")
title = page.title()
print(f"Page title: {title}")
print("Waiting 5 seconds before stopping the session")
time.sleep(5)
except Exception as error:
print(f"Automation failed: {error}")
finally:
# 4. Always stop the session
print("Stopping session...")
client.sessions.stop(session.id)
print("Session stopped")
if __name__ == "__main__":
run_automation()
```
## Automatic Timeout
Sessions automatically stop after some time based on their timeout. By default, this is based on your team's default Session Timeout setting which you can change on the [Settings page](https://app.hyperbrowser.ai/settings). You can also configure the timeout per session during session creation:
```typescript Node.js theme={null}
const session = await client.sessions.create({
timeoutMinutes: 15,
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
{
"timeout_minutes": 15,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
session = client.sessions.create(params=CreateSessionParams(timeout_minutes=15))
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{"timeoutMinutes": 15}'
```
Don't rely solely on automatic timeouts. Always explicitly stop sessions in your cleanup logic to avoid unexpected charges and ensure proper resource management.
## Error Handling
Use try-finally blocks to guarantee sessions are stopped, even when errors occur:
```typescript Node.js theme={null}
async function safeAutomation() {
let session;
try {
session = await client.sessions.create();
// Your automation code here
await doSomething(session);
} catch (error) {
console.error("Automation failed:", error);
} finally {
// This always runs, whether success or failure
if (session) {
await client.sessions.stop(session.id);
}
}
}
```
```python Python theme={null}
def safe_automation():
session = None
try:
session = client.sessions.create()
# Your automation code here
do_something(session)
except Exception as error:
print(f"Automation failed: {error}")
finally:
# This always runs, whether success or failure
if session:
client.sessions.stop(session.id)
```
## Long Running Sessions
By default, when you disconnect from a session with an automation library like Playwright or Puppeteer, your session will automatically stop. To keep your session alive across disconnects, you can add the `&keepAlive=true` query parameter to your session's WebSocket endpoint when you connect via CDP. This will keep your session alive until it times out based on the session's timeout value (default team setting or `timeoutMinutes` parameter passed in when you create the session) or if you stop the session manually via the API.
The `keepAlive` won't work if all the pages in the browser are closed. If all pages get closed, then the session will automatically stop.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms));
async function runAutomation() {
// 1. Create session with configuration
console.log("Creating session...");
const session = await client.sessions.create({
acceptCookies: true,
});
console.log(`Session created: ${session.id}`);
console.log(`Watch live: ${session.liveUrl}`);
try {
// 2. Connect with Playwright and add the keepAlive parameter
const browser = await chromium.connectOverCDP(
`${session.wsEndpoint}&keepAlive=true`
);
const context = browser.contexts()[0];
const page = context.pages()[0];
// 3. Use the session
console.log("Waiting 5 seconds before navigating to example.com");
await sleep(5000);
await page.goto("https://example.com");
const title = await page.title();
console.log(`Page title: ${title}`);
console.log("Waiting 5 seconds before disconnecting from the browser");
await sleep(5000);
await browser.close();
} catch (error) {
console.error("Automation failed:", error);
}
}
runAutomation().catch(console.error);
```
```python Python 1.0+ theme={null}
import os
import time
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def run_automation():
# 1. Create session with configuration
print("Creating session...")
session = client.sessions.create(
{
"accept_cookies": True,
}
)
print(f"Session created: {session.id}")
print(f"Watch live: {session.live_url}")
try:
# 2. Connect with Playwright and add the keepAlive parameter
# The context manager disconnects when its block finishes.
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(
f"{session.ws_endpoint}&keepAlive=true"
)
context = browser.contexts[0]
page = context.pages[0]
# 3. Use the session
print("Waiting 5 seconds before navigating to example.com")
time.sleep(5)
page.goto("https://example.com")
title = page.title()
print(f"Page title: {title}")
print("Waiting 5 seconds before disconnecting from the browser")
time.sleep(5)
except Exception as error:
print(f"Automation failed: {error}")
if __name__ == "__main__":
run_automation()
```
```python Python (legacy) theme={null}
import os
import time
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
from playwright.sync_api import sync_playwright
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def run_automation():
# 1. Create session with configuration
print("Creating session...")
session = client.sessions.create(
params=CreateSessionParams(
accept_cookies=True,
)
)
print(f"Session created: {session.id}")
print(f"Watch live: {session.live_url}")
try:
# 2. Connect with Playwright and add the keepAlive parameter
# The context manager disconnects when its block finishes.
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(
f"{session.ws_endpoint}&keepAlive=true"
)
context = browser.contexts[0]
page = context.pages[0]
# 3. Use the session
print("Waiting 5 seconds before navigating to example.com")
time.sleep(5)
page.goto("https://example.com")
title = page.title()
print(f"Page title: {title}")
print("Waiting 5 seconds before disconnecting from the browser")
time.sleep(5)
except Exception as error:
print(f"Automation failed: {error}")
if __name__ == "__main__":
run_automation()
```
## Best Practices
Follow these patterns to build reliable, cost-effective automation:
### 1. Always Use Try-Finally
Wrap session usage in try-finally blocks to guarantee cleanup:
```typescript theme={null}
let session;
try {
session = await client.sessions.create();
// ... use session
} finally {
if (session) await client.sessions.stop(session.id);
}
```
### 2. Set Appropriate Timeouts
Match timeout to task duration. Add a buffer for unexpected delays:
* **Quick tasks**: 5-10 minutes
* **Data scraping**: 15-30 minutes
* **Long workflows**: 30-60 minutes
### 3. Monitor Session State
Check session status before long-running operations:
```typescript theme={null}
const session = await client.sessions.get(sessionId);
if (session.status !== 'active') {
throw new Error('Session is no longer active');
}
```
### 4. Clean Up Orphaned Sessions
Periodically audit for abandoned sessions:
```typescript theme={null}
const response = await client.sessions.list({ status: 'active' });
// Review and stop any unexpected active sessions
```
### 5. Handle Network Failures
Network issues can leave sessions running. Always implement cleanup:
```typescript theme={null}
process.on('SIGTERM', async () => {
if (session) await client.sessions.stop(session.id);
process.exit(0);
});
```
## Next Steps
Control sessions with Puppeteer
Control sessions with Playwright
Persist browser state across sessions
Record and replay session activity
# Live View
Source: https://hyperbrowser.ai/docs/sessions/live-view
Watch and interact with your browser sessions in real-time
Live View lets you observe your browser sessions in real-time as they execute. This is essential for debugging automation scripts, monitoring long-running tasks, demonstrating workflows, or enabling human-in-the-loop interactions.
## How it Works
Every Hyperbrowser session automatically includes a unique `liveUrl` that streams the browser session in real-time. The URL remains valid as long as the session is active and the embedded authentication token hasn't expired (tokens expire after 12 hours).
When you create or retrieve a session, you'll receive a `liveUrl` that looks like:
```
https://app.hyperbrowser.ai/live?token=
```
You can open this URL in any modern browser to watch the session live, or embed it in your application.
## Getting the Live URL
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const session = await client.sessions.create();
// Access the Live View URL
console.log("Live View:", session.liveUrl);
console.log("Share this URL to let others watch in real-time");
```
```python Python theme={null}
from hyperbrowser import Hyperbrowser
import os
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create()
# Access the Live View URL
print("Live View:", session.live_url)
print("Share this URL to let others watch in real-time")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{}'
```
## Embedding Live View
You can embed Live View directly into your application using an iframe. This is useful for creating custom monitoring dashboards or providing seamless live views to end-users:
```html theme={null}
```
## Securing Live View
Live View URLs are secured with authentication and encryption. Only users with the correct URL can access the Live View.
Anyone with the URL can view (and potentially interact with) the session. Be sure to protect Live View URLs as sensitive secrets, especially if you're embedding them in a public web page.
**Token Expiration:** The token in the `liveUrl` will expire after 12 hours. To get a refreshed token, simply call the GET request for the session which will return a new `liveUrl` with an updated token.
```typescript Node.js theme={null}
// Get a fresh Live View URL
const session = await client.sessions.get("session-id");
console.log("Refreshed Live View:", session.liveUrl);
```
```python Python theme={null}
# Get a fresh Live View URL
session = client.sessions.get("session-id")
print("Refreshed Live View:", session.live_url)
```
```bash cURL theme={null}
curl https://api.hyperbrowser.ai/api/session/SESSION_ID \
-H "x-api-key: YOUR_API_KEY"
```
## Disabling Live View Interactions
By default, Live View allows users to interact with the session. To disable this, you can set the `viewOnlyLiveView` parameter to `true` when creating the session. This will make the Live View read-only and prevent users from interacting with the session.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const session = await client.sessions.create({
viewOnlyLiveView: true,
});
// Access the Live View URL
console.log("Live View:", session.liveUrl);
console.log("Share this URL to let others watch in real-time");
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(
params={
"view_only_live_view": True,
}
)
# Access the Live View URL
print("Live View:", session.live_url)
print("Share this URL to let others watch in real-time")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
import os
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(
params=CreateSessionParams(
view_only_live_view=True,
)
)
# Access the Live View URL
print("Live View:", session.live_url)
print("Share this URL to let others watch in real-time")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{"viewOnlyLiveView": true}'
```
## Next Steps
Record and replay browser sessions
Manage session lifecycle and cleanup
Persist browser state across sessions
Retrieve files from sessions
# Session Parameters
Source: https://hyperbrowser.ai/docs/sessions/parameters
Learn about the parameters that can be used to configure Hyperbrowser sessions
This page documents the parameters that can be used to configure Hyperbrowser sessions. These parameters are shared across New Session, Scrape, Crawl, Extract, and Agent Tasks.
You can view the Sessions API Reference at [Sessions API Reference](/docs/api-reference/create-new-session) for full REST details.
Proxy usage and CAPTCHA solving require a paid plan.
When using the Python SDK, all session parameters are in **snake\_case** (e.g., `use_proxy`), whereas in JavaScript/TypeScript they use **camelCase** (e.g., `useProxy`).
**Example:**
```python Python 1.0+ theme={null}
# Python SDK
{"use_proxy": True, "proxy_country": "US"}
```
```python Python (legacy) theme={null}
# Python SDK
from hyperbrowser.models import CreateSessionParams
CreateSessionParams(
use_proxy=True,
proxy_country="US",
)
```
## Session Parameters
When `true`, launches the session with advanced stealth techniques to reduce bot detection. This is only available on enterprise plans. Please contact us at [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) to get access.
When `true`, launches the session with standard stealth techniques to reduce bot detection.
When `true`, the session will be launched with a proxy.
Custom proxy server host (used only when `useProxy` is `true`). **Enterprise plan only.** Contact [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) to get access.
Username for authenticating with the custom proxy server. **Enterprise plan only.**
Password for authenticating with the custom proxy server. **Enterprise plan only.**
Country for proxy location (ISO 3166-1 alpha-2 code, e.g., `"US"`, `"GB"`, `"CA"`).
Check [here](/docs/api-reference/create-new-session#body-proxy-country) for the full list of supported countries.
Optional state code for proxies to US states. Is mutually exclusive with proxyCity. Takes in two letter state code.
Check [here](/docs/api-reference/create-new-session#body-proxy-state) for the full list of supported states.
Desired City. Is mutually exclusive with proxyState. Some cities might not be supported, so before using a new city, we recommend trying it out.
Region to run the browser session in.
Check [here](/docs/api-reference/create-new-session#body-region) for the full list of supported regions.
The `asia-south` region is only available on Enterprise and Select plans. [Get in touch](https://calendly.com/shri-hyperbrowser/demo) to learn more.
Screen resolution to emulate. Properties:
* `width` (number, default `1280`)
* `height` (number, default `720`)
When `true`, the session will attempt to automatically solve CAPTCHAs.
Optional CAPTCHA solver mode. Set to `"visual"` to use the visual reCAPTCHA solver when automatic CAPTCHA solving is enabled. In the Python SDK, use `solver_type`.
Block advertisements during the session.
Block trackers and other privacy-invasive technologies during the session.
Block common annoyances like pop-ups and overlays.
Automatically accept cookies on visited sites.
Enable web recording using rrweb.
Enable mp4 video recording which captures the entire screen.
Reuse browser state across sessions. Properties:
* `id` (string): Profile ID to use for the session
* `persistChanges` (boolean, default `false`): Persist changes back to the profile on session close
Static IP ID to use for the session. Check the [Static IPs](/docs/sessions/static-ips) page for more information.
Save downloads to the session. Check the [Downloads](/docs/sessions/downloads) page for more information.
Array of extension IDs to load. Check the [Extensions](/docs/sessions/extensions) page for more information.
URLs to block during the session.
Array of objects for customizing image captcha solving. Each object supports:
* `imageSelector` (string): CSS selector for the captcha image element.
* `inputSelector` (string): CSS selector for the input field where the captcha answer should be entered.
Session timeout in minutes. Overrides your default team timeout from [Settings](https://app.hyperbrowser.ai/settings).
Enable window manager to manage browser windows and tabs.
Enable window manager taskbar to manage browser windows and tabs.
Enable read-only Live View which disables user interactions.
Disable the browser password manager popup on logins.
Always open PDFs externally instead of in browser. This will download the PDF instead of opening it in the browser PDF viewer.
# Connect with Playwright
Source: https://hyperbrowser.ai/docs/sessions/playwright
Control Hyperbrowser sessions using Playwright
Playwright is a powerful browser automation framework that works seamlessly with Hyperbrowser sessions. This guide shows you how to connect Playwright to your cloud browser sessions.
## Installation
First, install Playwright:
```bash npm theme={null}
npm install @hyperbrowser/sdk playwright-core dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk playwright-core dotenv
```
```bash pip theme={null}
pip install hyperbrowser playwright python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser playwright python-dotenv
```
## Basic Connection
Connect Playwright to a Hyperbrowser session:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
// Create session
const session = await client.sessions.create({
acceptCookies: true,
});
try {
// Connect Playwright
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
// Get the default page
const page = defaultContext.pages()[0];
// Navigate and interact
await page.goto("https://example.com");
console.log("Title:", await page.title());
} catch (err) {
console.error("Encountered error:", err);
} finally {
// Clean up
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```python Python theme={null}
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
def main():
# Create session
session = client.sessions.create()
try:
with sync_playwright() as p:
# Connect Playwright
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
# Get the default page
page = default_context.pages[0]
# Navigate and interact
page.goto("https://example.com")
print(f"Title: {page.title()}")
except Exception as e:
print(f"Encountered error: {e}")
finally:
# Clean up
client.sessions.stop(session.id)
main()
```
## Next Steps
Connect with Puppeteer
Manage sessions
Configure anti-detection
Persist browser state
# Profiles
Source: https://hyperbrowser.ai/docs/sessions/profiles
Persist browser state across sessions
Profiles let you save and reuse browser state, which includes cookies, local storage, session storage, and cache, across multiple sessions. Perfect for maintaining login states, preserving user preferences, or building realistic multi session workflows.
## How Profiles Work
A profile is essentially a saved snapshot of a browser's user data directory. By default, each Hyperbrowser session uses a fresh user data directory to ensure isolation.
When you create a profile and then use and persist it in a new session, Hyperbrowser saves that session's user data directory. You can then attach the profile to future sessions.
## Create a Profile
Create a profile to store browser state. Optionally, give it a name to keep track of different profiles:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
// Create a profile with optional name
const profile = await client.profiles.create({
name: "my-profile",
});
console.log("Profile created:", profile.id);
console.log("Profile name:", profile.name);
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
# Create a profile with optional name
profile = client.profiles.create(params={"name": "my-profile"})
print(f"Profile created: {profile.id}")
print(f"Profile name: {profile.name}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateProfileParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
# Create a profile with optional name
profile = client.profiles.create(params=CreateProfileParams(name="my-profile"))
print(f"Profile created: {profile.id}")
print(f"Profile name: {profile.name}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/profile \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "my-profile"
}'
```
## Use a Profile in a Session
Attach a profile to a session to load and save browser state:
```typescript Node.js theme={null}
const session = await client.sessions.create({
profile: {
id: profile.id,
// Optional: Persist changes made to the profile during the session
// Set to true for the first session of a new profile
// persistChanges: true,
},
});
console.log("Session created with profile:", session.id);
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
params={
"profile": {
"id": profile.id,
# Optional: Persist changes made to the profile during the session
# Set to True for the first session of a new profile
# "persist_changes": True,
}
}
)
print(f"Session created with profile: {session.id}")
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams, CreateSessionProfile
session = client.sessions.create(
params=CreateSessionParams(
profile=CreateSessionProfile(
id=profile.id,
# Optional: Persist changes made to the profile during the session
# Set to True for the first session of a new profile
# persist_changes=True,
)
)
)
print(f"Session created with profile: {session.id}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"profile": {
"id": "profile-id-here",
"persistChanges": true
}
}'
```
### Persist Changes
The `persistChanges` parameter controls whether session changes are saved. You will need to do this for the first time you use a new profile in a session so it can be used in subsequent sessions.
* `true` - Persist changes made to the profile during the session
* `false` (default) - Use the profile as read-only; don't persist changes made to the profile during the session
Once whatever changes to the profile have been made, the session should be safely closed to ensure that the profile is saved. After that unless the "persistChanges": true option is passed to the session again, the profile will be immutable.
When using browser automation libraries like Playwright or Puppeteer, it is important to always use the default context in order for profiles to work properly.
### Persist Network Cache
The `persistNetworkCache` parameter controls whether the browser's network cache (HTTP cache) is persisted along with other profile data when persisting changes.
* `true` - Persist network cache
* `false` (default) - Don't persist network cache
Persisting Network Cache requires the `persistChanges` parameter to be set to `true`.
Here’s a minimal example of creating a session with network cache persistence enabled:
```typescript Node.js theme={null}
const session = await client.sessions.create({
profile: {
id: profile.id,
persistChanges: true,
persistNetworkCache: true,
},
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
params={
"profile": {
"id": profile.id,
"persist_changes": True,
"persist_network_cache": True,
}
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams, CreateSessionProfile
session = client.sessions.create(
params=CreateSessionParams(
profile=CreateSessionProfile(
id=profile.id,
persist_changes=True,
persist_network_cache=True,
)
)
)
```
Persisting Network Cache is currently available by request. Please contact us at [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) or on our support chat to enable network cache persistence for your team.
## Profile Login Example
### Step 1: Login and Save to Profile
First, create a profile and perform a login. When the session ends, all cookies and browser state are automatically saved to the profile.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function loginAndSaveProfile() {
// 1. Create a profile
const profile = await client.profiles.create({
name: "hackernews-profile",
});
console.log("Profile created:", profile.id);
// 2. Create session with profile
const session = await client.sessions.create({
profile: {
id: profile.id,
persistChanges: true, // Save browser state to profile
},
});
const HACKER_NEWS_USERNAME = "your_username";
const HACKER_NEWS_PASSWORD = "your_password";
try {
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
// 3. Log in to Hacker News
await page.goto("https://news.ycombinator.com/login");
await page.fill('input[name="acct"]', HACKER_NEWS_USERNAME);
await page.fill('input[name="pw"]', HACKER_NEWS_PASSWORD);
await page.click('input[type="submit"]');
await page.waitForLoadState("networkidle");
const pageUrl = page.url();
const isLoggedIn = pageUrl === "https://news.ycombinator.com/";
if (isLoggedIn) {
console.log("Logged in successfully to Hacker News");
} else {
console.error("Failed to log in to Hacker News");
}
} catch (err) {
console.error("Error logging in to Hacker News:", err);
} finally {
// 4. Stop session - profile now contains login cookies
await client.sessions.stop(session.id);
}
return profile.id;
}
// Save the profile ID for later use
loginAndSaveProfile().then((profileId) => {
console.log("Saved profile ID:", profileId);
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def login_and_save_profile():
# 1. Create a profile
profile = client.profiles.create(params={"name": "hackernews-profile"})
print(f"Profile created: {profile.id}")
# 2. Create session with profile
session = client.sessions.create(
params={
"profile": {
"id": profile.id,
"persist_changes": True, # Save browser state to profile
}
}
)
HACKER_NEWS_USERNAME = "your_username"
HACKER_NEWS_PASSWORD = "your_password"
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
# 3. Log in to Hacker News
page.goto("https://news.ycombinator.com/login")
page.fill('input[name="acct"]', HACKER_NEWS_USERNAME)
page.fill('input[name="pw"]', HACKER_NEWS_PASSWORD)
page.click('input[type="submit"]')
page.wait_for_load_state("networkidle", timeout=5000)
page_url = page.url
is_logged_in = page_url == "https://news.ycombinator.com/"
if is_logged_in:
print("Logged in successfully to Hacker News")
else:
print("Failed to log in to Hacker News")
except Exception as e:
print(f"Error logging in to Hacker News: {e}")
finally:
# 4. Stop session - profile now contains login cookies
client.sessions.stop(session.id)
return profile.id
# Save the profile ID for later use
profile_id = login_and_save_profile()
print(f"Saved profile ID: {profile_id}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import (
CreateSessionParams,
CreateProfileParams,
CreateSessionProfile,
)
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def login_and_save_profile():
# 1. Create a profile
profile = client.profiles.create(
params=CreateProfileParams(name="hackernews-profile")
)
print(f"Profile created: {profile.id}")
# 2. Create session with profile
session = client.sessions.create(
params=CreateSessionParams(
profile=CreateSessionProfile(
id=profile.id,
persist_changes=True, # Save browser state to profile
)
)
)
HACKER_NEWS_USERNAME = "your_username"
HACKER_NEWS_PASSWORD = "your_password"
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
# 3. Log in to Hacker News
page.goto("https://news.ycombinator.com/login")
page.fill('input[name="acct"]', HACKER_NEWS_USERNAME)
page.fill('input[name="pw"]', HACKER_NEWS_PASSWORD)
page.click('input[type="submit"]')
page.wait_for_load_state("networkidle", timeout=5000)
page_url = page.url
is_logged_in = page_url == "https://news.ycombinator.com/"
if is_logged_in:
print("Logged in successfully to Hacker News")
else:
print("Failed to log in to Hacker News")
except Exception as e:
print(f"Error logging in to Hacker News: {e}")
finally:
# 4. Stop session - profile now contains login cookies
client.sessions.stop(session.id)
return profile.id
# Save the profile ID for later use
profile_id = login_and_save_profile()
print(f"Saved profile ID: {profile_id}")
```
A new profile will be created that stores your login credentials and can be used to in future sessions.
You can view and manage your profiles in the [Profiles](https://app.hyperbrowser.ai/features/profiles) page in the dashboard.
Profiles can take a few seconds after the browser session is closed to completely save.
### Step 2: Reuse the Saved Profile
After waiting a couple seconds, now use the saved profile in a new session. The browser will automatically load the saved cookies, so you're already logged in without needing to authenticate again.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function useExistingProfile(profileId) {
// Use the same profile in a new session
const session = await client.sessions.create({
profile: {
id: profileId,
},
});
try {
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
// Navigate directly to Hacker News - already logged in!
await page.goto("https://news.ycombinator.com/", {
waitUntil: "networkidle",
});
try {
await page.waitForSelector("#me", { timeout: 7_000 });
} catch {
console.error("Timed out waiting for selector #me");
}
const username = await page.$eval("#me", (el) =>
el ? el.textContent : null
);
if (username) {
console.log("Still logged in as:", username);
} else {
console.error("Not logged in to Hacker News");
}
} catch (err) {
console.error("Error using existing profile:", err);
} finally {
await client.sessions.stop(session.id);
}
}
useExistingProfile("your-profile-id").then(() => {
console.log("Finished reusing profile!");
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def use_existing_profile(profile_id):
session = client.sessions.create(
params={
"profile": {
"id": profile_id,
}
}
)
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
page.goto("https://news.ycombinator.com/", wait_until="networkidle")
try:
page.wait_for_selector("#me", timeout=7_000)
except:
print("Timed out waiting for selector #me")
username = page.evaluate("""
() => {
const meElement = document.querySelector('#me');
return meElement ? meElement.textContent : null;
}
""")
if username:
print("Still logged in as:", username)
else:
print("Not logged in to Hacker News")
except Exception as e:
print(f"Error using existing profile: {e}")
finally:
client.sessions.stop(session.id)
use_existing_profile("your-profile-id")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams, CreateSessionProfile
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def use_existing_profile(profile_id):
session = client.sessions.create(
params=CreateSessionParams(
profile=CreateSessionProfile(
id=profile_id,
)
)
)
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
page.goto("https://news.ycombinator.com/", wait_until="networkidle")
try:
page.wait_for_selector("#me", timeout=7_000)
except:
print("Timed out waiting for selector #me")
username = page.evaluate("""
() => {
const meElement = document.querySelector('#me');
return meElement ? meElement.textContent : null;
}
""")
if username:
print("Still logged in as:", username)
else:
print("Not logged in to Hacker News")
except Exception as e:
print(f"Error using existing profile: {e}")
finally:
client.sessions.stop(session.id)
use_existing_profile("your-profile-id")
```
**What happens here:**
1. A new session is created using the same profile ID
2. The browser automatically loads the saved cookies and state from the profile
3. You can navigate directly to Hacker News and you're already authenticated
4. No login required - the session picks up right where you left off
This pattern is useful for scenarios where you need to:
* Avoid repeated logins across multiple automation runs
* Test authenticated user flows
* Manage multiple accounts with separate profiles
* Maintain session state over time
## Forking Profiles
Forking lets you create an independent copy of a profile. This is useful when you want to start from an existing authenticated state but save new changes to a separate profile — for example, adding a second login without modifying the original.
Use `profiles.fork()` to create a standalone copy of a profile at any time. The original profile is never modified.
```typescript Node.js theme={null}
// Fork profile A into a new independent profile B
const forkedProfile = await client.profiles.fork(profileA.id, {
name: "profile-b",
});
console.log("Forked profile:", forkedProfile.id);
// Use the forked profile in a new session
const session = await client.sessions.create({
profile: {
id: forkedProfile.id,
persistChanges: true,
},
});
```
```python Python 1.0+ theme={null}
# Fork profile A into a new independent profile B
forked_profile = client.profiles.fork(
profile_a.id,
params={"name": "profile-b"},
)
print(f"Forked profile: {forked_profile.id}")
# Use the forked profile in a new session
session = client.sessions.create(
params={
"profile": {
"id": forked_profile.id,
"persist_changes": True,
}
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import (
ForkProfileParams,
CreateSessionParams,
CreateSessionProfile,
)
# Fork profile A into a new independent profile B
forked_profile = client.profiles.fork(
profile_a.id,
params=ForkProfileParams(name="profile-b"),
)
print(f"Forked profile: {forked_profile.id}")
# Use the forked profile in a new session
session = client.sessions.create(
params=CreateSessionParams(
profile=CreateSessionProfile(
id=forked_profile.id,
persist_changes=True,
)
)
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/profile/PROFILE_A_ID/fork \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"name": "profile-b"
}'
```
## Managing Profiles
### List Profiles
Retrieve all your profiles with optional pagination and filtering:
```typescript Node.js theme={null}
const result = await client.profiles.list({
page: 1,
name: "hackernews-profile",
});
console.log(`Found ${result.totalCount} profiles`);
result.profiles.forEach((profile) => {
console.log(`- ${profile.name}: ${profile.id}`);
});
```
```python Python 1.0+ theme={null}
result = client.profiles.list(
params={
"page": 1,
"name": "hackernews-profile",
}
)
print(f"Found {result.total_count} profiles")
for profile in result.profiles:
print(f"- {profile.name}: {profile.id}")
```
```python Python (legacy) theme={null}
from hyperbrowser.models import ProfileListParams
result = client.profiles.list(
params=ProfileListParams(
page=1,
name="hackernews-profile",
)
)
print(f"Found {result.total_count} profiles")
for profile in result.profiles:
print(f"- {profile.name}: {profile.id}")
```
```bash cURL theme={null}
curl -X GET "https://api.hyperbrowser.ai/api/profiles?page=1&name=hackernews-profile" \
-H "x-api-key: YOUR_API_KEY"
```
### Get Profile Details
Retrieve information about a specific profile:
```typescript Node.js theme={null}
const profile = await client.profiles.get("your-profile-id");
console.log("Profile ID:", profile.id);
console.log("Profile Name:", profile.name);
console.log("Created At:", profile.createdAt);
```
```python Python theme={null}
profile = client.profiles.get("your-profile-id")
print(f"Profile ID: {profile.id}")
print(f"Profile Name: {profile.name}")
print(f"Created At: {profile.created_at}")
```
```bash cURL theme={null}
curl -X GET https://api.hyperbrowser.ai/api/profile/profile-id \
-H "x-api-key: YOUR_API_KEY"
```
### Delete a Profile
Delete a profile when you no longer need it:
```typescript Node.js theme={null}
await client.profiles.delete("profile-id");
console.log("Profile deleted");
```
```python Python theme={null}
client.profiles.delete("profile-id")
print("Profile deleted")
```
```bash cURL theme={null}
curl -X DELETE https://api.hyperbrowser.ai/api/profile/profile-id \
-H "x-api-key: YOUR_API_KEY"
```
## Read-Only Profile Usage
Load a previously persisted profile without saving any changes made during this new session:
```typescript Node.js theme={null}
const session = await client.sessions.create({
profile: {
id: "your-profile-id",
persistChanges: false, // Don't save changes
},
});
// Use the session - changes will not be persisted
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
params={
"profile": {
"id": "your-profile-id",
"persist_changes": False, # Don't save changes
}
}
)
# Use the session - changes will not be persisted
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams, CreateSessionProfile
session = client.sessions.create(
params=CreateSessionParams(
profile=CreateSessionProfile(
id="your-profile-id",
persist_changes=False, # Don't save changes
)
)
)
# Use the session - changes will not be persisted
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"profile": {
"id": "your-profile-id",
"persistChanges": false
}
}'
```
**Use cases for read-only profiles:**
* Test workflows without affecting the saved state
* Run parallel sessions with the same profile safely
* Maintain a clean baseline profile for repeated use
By default, `persistChanges` is set to `false` so you don't have to explicitly specify it.
## Next Steps
Manage session lifecycle
Record session activity
Watch sessions in real-time
Load browser extensions
# Proxy Configuration
Source: https://hyperbrowser.ai/docs/sessions/proxy
Route your sessions through proxy servers for geo-targeting and IP rotation
Route browser sessions through proxy servers to access geo-restricted content, rotate IPs, and distribute requests across different locations.
Proxy features require a paid plan.
## Quick Start
Enable Hyperbrowser's managed proxy network with a single parameter:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const session = await client.sessions.create({
useProxy: true,
});
console.log("Session ID:", session.id);
console.log("WebSocket endpoint:", session.wsEndpoint);
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(
params={
"use_proxy": True,
}
)
print(f"Session ID: {session.id}")
print(f"WebSocket endpoint: {session.ws_endpoint}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(
params=CreateSessionParams(
use_proxy=True,
)
)
print(f"Session ID: {session.id}")
print(f"WebSocket endpoint: {session.ws_endpoint}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true
}'
```
## Update Proxy on a Running Session
You can update proxy settings on an active session without recreating it by calling `PUT /session/:id/update` with `type: "proxy"`.
* Set `enabled: false` to disable proxying for the current session.
* Set `enabled: true` with `staticIpId` to switch the session onto a static IP.
* Set `enabled: true` with an optional `location` object to use Hyperbrowser's managed proxy network.
* If `location.country` is omitted, Hyperbrowser defaults it to `US`.
* Custom proxy values cannot be set through this update route. If you need `proxyServer`, `proxyServerUsername`, or `proxyServerPassword`, create the session with those values instead.
* Once a session has used managed proxy or static IP, you can disable and re-enable that same proxy type, but you cannot switch between managed proxy and static IP within the same session.
### Managed Proxy Update
Use `location` to target a managed proxy location on an existing session.
```typescript Node.js theme={null}
await client.sessions.updateProxyParams(session.id, {
enabled: true,
location: {
country: "GB",
city: "london",
},
});
// If country is omitted, it defaults to US.
await client.sessions.updateProxyParams(session.id, {
enabled: true,
location: {
state: "CA",
},
});
```
```python Python 1.0+ theme={null}
client.sessions.update_proxy_params(
session.id,
{
"enabled": True,
"location": {
"country": "GB",
"city": "London",
},
},
)
# If country is omitted, it defaults to US.
client.sessions.update_proxy_params(
session.id,
{
"enabled": True,
"location": {
"state": "CA",
},
},
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import (
UpdateSessionProxyLocationParams,
UpdateSessionProxyParams,
)
client.sessions.update_proxy_params(
session.id,
UpdateSessionProxyParams(
enabled=True,
location=UpdateSessionProxyLocationParams(
country="GB",
city="London",
),
),
)
# If country is omitted, it defaults to US.
client.sessions.update_proxy_params(
session.id,
UpdateSessionProxyParams(
enabled=True,
location=UpdateSessionProxyLocationParams(
state="CA",
),
),
)
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/session/YOUR_SESSION_ID/update \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"type": "proxy",
"params": {
"enabled": true,
"location": {
"country": "GB",
"city": "London"
}
}
}'
curl -X PUT https://api.hyperbrowser.ai/api/session/YOUR_SESSION_ID/update \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"type": "proxy",
"params": {
"enabled": true,
"location": {
"state": "CA"
}
}
}'
```
### Static IP Update
Provide `staticIpId` to move an existing session onto one of your team's active static IPs.
```typescript Node.js theme={null}
await client.sessions.updateProxyParams(session.id, {
enabled: true,
staticIpId: "YOUR_STATIC_IP_ID",
});
```
```python Python 1.0+ theme={null}
client.sessions.update_proxy_params(
session.id,
{
"enabled": True,
"static_ip_id": "YOUR_STATIC_IP_ID",
},
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import UpdateSessionProxyParams
client.sessions.update_proxy_params(
session.id,
UpdateSessionProxyParams(
enabled=True,
static_ip_id="YOUR_STATIC_IP_ID",
),
)
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/session/YOUR_SESSION_ID/update \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"type": "proxy",
"params": {
"enabled": true,
"staticIpId": "YOUR_STATIC_IP_ID"
}
}'
```
### Disable Proxy
Disable proxying for the active session without ending the session itself.
```typescript Node.js theme={null}
await client.sessions.updateProxyParams(session.id, {
enabled: false,
});
```
```python Python 1.0+ theme={null}
client.sessions.update_proxy_params(
session.id,
{
"enabled": False,
},
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import UpdateSessionProxyParams
client.sessions.update_proxy_params(
session.id,
UpdateSessionProxyParams(
enabled=False,
),
)
```
```bash cURL theme={null}
curl -X PUT https://api.hyperbrowser.ai/api/session/YOUR_SESSION_ID/update \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"type": "proxy",
"params": {
"enabled": false
}
}'
```
## Country-Level Targeting
Target specific countries using ISO 3166-1 alpha-2 country codes. Hyperbrowser supports 100+ countries including:
| Region | Countries |
| - | - |
| **North America** | `US`, `CA`, `MX` |
| **Europe** | `GB`, `DE`, `FR`, `ES`, `IT`, `NL`, `SE`, `CH` |
| **Asia Pacific** | `JP`, `AU`, `SG`, `IN`, `KR`, `HK`, `CN` |
| **South America** | `BR`, `AR`, `CL` |
| **Middle East** | `AE`, `SA`, `IL` |
```typescript Node.js theme={null}
// Target specific countries
const usSession = await client.sessions.create({
useProxy: true,
proxyCountry: "US",
});
const gbSession = await client.sessions.create({
useProxy: true,
proxyCountry: "GB",
});
const jpSession = await client.sessions.create({
useProxy: true,
proxyCountry: "JP",
});
```
```python Python 1.0+ theme={null}
# Target specific countries
us_session = client.sessions.create(params={"use_proxy": True, "proxy_country": "US"})
gb_session = client.sessions.create(params={"use_proxy": True, "proxy_country": "GB"})
jp_session = client.sessions.create(params={"use_proxy": True, "proxy_country": "JP"})
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
# Target specific countries
us_session = client.sessions.create(
params=CreateSessionParams(use_proxy=True, proxy_country="US")
)
gb_session = client.sessions.create(
params=CreateSessionParams(use_proxy=True, proxy_country="GB")
)
jp_session = client.sessions.create(
params=CreateSessionParams(use_proxy=True, proxy_country="JP")
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"proxyCountry": "GB"
}'
```
## US State-Level Targeting
For US proxies, target specific states for more precise geo-targeting:
```typescript Node.js theme={null}
// Target California
const caSession = await client.sessions.create({
useProxy: true,
proxyCountry: "US",
proxyState: "CA",
});
// Target New York
const nySession = await client.sessions.create({
useProxy: true,
proxyCountry: "US",
proxyState: "NY",
});
// Target Texas
const txSession = await client.sessions.create({
useProxy: true,
proxyCountry: "US",
proxyState: "TX",
});
```
```python Python 1.0+ theme={null}
# Target California
ca_session = client.sessions.create(
params={"use_proxy": True, "proxy_country": "US", "proxy_state": "CA"}
)
# Target New York
ny_session = client.sessions.create(
params={"use_proxy": True, "proxy_country": "US", "proxy_state": "NY"}
)
# Target Texas
tx_session = client.sessions.create(
params={"use_proxy": True, "proxy_country": "US", "proxy_state": "TX"}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
# Target California
ca_session = client.sessions.create(
params=CreateSessionParams(use_proxy=True, proxy_country="US", proxy_state="CA")
)
# Target New York
ny_session = client.sessions.create(
params=CreateSessionParams(use_proxy=True, proxy_country="US", proxy_state="NY")
)
# Target Texas
tx_session = client.sessions.create(
params=CreateSessionParams(use_proxy=True, proxy_country="US", proxy_state="TX")
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"proxyCountry": "US",
"proxyState": "CA"
}'
```
State codes use the standard two-letter US state abbreviations (AL, AK, AZ,
AR, CA, etc.)
## City-Level Targeting
Target specific cities for ultra-precise geo-location. Common cities include New York, Los Angeles, Chicago, London, Tokyo, and more.
```typescript Node.js theme={null}
// Target New York City
const nycSession = await client.sessions.create({
useProxy: true,
proxyCountry: "US",
proxyCity: "New York",
});
// Target London
const londonSession = await client.sessions.create({
useProxy: true,
proxyCountry: "GB",
proxyCity: "London",
});
```
```python Python 1.0+ theme={null}
# Target New York City
nyc_session = client.sessions.create(
params={"use_proxy": True, "proxy_country": "US", "proxy_city": "New York"}
)
# Target London
london_session = client.sessions.create(
params={"use_proxy": True, "proxy_country": "GB", "proxy_city": "London"}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
# Target New York City
nyc_session = client.sessions.create(
params=CreateSessionParams(
use_proxy=True, proxy_country="US", proxy_city="New York"
)
)
# Target London
london_session = client.sessions.create(
params=CreateSessionParams(use_proxy=True, proxy_country="GB", proxy_city="London")
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"proxyCountry": "US",
"proxyCity": "New York"
}'
```
`proxyCity` and `proxyState` are mutually exclusive. Use one or the other, not
both.
While Hyperbrowser supports nearly all countries and a lot of cities, not all areas are guaranteed to be available. Especially for areas with low population density, separate testing should be done to ensure that adequate coverage is available.
## Custom Proxy Servers
Custom proxy servers are only available on the Enterprise plan. Reach out to us at [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) to learn more.
Use your own proxy infrastructure by providing server credentials. Supports HTTP, HTTPS, SOCKS5, SOCKS5h proxies.
```typescript Node.js theme={null}
const session = await client.sessions.create({
useProxy: true,
proxyServer: "scheme://proxy.example.com:8080",
proxyServerUsername: "proxy-username",
proxyServerPassword: "proxy-password",
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
params={
"use_proxy": True,
"proxy_server": "scheme://proxy.example.com:8080",
"proxy_server_username": "proxy-username",
"proxy_server_password": "proxy-password",
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
session = client.sessions.create(
params=CreateSessionParams(
use_proxy=True,
proxy_server="scheme://proxy.example.com:8080",
proxy_server_username="proxy-username",
proxy_server_password="proxy-password",
)
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"proxyServer": "scheme://proxy.example.com:8080",
"proxyServerUsername": "proxy-username",
"proxyServerPassword": "proxy-password"
}'
```
When setting the `proxyServer`, ensure that you start it with the scheme like `http://` for example.
## Best Practices
Use proxies in the same region as your target content for better performance and to avoid geo-blocking.
Always enable `useUltraStealth` or `useStealth` when using proxies to maximize
detection evasion.
Verify your proxy configuration works with target sites before running large-scale operations.
## Related Features
Advanced bot detection evasion
Use dedicated IP addresses
Persist cookies and local storage
Watch sessions in real-time
# Connect with Puppeteer
Source: https://hyperbrowser.ai/docs/sessions/puppeteer
Control Hyperbrowser sessions using Puppeteer
Puppeteer is a popular Node.js library for browser automation that integrates seamlessly with Hyperbrowser sessions. This guide shows you how to connect Puppeteer to your cloud browsers.
Puppeteer is a Node.js integration. Python clients should use the
[Playwright integration](/docs/sessions/playwright); Pyppeteer's WebSocket
dependency is incompatible with the Hyperbrowser Python SDK.
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk puppeteer-core dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk puppeteer-core dotenv
```
For Node.js, use `puppeteer-core` instead of `puppeteer` since Hyperbrowser provides the
browser. This saves disk space and installation time.
## Basic Connection
Connect Puppeteer to a Hyperbrowser session:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { connect } from "puppeteer-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
// Create session
const session = await client.sessions.create();
try {
// Connect Puppeteer
const browser = await connect({
browserWSEndpoint: session.wsEndpoint,
defaultViewport: null,
});
const defaultContext = browser.defaultBrowserContext();
// Get or create a page
const pages = await defaultContext.pages();
const page = pages[0];
// Navigate and interact
await page.goto("https://example.com");
console.log("Title:", await page.title());
} catch (err) {
console.error("Encountered error:", err);
} finally {
// Clean up
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
## Next Steps
Connect with Playwright
Manage sessions
Configure anti-detection
Persist browser state
# Recordings
Source: https://hyperbrowser.ai/docs/sessions/recordings
Record and replay your browser sessions with web recordings and video replays
Record and replay your browser sessions to debug failures, analyze behavior, and share reproducible bug reports. Hyperbrowser supports both web recordings (using [rrweb](https://www.rrweb.io/)) that capture DOM changes and interactions in a lightweight format, and traditional MP4 video recordings for easy sharing.
## Recording Types
Hyperbrowser supports two types of recordings:
* **Web Recording (rrweb)**: Captures DOM changes, interactions, and network requests in a lightweight JSON format that can be replayed with the rrweb player
* **Video Recording (MP4)**: Records the session as a standard video file for easy sharing and viewing
## Enabling Session Recording
To record a session, set `enableWebRecording` to `true` when creating a new session. This will record all browser interactions, DOM changes, and network requests for the duration of the session.
The rrweb session recording is enabled by default, so you don't need to explicitly set the `enableWebRecording`/`enable_web_recording` parameter to `true`/`True`.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const session = await client.sessions.create({
enableWebRecording: true,
});
console.log(`Session ID: ${session.id}`);
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(params={"enable_web_recording": True})
print(f"Session ID: {session.id}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(params=CreateSessionParams(enable_web_recording=True))
print(f"Session ID: {session.id}")
```
```bash cURL theme={null}
curl -X POST 'https://api.hyperbrowser.ai/api/session' \
-H 'x-api-key: YOUR_API_KEY' \
-H 'Content-Type: application/json' \
-d '{
"enableWebRecording": true
}'
```
## Enabling Video Recording
To create a video screen recording (MP4 format), set both `enableWebRecording` and `enableVideoWebRecording` to `true`. Both must be set to `true`/`True`.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const session = await client.sessions.create({
enableWebRecording: true,
enableVideoWebRecording: true,
});
console.log(`Session ID: ${session.id}`);
```
```python Python 1.0+ theme={null}
import os
from hyperbrowser import Hyperbrowser
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(
params={"enable_web_recording": True, "enable_video_web_recording": True}
)
print(f"Session ID: {session.id}")
```
```python Python (legacy) theme={null}
import os
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(
params=CreateSessionParams(
enable_web_recording=True, enable_video_web_recording=True
)
)
print(f"Session ID: {session.id}")
```
```bash cURL theme={null}
curl -X POST 'https://api.hyperbrowser.ai/api/session' \
-H 'x-api-key: YOUR_API_KEY' \
-H 'Content-Type: application/json' \
-d '{
"enableWebRecording": true,
"enableVideoWebRecording": true
}'
```
`enableWebRecording` must be `true` for video recording to work.
## Retrieving Recordings
To retrieve a recording, you need to:
1. Note the session `id` when you create the session
2. Ensure the session has been stopped before fetching the recording
3. Poll the recording URL endpoint until the status is `completed` or `failed`
### Get Web Recording URL
Retrieve the URL for the rrweb recording:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
// Poll until recording is ready
let recordingData = await client.sessions.getRecordingURL("session-id");
while (recordingData.status === "pending" || recordingData.status === "in_progress") {
console.log(`Recording status: ${recordingData.status}, waiting...`);
await new Promise(resolve => setTimeout(resolve, 1000));
recordingData = await client.sessions.getRecordingURL("session-id");
}
console.log(recordingData.status); // "not_enabled" | "completed" | "failed"
if (recordingData.status === "completed" && recordingData.recordingUrl) {
// Fetch the recording data
const response = await fetch(recordingData.recordingUrl);
const recordingEvents = await response.json();
console.log("Recording events:", recordingEvents);
} else if (recordingData.status === "failed") {
console.error("Recording failed:", recordingData.error);
}
```
```python Python theme={null}
import time
import requests
from hyperbrowser import Hyperbrowser
import os
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
# Poll until recording is ready
recording_data = client.sessions.get_recording_url("session-id")
while recording_data.status in ["pending", "in_progress"]:
print(f"Recording status: {recording_data.status}, waiting...")
time.sleep(1)
recording_data = client.sessions.get_recording_url("session-id")
print(f"Final status: {recording_data.status}")
if recording_data.status == "completed" and recording_data.recording_url:
response = requests.get(recording_data.recording_url)
recording_events = response.json()
print("Recording events:", recording_events)
elif recording_data.status == "failed":
print(f"Recording failed: {recording_data.error}")
```
```bash cURL theme={null}
curl -X GET 'https://api.hyperbrowser.ai/api/session/{sessionId}/recording-url' \
-H 'x-api-key: YOUR_API_KEY'
```
### Get Video Recording URL
Retrieve the URL for the video recording (MP4):
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
// Poll until video recording is ready
let recordingData = await client.sessions.getVideoRecordingURL("session-id");
while (recordingData.status === "pending" || recordingData.status === "in_progress") {
console.log(`Video recording status: ${recordingData.status}, waiting...`);
await new Promise(resolve => setTimeout(resolve, 1000));
recordingData = await client.sessions.getVideoRecordingURL("session-id");
}
console.log(recordingData.status); // "not_enabled" | "completed" | "failed"
if (recordingData.status === "completed" && recordingData.recordingUrl) {
console.log(`Download video: ${recordingData.recordingUrl}`);
} else if (recordingData.status === "failed") {
console.error("Video recording failed:", recordingData.error);
}
```
```python Python theme={null}
import time
from hyperbrowser import Hyperbrowser
import os
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
# Poll until video recording is ready
recording_data = client.sessions.get_video_recording_url("session-id")
while recording_data.status in ["pending", "in_progress"]:
print(f"Video recording status: {recording_data.status}, waiting...")
time.sleep(1)
recording_data = client.sessions.get_video_recording_url("session-id")
print(f"Final status: {recording_data.status}")
if recording_data.status == "completed" and recording_data.recording_url:
print(f"Download video: {recording_data.recording_url}")
elif recording_data.status == "failed":
print(f"Video recording failed: {recording_data.error}")
```
```bash cURL theme={null}
curl -X GET 'https://api.hyperbrowser.ai/api/session/{sessionId}/video-recording-url' \
-H 'x-api-key: YOUR_API_KEY'
```
### Response Format
Both recording endpoints return the same response structure:
```json theme={null}
{
"status": "completed",
"recordingUrl": "https://...",
"error": null
}
```
**Status values:**
* `not_enabled` - Recording was not enabled for this session
* `pending` - Recording is queued for processing
* `in_progress` - Recording is being processed
* `completed` - Recording is ready (check `recordingUrl`)
* `failed` - Recording failed (check `error` for details)
## Replaying rrweb Recordings
### Using the rrweb Player
Once you have the recording data, you can replay it using rrweb's player:
```html theme={null}
```
This launches an interactive player UI that allows you to play, pause, rewind, and inspect the recorded session.
### Building a Custom Player
You can also use rrweb's APIs to build your own playback UI. Refer to the [rrweb documentation](https://github.com/rrweb-io/rrweb/blob/master/guide.md#replay) for details on customizing the Replayer.
## Complete Example
Here's a full workflow that creates a session, performs automation, and retrieves the recording:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function sleep(ms: number) {
return new Promise((resolve) => setTimeout(resolve, ms));
}
async function runWithRecording() {
// Create session with recording enabled
const session = await client.sessions.create({
enableWebRecording: true,
enableVideoWebRecording: true,
});
console.log(`Session created: ${session.id}`);
try {
// Connect and run automation
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
await page.goto("https://example.com");
await page.click("a");
await page.goto("https://hackernews.com");
} catch (error) {
console.error("Error during automation:", error);
} finally {
// Stop the session
await client.sessions.stop(session.id);
}
// Poll for web recording
console.log("Waiting for web recording to be processed...");
let webRecording = await client.sessions.getRecordingURL(session.id);
while (
webRecording.status === "pending" ||
webRecording.status === "in_progress"
) {
await sleep(1000);
webRecording = await client.sessions.getRecordingURL(session.id);
}
if (webRecording.status === "completed") {
console.log(`Web recording: ${webRecording.recordingUrl}`);
} else if (webRecording.status === "failed") {
console.error("Web recording failed:", webRecording.error);
}
// Poll for video recording
console.log("Waiting for video recording to be processed...");
let videoRecording = await client.sessions.getVideoRecordingURL(session.id);
while (
videoRecording.status === "pending" ||
videoRecording.status === "in_progress"
) {
await sleep(1000);
videoRecording = await client.sessions.getVideoRecordingURL(session.id);
}
if (videoRecording.status === "completed") {
console.log(`Video recording: ${videoRecording.recordingUrl}`);
} else if (videoRecording.status === "failed") {
console.error("Video recording failed:", videoRecording.error);
}
}
runWithRecording();
```
```python Python 1.0+ theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from playwright.async_api import async_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = AsyncHyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
async def run_with_recording():
# Create session with recording enabled
session = await client.sessions.create(
params={
"enable_web_recording": True,
"enable_video_web_recording": True,
}
)
print(f"Session created: {session.id}")
try:
# Connect and run automation
async with async_playwright() as p:
browser = await p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
await page.goto("https://example.com")
await page.click("a")
await page.goto("https://hackernews.com")
except Exception as error:
print(f"Error during automation: {error}")
finally:
# Stop the session
await client.sessions.stop(session.id)
# Poll for web recording
print("Waiting for web recording to be processed...")
web_recording = await client.sessions.get_recording_url(session.id)
while web_recording.status in ["pending", "in_progress"]:
await asyncio.sleep(1)
web_recording = await client.sessions.get_recording_url(session.id)
if web_recording.status == "completed":
print(f"Web recording: {web_recording.recording_url}")
elif web_recording.status == "failed":
print(f"Web recording failed: {web_recording.error}")
# Poll for video recording
print("Waiting for video recording to be processed...")
video_recording = await client.sessions.get_video_recording_url(session.id)
while video_recording.status in ["pending", "in_progress"]:
await asyncio.sleep(1)
video_recording = await client.sessions.get_video_recording_url(session.id)
if video_recording.status == "completed":
print(f"Video recording: {video_recording.recording_url}")
elif video_recording.status == "failed":
print(f"Video recording failed: {video_recording.error}")
if __name__ == "__main__":
asyncio.run(run_with_recording())
```
```python Python (legacy) theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from hyperbrowser.models import CreateSessionParams
from playwright.async_api import async_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = AsyncHyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
async def run_with_recording():
# Create session with recording enabled
session = await client.sessions.create(
params=CreateSessionParams(
enable_web_recording=True,
enable_video_web_recording=True,
)
)
print(f"Session created: {session.id}")
try:
# Connect and run automation
async with async_playwright() as p:
browser = await p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
await page.goto("https://example.com")
await page.click("a")
await page.goto("https://hackernews.com")
except Exception as error:
print(f"Error during automation: {error}")
finally:
# Stop the session
await client.sessions.stop(session.id)
# Poll for web recording
print("Waiting for web recording to be processed...")
web_recording = await client.sessions.get_recording_url(session.id)
while web_recording.status in ["pending", "in_progress"]:
await asyncio.sleep(1)
web_recording = await client.sessions.get_recording_url(session.id)
if web_recording.status == "completed":
print(f"Web recording: {web_recording.recording_url}")
elif web_recording.status == "failed":
print(f"Web recording failed: {web_recording.error}")
# Poll for video recording
print("Waiting for video recording to be processed...")
video_recording = await client.sessions.get_video_recording_url(session.id)
while video_recording.status in ["pending", "in_progress"]:
await asyncio.sleep(1)
video_recording = await client.sessions.get_video_recording_url(session.id)
if video_recording.status == "completed":
print(f"Video recording: {video_recording.recording_url}")
elif video_recording.status == "failed":
print(f"Video recording failed: {video_recording.error}")
if __name__ == "__main__":
asyncio.run(run_with_recording())
```
## Storage and Retention
Session recordings are stored securely in Hyperbrowser's cloud infrastructure. Recordings are retained according to your plan's data retention policy.
## Limitations
* Session recordings capture only the visual state of the page. They do not include server-side logs, database changes, or other non-DOM modifications.
* Recordings may not perfectly reproduce complex WebGL or canvas-based animations.
## Best Practices
1. **Always enable recordings for debugging**: Recordings are invaluable when troubleshooting automation failures
2. **Poll for completion**: After stopping a session, poll the recording URL endpoint until the status is `completed`
3. **Handle failures gracefully**: Check the `error` field if the status is `failed`
4. **Use video recordings for sharing**: MP4 videos are easier to share with non-technical stakeholders
## Next Steps
Watch sessions in real-time as they run
Save and retrieve files from sessions
Track session events and lifecycle
View all your session recordings
# Multi-Region Support
Source: https://hyperbrowser.ai/docs/sessions/regions
Deploy browser sessions across multiple geographic regions for optimal performance and compliance
Hyperbrowser supports deploying browser sessions across multiple geographic regions worldwide. This allows you to optimize for latency and improve performance for your users.
## Available Regions
Hyperbrowser currently supports the following regions:
| Region Code | Location | Description |
| - | - | - |
| `us-central` | United States (Central) | Optimal for North American users |
| `us-east` | United States (East) | East coast US region |
| `us-west` | United States (West) | West coast US region |
| `europe-west` | Europe (West) | Optimal for European users |
| `asia-south` | Asia (South) | Optimal for Asian users. Available on Enterprise and Select plans. |
The `asia-south` region is only available on Enterprise and Select plans. [Get in touch](https://calendly.com/shri-hyperbrowser/demo) to learn more.
## Why Use Multi-Region Support?
**Reduced Latency**: Deploy sessions closer to your target websites or end users to minimize network latency and improve response times.
**Geographic Testing**: Test how websites behave from different geographic locations, useful for geo-restricted content or region-specific features.
**Load Distribution**: Distribute workloads across multiple regions for better scalability and reliability.
## Quick Start
Specify the region when creating a session:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function main() {
// Create session in Europe
const session = await client.sessions.create({
region: "europe-west",
});
console.log("Session ID:", session.id);
console.log("WebSocket:", session.wsEndpoint);
}
main();
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
# Create session in Europe
session = client.sessions.create(params={"region": "europe-west"})
print(f"Session ID: {session.id}")
print(f"WebSocket: {session.ws_endpoint}")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
import os
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
# Create session in Europe
session = client.sessions.create(params=CreateSessionParams(region="europe-west"))
print(f"Session ID: {session.id}")
print(f"WebSocket: {session.ws_endpoint}")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"region": "europe-west"
}'
```
## Region Selection Best Practices
### Combine with Proxy Settings
You can also combine region selection with proxy settings from the same geographic area:
```typescript Node.js theme={null}
// European session with European proxy
const session = await client.sessions.create({
region: "europe-west",
useProxy: true,
proxyCountry: "DE", // Germany
useStealth: true,
});
```
```python Python 1.0+ theme={null}
# European session with European proxy
session = client.sessions.create(
params={
"region": "europe-west",
"use_proxy": True,
"proxy_country": "DE", # Germany
"use_stealth": True,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
# European session with European proxy
session = client.sessions.create(
params=CreateSessionParams(
region="europe-west",
use_proxy=True,
proxy_country="DE", # Germany
use_stealth=True,
)
)
```
Matching your session region with your proxy country can reduce latency and make your traffic patterns appear more natural.
## Troubleshooting
### High Latency
If you're experiencing high latency:
1. **Check region selection**: Ensure you're using the region closest to your target
2. **Verify network connectivity**: Test your connection to the region
3. **Consider proxy settings**: Proxy routing can add latency
### Session Creation Timeout
If session creation times out in a specific region:
1. **Try a different region**: Some regions may be experiencing high load
2. **Check region status**: Contact support for region availability
3. **Increase timeout**: Adjust your client timeout settings
## API Reference
For complete API details on the `region` parameter, see:
* [Create Session API Reference](/docs/api-reference/create-new-session)
* [Session Parameters Documentation](/docs/sessions/parameters)
## Next Steps
Combine regions with proxy settings for optimal performance
Explore all available session configuration options
Learn about anti-detection features
Use static IPs with regional sessions
# Connect with Selenium
Source: https://hyperbrowser.ai/docs/sessions/selenium
Control Hyperbrowser sessions using Selenium
Selenium is a powerful browser automation framework that works seamlessly with Hyperbrowser sessions. This guide shows you how to connect Selenium to your cloud browser sessions.
## Installation
First, install Selenium:
```bash npm theme={null}
npm install @hyperbrowser/sdk selenium-webdriver dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk selenium-webdriver dotenv
```
```bash pip theme={null}
pip install hyperbrowser selenium python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser selenium python-dotenv
```
## Basic Connection
Connect Selenium to a Hyperbrowser session:
```typescript Node.js theme={null}
import dotenv from "dotenv";
import https from "https";
import { Builder, WebDriver } from "selenium-webdriver";
import fs from "fs";
import { Options } from "selenium-webdriver/chrome";
import { Hyperbrowser } from "@hyperbrowser/sdk";
// Load environment variables from .env file
dotenv.config();
const client = new Hyperbrowser({ apiKey: process.env.HYPERBROWSER_API_KEY });
async function main() {
const session = await client.sessions.create();
if (!session.webdriverEndpoint) {
await client.sessions.stop(session.id);
throw new Error("No webdriver endpoint found");
}
const customHttpsAgent = new https.Agent({});
(customHttpsAgent as any).addRequest = (req: any, options: any) => {
req.setHeader("x-hyperbrowser-token", session.token);
(https.Agent.prototype as any).addRequest.call(
customHttpsAgent,
req,
options
);
};
const driver: WebDriver = await new Builder()
.forBrowser("chrome")
.usingHttpAgent(customHttpsAgent)
.usingServer(session.webdriverEndpoint)
.setChromeOptions(new Options())
.build();
try {
// Navigate to a URL
await driver.get("https://www.google.com");
console.log("Navigated to Google");
// Search
const searchBox = await driver.findElement({ name: "q" });
await searchBox.sendKeys("Selenium WebDriver");
await searchBox.submit();
console.log("Performed search");
// Screenshot
await driver.takeScreenshot().then((data) => {
fs.writeFileSync("search_results.png", data, "base64");
});
console.log("Screenshot saved");
} finally {
await driver.quit();
await client.sessions.stop(session.id);
}
}
main().catch(console.error);
```
```python Python theme={null}
import os
from dotenv import load_dotenv
from selenium import webdriver
from selenium.webdriver.remote.client_config import ClientConfig
from selenium.webdriver.remote.remote_connection import RemoteConnection
from selenium.webdriver.chrome.options import Options
from hyperbrowser import Hyperbrowser
# Load environment variables from .env file
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class CustomRC(RemoteConnection):
_signing_key = None
def __init__(self, server: str, token: str):
super().__init__(
client_config=ClientConfig(
remote_server_addr=server,
extra_headers={"x-hyperbrowser-token": token},
)
)
def main():
session = client.sessions.create()
driver = None
try:
custom_conn = CustomRC(session.webdriver_endpoint, session.token)
driver = webdriver.Remote(custom_conn, options=Options())
# Navigate to a URL
driver.get("https://www.google.com")
print("Navigated to Google")
# Search
search_box = driver.find_element("name", "q")
search_box.send_keys("Selenium WebDriver")
search_box.submit()
print("Performed search")
# Screenshot
driver.save_screenshot("search_results.png")
print("Screenshot saved")
except Exception as e:
print(f"Error: {e}")
finally:
if driver is not None:
driver.quit()
client.sessions.stop(session.id)
if __name__ == "__main__":
main()
```
Please contact us at [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) to get access to using Selenium with Hyperbrowser.
## Next Steps
Connect with Puppeteer
Connect with Playwright
Configure anti-detection
Persist browser state
# Static IPs
Source: https://hyperbrowser.ai/docs/sessions/static-ips
Use dedicated IP addresses for your sessions
Static IPs give you dedicated IP addresses that stay consistent across sessions. Perfect for API whitelisting, maintaining identity with authenticated sessions, or building reputation without the noise from shared IPs.
## Why Use Static IPs?
Standard proxy rotation gives you a different IP each time. Static IPs give you:
* **Consistency** - Same IP across all your sessions
* **Dedicated** - Your IP, not shared with anyone else
* **Whitelist-ready** - Perfect for APIs and services that only allow known IPs
* **Clean reputation** - Build trust without worrying about what others did with that IP
* **Compliance** - Meet security requirements for fixed source IPs
## Use Cases
### 1. API Access with IP Whitelisting
Many enterprise APIs only accept requests from whitelisted IPs. Here's how to use your static IP:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function accessWhitelistedAPI() {
const session = await client.sessions.create({
useProxy: true, // Make sure to set this to true
staticIpId: "your-static-ip-id", // Your assigned static IP
});
try {
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
// Access API that requires whitelisted IP
await page.goto("https://example.com");
const data = await page.content();
console.log("API Response:", data);
} catch (err) {
console.error("Error accessing whitelisted API:", err);
} finally {
await client.sessions.stop(session.id);
}
}
accessWhitelistedAPI();
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
def access_whitelisted_api():
session = client.sessions.create(
params={
"use_proxy": True, # Make sure to set this to True
"static_ip_id": "your-static-ip-id", # Your assigned static IP
}
)
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
# Access API that requires whitelisted IP
page.goto("https://example.com")
data = page.content()
print(f"API Response: {data}")
except Exception as e:
print(f"Error accessing whitelisted API: {e}")
finally:
client.sessions.stop(session.id)
access_whitelisted_api()
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
def access_whitelisted_api():
session = client.sessions.create(
params=CreateSessionParams(
use_proxy=True, # Make sure to set this to True
static_ip_id="your-static-ip-id", # Your assigned static IP
)
)
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
# Access API that requires whitelisted IP
page.goto("https://example.com")
data = page.content()
print(f"API Response: {data}")
except Exception as e:
print(f"Error accessing whitelisted API: {e}")
finally:
client.sessions.stop(session.id)
access_whitelisted_api()
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"staticIpId": "your-static-ip-id"
}'
```
It is important that you have `useProxy`/`use_proxy` set to `true`/`True` along with your `staticIpId`/`static_ip_id` in your session params when using Static IPs.
### 2. Consistent Identity for Authenticated Sessions
Combine static IPs with profiles to maintain consistent identity across sessions:
```typescript Node.js theme={null}
async function maintainAuthenticatedSession(profileId, staticIpId) {
const session = await client.sessions.create({
useProxy: true, // Make sure to set this to true
staticIpId: staticIpId,
profile: {
id: profileId,
persistChanges: true,
},
});
try {
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
// Access authenticated content
// Same IP + saved cookies = consistent identity
await page.goto("https://example.com/");
} catch (err) {
console.error("Error accessing authenticated session:", err);
} finally {
await client.sessions.stop(session.id);
}
}
```
```python Python 1.0+ theme={null}
def maintain_authenticated_session(profile_id: str, static_ip_id: str):
session = client.sessions.create(
params={
"use_proxy": True, # Make sure to set this to true
"static_ip_id": static_ip_id,
"profile": {"id": profile_id, "persist_changes": True},
}
)
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
# Access authenticated content
# Same IP + saved cookies = consistent identity
page.goto("https://example.com/")
except Exception as e:
print(f"Error accessing authenticated session: {e}")
finally:
client.sessions.stop(session.id)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams, CreateSessionProfile
def maintain_authenticated_session(profile_id: str, static_ip_id: str):
session = client.sessions.create(
params=CreateSessionParams(
use_proxy=True, # Make sure to set this to true
static_ip_id=static_ip_id,
profile=CreateSessionProfile(
id=profile_id,
persist_changes=True,
),
)
)
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
# Access authenticated content
# Same IP + saved cookies = consistent identity
page.goto("https://example.com/")
except Exception as e:
print(f"Error accessing authenticated session: {e}")
finally:
client.sessions.stop(session.id)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"staticIpId": "your-static-ip-id",
"profile": {
"id": "your-profile-id",
"persistChanges": true
}
}'
```
### 3. Multi-Step Workflows
Some sites flag IP changes mid-session as suspicious. Static IPs keep you consistent:
```typescript Node.js theme={null}
async function performMultiStepWorkflow(staticIpId) {
const session = await client.sessions.create({
useProxy: true, // Make sure to set this to true
staticIpId: staticIpId,
});
try {
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const defaultContext = browser.contexts()[0];
const page = defaultContext.pages()[0];
// Multi-step process with consistent IP
await page.goto("https://example.com/step1");
// ... complete step 1
await page.goto("https://example.com/step2");
// ... complete step 2
await page.goto("https://example.com/step3");
// ... complete step 3
// All steps appear from the same IP
} catch (err) {
console.error("Error performing multi-step workflow:", err);
} finally {
await client.sessions.stop(session.id);
}
}
```
```python Python 1.0+ theme={null}
def perform_multi_step_workflow(static_ip_id: str):
session = client.sessions.create(
params={
"use_proxy": True, # Make sure to set this to true
"static_ip_id": static_ip_id,
}
)
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
# Multi-step process with consistent IP
page.goto("https://example.com/step1")
# ... complete step 1
page.goto("https://example.com/step2")
# ... complete step 2
page.goto("https://example.com/step3")
# ... complete step 3
# All steps appear from the same IP
except Exception as e:
print(f"Error performing multi-step workflow: {e}")
finally:
client.sessions.stop(session.id)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
def perform_multi_step_workflow(static_ip_id: str):
session = client.sessions.create(
params=CreateSessionParams(
use_proxy=True, # Make sure to set this to true
static_ip_id=static_ip_id,
)
)
try:
with sync_playwright() as p:
browser = p.chromium.connect_over_cdp(session.ws_endpoint)
default_context = browser.contexts[0]
page = default_context.pages[0]
# Multi-step process with consistent IP
page.goto("https://example.com/step1")
# ... complete step 1
page.goto("https://example.com/step2")
# ... complete step 2
page.goto("https://example.com/step3")
# ... complete step 3
# All steps appear from the same IP
except Exception as e:
print(f"Error performing multi-step workflow: {e}")
finally:
client.sessions.stop(session.id)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"staticIpId": "your-static-ip-id"
}'
```
## Verifying Your Static IP
Quick check to confirm your static IP is working:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
async function verifyStaticIP(staticIpId) {
const session1 = await client.sessions.create({
useProxy: true, // Make sure to set this to true
staticIpId: staticIpId,
});
const session2 = await client.sessions.create({
useProxy: true, // Make sure to set this to true
staticIpId: staticIpId,
});
try {
// Check IP from first session
const browser1 = await chromium.connectOverCDP(session1.wsEndpoint);
const defaultContext = browser1.contexts()[0];
const page1 = defaultContext.pages()[0];
await page1.goto("https://api.ipify.org?format=json");
const ip1 = await page1.evaluate(() => document.body.textContent);
// Check IP from second session
const browser2 = await chromium.connectOverCDP(session2.wsEndpoint);
const defaultContext2 = browser2.contexts()[0];
const page2 = defaultContext2.pages()[0];
await page2.goto("https://api.ipify.org?format=json");
const ip2 = await page2.evaluate(() => document.body.textContent);
console.log("Session 1 IP:", ip1); // Should be the same as the static IP address shown in the dashboard
console.log("Session 2 IP:", ip2); // Should be the same as the static IP address shown in the dashboard
console.log("IPs match:", ip1 === ip2); // Should be true
} catch (err) {
console.error("Error verifying static IP:", err);
} finally {
await client.sessions.stop(session1.id);
await client.sessions.stop(session2.id);
}
}
verifyStaticIP("your-static-ip-id");
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
def verify_static_ip(static_ip_id: str):
session1 = client.sessions.create(
params={"use_proxy": True, "static_ip_id": static_ip_id}
)
session2 = client.sessions.create(
params={"use_proxy": True, "static_ip_id": static_ip_id}
)
try:
with sync_playwright() as p:
# Check IP from first session
browser1 = p.chromium.connect_over_cdp(session1.ws_endpoint)
default_context = browser1.contexts[0]
page1 = default_context.pages[0]
page1.goto("https://api.ipify.org?format=json")
ip1 = page1.evaluate("() => document.body.textContent")
# Check IP from second session
browser2 = p.chromium.connect_over_cdp(session2.ws_endpoint)
default_context2 = browser2.contexts[0]
page2 = default_context2.pages[0]
page2.goto("https://api.ipify.org?format=json")
ip2 = page2.evaluate("() => document.body.textContent")
print(
f"Session 1 IP: {ip1}"
) # Should be the same as the static IP address shown in the dashboard
print(
f"Session 2 IP: {ip2}"
) # Should be the same as the static IP address shown in the dashboard
print(f"IPs match: {ip1 == ip2}") # Should be True
except Exception as e:
print(f"Error verifying static IP: {e}")
finally:
client.sessions.stop(session1.id)
client.sessions.stop(session2.id)
verify_static_ip("your-static-ip-id")
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import CreateSessionParams
from playwright.sync_api import sync_playwright
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
def verify_static_ip(static_ip_id: str):
session1 = client.sessions.create(
params=CreateSessionParams(use_proxy=True, static_ip_id=static_ip_id)
)
session2 = client.sessions.create(
params=CreateSessionParams(use_proxy=True, static_ip_id=static_ip_id)
)
try:
with sync_playwright() as p:
# Check IP from first session
browser1 = p.chromium.connect_over_cdp(session1.ws_endpoint)
default_context = browser1.contexts[0]
page1 = default_context.pages[0]
page1.goto("https://api.ipify.org?format=json")
ip1 = page1.evaluate("() => document.body.textContent")
# Check IP from second session
browser2 = p.chromium.connect_over_cdp(session2.ws_endpoint)
default_context2 = browser2.contexts[0]
page2 = default_context2.pages[0]
page2.goto("https://api.ipify.org?format=json")
ip2 = page2.evaluate("() => document.body.textContent")
print(
f"Session 1 IP: {ip1}"
) # Should be the same as the static IP address shown in the dashboard
print(
f"Session 2 IP: {ip2}"
) # Should be the same as the static IP address shown in the dashboard
print(f"IPs match: {ip1 == ip2}") # Should be True
except Exception as e:
print(f"Error verifying static IP: {e}")
finally:
client.sessions.stop(session1.id)
client.sessions.stop(session2.id)
verify_static_ip("your-static-ip-id")
```
```bash cURL theme={null}
# Create two sessions with the same static IP
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"staticIpId": "your-static-ip-id"
}'
# Both sessions will use the same IP address
# Connect to each liveUrl or wsEndpoint and check IP at https://api.ipify.org
```
## Getting Started with Static IPs
Purchase and allocate static IPs to your team
Follow these steps to get your own static IPs:
1. **Purchase a plan**
* On the Hyperbrowser dashboard, go the [Static IPs Tab](https://app.hyperbrowser.ai/settings?tab=static-ips) and choose a plan. Your plan determines how many static IPs you can allocate to your team.
2. **Allocate IPs to your team**
* On the same page, once you have purchased a plan, allocate available static IPs to your team.
* You can deactivate or delete an IP from your team at any time.
* Once a Static IP is deleted from your team, there is no guarantee that it will be available again.
3. **Copy your Static IP ID(s)**
* After allocation, copy the `ID` shown for the Static IP in the table. Use this ID in your session configuration.
* The table also contains the actual IP address of the Static IP.
4. **Manage limits and upgrades**
* Your total allocatable IPs are limited by your plan. Upgrade your plan to increase your static IP allocation.
* If you downgrade or cancel your plan, any Static IPs over your allocation limit will be randomly deactivated.
## Combining with Other Features
Static IPs work great with other Hyperbrowser capabilities:
```typescript Node.js theme={null}
const session = await client.sessions.create({
// Static IP
useProxy: true,
staticIpId: "your-static-ip-id",
// Anti-detection
useUltraStealth: true,
// Profile persistence
profile: {
id: profileId,
persistChanges: true,
},
// Privacy features
adblock: true,
trackers: true,
annoyances: true,
// Captcha solving
solveCaptchas: true,
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
params={
# Static IP
"use_proxy": True,
"static_ip_id": "your-static-ip-id",
# Anti-detection
"use_ultra_stealth": True,
# Profile persistence
"profile": {"id": profile_id, "persist_changes": True},
# Privacy features
"adblock": True,
"trackers": True,
"annoyances": True,
# Captcha solving
"solve_captchas": True,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams, CreateSessionProfile
session = client.sessions.create(
params=CreateSessionParams(
# Static IP
use_proxy=True,
static_ip_id="your-static-ip-id",
# Anti-detection
use_ultra_stealth=True,
# Profile persistence
profile=CreateSessionProfile(id=profile_id, persist_changes=True),
# Privacy features
adblock=True,
trackers=True,
annoyances=True,
# Captcha solving
solve_captchas=True,
)
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useProxy": true,
"staticIpId": "your-static-ip-id",
"useUltraStealth": true,
"profile": {
"id": "your-profile-id",
"persistChanges": true
},
"adblock": true,
"trackers": true,
"annoyances": true,
"solveCaptchas": true
}'
```
## Best Practices
Keep an eye on your static IP's reputation. If a target site starts blocking it, you may need to adjust your automation patterns or rotate to a different static IP.
Having a static IP doesn't mean you can hammer a site. Implement proper delays and rate limiting to stay under the radar.
Set up IP whitelisting well before you need it. Some services take days to process whitelist requests.
Keep notes on which static IPs you're using for which purposes. Makes troubleshooting and management much easier.
Always test with your static IP in a dev environment before going to production. Verify the IP is working as expected.
## Troubleshooting
**Symptoms:** Different IP on each session
**Fixes:**
* Check that `staticIpId` is set correctly in your session params
* Make sure that you have `useProxy`/`use_proxy` set to `true`/`True` in your session params
* Make sure you're using the correct static IP ID (not the actual IP address)
* Contact [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) to confirm your static IP pool is active
**Symptoms:** Can't access whitelisted services
**Fixes:**
* Double-check the service whitelisted the correct IP addresses
* Verify no additional firewall rules are blocking you
* Confirm the service completed their whitelist setup
* Test with a simple curl command first to isolate the issue
* Contact the service provider to verify whitelist status
**Symptoms:** Getting blocked or flagged by target sites
**Fixes:**
* Review your automation - are you being too aggressive?
* Add delays and proper rate limiting
* Check if you're respecting robots.txt
* Contact support to discuss IP rotation or getting a fresh IP
* Consider using multiple static IPs to distribute load
**Symptoms:** Don't know what to pass as `staticIpId`
**Fixes:**
* Check the [Static IPs Tab](https://app.hyperbrowser.ai/settings?tab=static-ips) in the dashboard
* The ID is a UUID format string, not the actual IP address
* Contact [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) if you can't locate it
If you have other Static IP needs, please contact us at [info@hyperbrowser.ai](mailto:info@hyperbrowser.ai).
## Next Steps
Learn about proxy options
Configure anti-detection
Persist browser state
Record and replay sessions
# Stealth Mode
Source: https://hyperbrowser.ai/docs/sessions/stealth
Configure anti-detection features for your browser sessions
Stealth mode applies anti-detection techniques to help your automated browser sessions bypass bot detection. Use it when interacting with sites that employ bot protection mechanisms.
## Basic Usage
Enable standard stealth mode when creating a session:
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const session = await client.sessions.create({
useStealth: true,
});
```
```python Python 1.0+ theme={null}
from hyperbrowser import Hyperbrowser
import os
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(params={"use_stealth": True})
```
```python Python (legacy) theme={null}
from hyperbrowser import Hyperbrowser
import os
from hyperbrowser.models import CreateSessionParams
from dotenv import load_dotenv
load_dotenv()
client = Hyperbrowser(api_key=os.environ["HYPERBROWSER_API_KEY"])
session = client.sessions.create(params=CreateSessionParams(use_stealth=True))
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useStealth": true
}'
```
## Ultra Stealth Mode
For more advanced bot detection evasion, use ultra stealth mode:
Ultra stealth is only available on enterprise plans. Contact us at
[info@hyperbrowser.ai](mailto:info@hyperbrowser.ai) to learn more.
```typescript Node.js theme={null}
const session = await client.sessions.create({
useUltraStealth: true,
useProxy: true,
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
params={
"use_ultra_stealth": True,
"use_proxy": True,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
session = client.sessions.create(
params=CreateSessionParams(
use_ultra_stealth=True,
use_proxy=True,
)
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useUltraStealth": true,
"useProxy": true
}'
```
## Combine with Proxies
Stealth mode is most effective when combined with proxies:
```typescript Node.js theme={null}
const session = await client.sessions.create({
useStealth: true,
useProxy: true,
});
```
```python Python 1.0+ theme={null}
session = client.sessions.create(
params={
"use_stealth": True,
"use_proxy": True,
}
)
```
```python Python (legacy) theme={null}
from hyperbrowser.models import CreateSessionParams
session = client.sessions.create(
params=CreateSessionParams(
use_stealth=True,
use_proxy=True,
)
)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/session \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"useStealth": true,
"useProxy": true
}'
```
## Tips
* **Use ultra stealth for challenging sites** - `useUltraStealth` / `use_ultra_stealth` provides the strongest protection
* **Combine with proxies** - Pair stealth with proxies for better results
## Related Features
Route sessions through proxies
Persist browser state across sessions
Use dedicated IP addresses
# File Uploads
Source: https://hyperbrowser.ai/docs/sessions/uploads
Learn how to upload files to your browser sessions
Hyperbrowser allows you to upload files to your browser sessions. This is useful for a variety of use cases, such as:
* Uploading a CSV file to a website to be used in a form
* Uploading a PDF file to a website to be used in a form
* Uploading a Word document to a website to be used in a form
You can upload files to your remote browser session via the Session Uploads API. The name of the file that is uploaded is what it will be saved as in the remote browser session and be stored in the `/tmp/uploads` directory.
**Self-Hosted Hyperbrowser**: For some instances of self-hosted Hyperbrowser, uploads will be available at `/tmp//uploads` on the host machine.
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { chromium } from "playwright-core";
import { config } from "dotenv";
// import fs from "fs";
config();
const hbClient = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms));
const main = async () => {
const session = await hbClient.sessions.create();
const liveUrl = session.liveUrl;
console.log("Live URL:", liveUrl);
const fileName = "FILENAME.EXTENSION";
try {
const result = await hbClient.sessions.uploadFile(session.id, {
fileInput: fileName,
});
// or
// const fileStream = fs.createReadStream(fileName);
// const result = await hbClient.sessions.uploadFile(session.id, {
// fileInput: fileStream,
// });
console.log("Upload success:", result);
const browser = await chromium.connectOverCDP(session.wsEndpoint);
const context = browser.contexts()[0];
const page = context.pages()[0];
await page.goto("https://browser-tests-alpha.vercel.app/api/upload-test");
const cdp = await context.newCDPSession(page);
const root = await cdp.send("DOM.getDocument");
const inputNode = await cdp.send("DOM.querySelector", {
nodeId: root["root"]["nodeId"],
selector: "#fileUpload",
});
const remoteFilePath = result.filePath;
if (!remoteFilePath) {
console.error("Remote file path not found");
return;
}
await cdp.send("DOM.setFileInputFiles", {
files: [remoteFilePath],
nodeId: inputNode["nodeId"],
});
await sleep(20_000);
} catch (err) {
console.error("Error uploading file:", err);
} finally {
await hbClient.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(err);
process.exit(1);
});
```
```python Python theme={null}
import asyncio
from hyperbrowser import AsyncHyperbrowser
from playwright.async_api import async_playwright
from dotenv import load_dotenv
import os
load_dotenv()
hb_client = AsyncHyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
async def main():
session = await hb_client.sessions.create()
live_url = session.live_url
print("Live URL:", live_url)
file_name = "FILENAME.EXTENSION"
try:
result = await hb_client.sessions.upload_file(session.id, file_name)
# or upload the file itself
# with open(file_name, "rb") as f:
# result = await hb_client.sessions.upload_file(session.id, f)
print("Upload success:", result)
async with async_playwright() as p:
browser = await p.chromium.connect_over_cdp(session.ws_endpoint)
context = browser.contexts[0]
page = context.pages[0]
# Navigate to a website
await page.goto("https://browser-tests-alpha.vercel.app/api/upload-test")
cdp = await context.new_cdp_session(page)
root = await cdp.send("DOM.getDocument")
input_node = await cdp.send(
"DOM.querySelector",
{"nodeId": root["root"]["nodeId"], "selector": "#fileUpload"},
)
remote_file_path = result.file_path
if not remote_file_path:
print("Remote file path not found")
return
await cdp.send(
"DOM.setFileInputFiles",
{"files": [remote_file_path], "nodeId": input_node["nodeId"]},
)
await asyncio.sleep(20)
except Exception as e:
print(f"Error: {e}")
finally:
await hb_client.sessions.stop(session.id)
if __name__ == "__main__":
asyncio.run(main())
```
```typescript Node.js theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { connect } from "puppeteer-core";
import { config } from "dotenv";
// import fs from "fs";
config();
const hbClient = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms));
const main = async () => {
const session = await hbClient.sessions.create();
const liveUrl = session.liveUrl;
console.log("Live URL:", liveUrl);
const fileName = "FILENAME.EXTENSION";
try {
const result = await hbClient.sessions.uploadFile(session.id, {
fileInput: fileName,
});
// or
// const fileStream = fs.createReadStream(fileName);
// const result = await hbClient.sessions.uploadFile(session.id, {
// fileInput: fileStream,
// });
console.log("Upload success:", result);
const browser = await connect({ browserWSEndpoint: session.wsEndpoint });
const defaultContext = browser.defaultBrowserContext();
const page = await defaultContext.newPage();
await page.goto("https://browser-tests-alpha.vercel.app/api/upload-test");
const remoteFilePath = result.filePath;
if (!remoteFilePath) {
console.error("Remote file path not found");
return;
}
const input = await page.$("#fileUpload");
if (input) {
await (input as any).uploadFile(remoteFilePath);
} else {
console.error("Input element not found");
return;
}
await sleep(20_000);
} catch (err) {
console.error("Error uploading file:", err);
} finally {
await hbClient.sessions.stop(session.id);
}
};
main().catch((err) => {
console.error(err);
process.exit(1);
});
```
# Advanced Guide
Source: https://hyperbrowser.ai/docs/web-scraping/advanced-guide
End-to-end guide to scrape, crawl, and extract structured data
This guide shows how to use Hyperbrowser to scrape a single page, crawl multiple pages, and extract structured data. It also documents the most important parameters.
You can also see dedicated pages for [Scrape](/docs/web-scraping/scrape), [Crawl](/docs/web-scraping/crawl), and [Extract](/docs/web-scraping/extract).
For session configuration details, see [Configuration Parameters](/docs/sessions/parameters).
For full schemas, see the API Reference.
## Scraping a web page
With just a URL, you can extract page contents in your chosen formats using the `/scrape` endpoint.
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
// Handles both starting and waiting for scrape job response
const scrapeResult = await client.scrape.startAndWait({
url: "https://example.com",
});
console.log("Scrape result:", scrapeResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Start scraping and wait for completion
scrape_result = client.scrape.start_and_wait({"url": "https://example.com"})
print("Scrape result:\n", scrape_result.model_dump_json(indent=2))
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartScrapeJobParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Start scraping and wait for completion
scrape_result = client.scrape.start_and_wait(
StartScrapeJobParams(url="https://example.com")
)
print("Scrape result:\n", scrape_result.model_dump_json(indent=2))
```
```bash cURL theme={null}
# Start Scrape Job
curl -X POST https://api.hyperbrowser.ai/api/scrape \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com"
}'
# Get Scrape Job Status
curl https://api.hyperbrowser.ai/api/scrape/{jobId}/status \
-H 'x-api-key: '
# Get Scrape Job Status and Data
curl https://api.hyperbrowser.ai/api/scrape/{jobId} \
-H 'x-api-key: '
```
### Session Options
All Scraping APIs (scrape, crawl, extract) support session parameters. See [Session Parameters](/docs/sessions/parameters) for all options.
### Scrape Options
Output formats to include in the response. One or more of: `"html"`, `"links"`, `"markdown"`, `"screenshot"`.
CSS selectors (tags, classes, IDs) to explicitly include. Only matching elements are returned.
CSS selectors (tags, classes, IDs) to exclude from the scraped content.
When `true`, attempts to extract only main content (omits headers/nav/footers).
Milliseconds to wait after initial load before scraping (useful for dynamic content and CAPTCHA detection when `sessionOptions.solveCaptchas` is enabled).
Maximum time (ms) to wait for navigation to complete. Equivalent to `page.goto(url, { waitUntil: "load", timeout })`.
Load condition: `"load"`, `"domcontentloaded"`, or `"networkidle"`.
Screenshot settings (effective only when `formats` includes `"screenshot"`). Both `fullPage` and `cropToContent` cannot be true at the same time.
* `fullPage` (`boolean`, default `false`) — capture full page beyond viewport
* `format` (`"webp" | "jpeg" | "png"`, default `"webp"`)
* `cropToContent` (`boolean`, default `false`) — Automatically adjusts the screenshot height to match the page's actual content. If the page is shorter than the viewport, the screenshot is trimmed to remove any empty space below the content. If the page is taller than the viewport, the screenshot is cropped to the height of the viewport.
* `cropToContentMaxHeight` (`number`, optional) — The maximum height of the screenshot when `cropToContent` is true. Overrides the height set in the `screen` configuration.
* `cropToContentMinHeight` (`number`, optional) — The minimum height of the screenshot when `cropToContent` is true. Overrides the height set in the `screen` configuration.
Set the storage state of the page before scraping.
Properties:
* `localStorage` (`object`, optional) — Local storage data (key-value pairs where both keys and values must be strings)
* `sessionStorage` (`object`, optional) — Session storage data (key-value pairs where both keys and values must be strings)
#### Example with options
By configuring these options when making a scrape request, you can control the format and content of the scraped data, as well as the behavior of the scraper itself.
For example, to scrape a page with the following:
* In stealth mode
* Automatically accept cookies
* Return only the main content as HTML
* Exclude any `` elements
* Wait 2 seconds after the page loads and before scraping
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const scrapeResult = await client.scrape.startAndWait({
url: "https://example.com",
sessionOptions: {
useStealth: true,
acceptCookies: true,
},
scrapeOptions: {
formats: ["html"],
onlyMainContent: true,
excludeTags: ["span"],
waitFor: 2000,
},
});
console.log("Scrape result:", scrapeResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
scrape_result = client.scrape.start_and_wait(
{
"url": "https://example.com",
"session_options": {"use_stealth": True, "accept_cookies": True},
"scrape_options": {
"formats": ["html"],
"only_main_content": True,
"exclude_tags": ["span"],
"wait_for": 2000,
},
}
)
print("Scrape result:\n", scrape_result.model_dump_json(indent=2))
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartScrapeJobParams, CreateSessionParams, ScrapeOptions
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
scrape_result = client.scrape.start_and_wait(
StartScrapeJobParams(
url="https://example.com",
session_options=CreateSessionParams(use_stealth=True, accept_cookies=True),
scrape_options=ScrapeOptions(
formats=["html"],
only_main_content=True,
exclude_tags=["span"],
wait_for=2000,
),
)
)
print("Scrape result:\n", scrape_result.model_dump_json(indent=2))
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/scrape \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"sessionOptions": {
"useStealth": true,
"acceptCookies": true
},
"scrapeOptions": {
"formats": ["html"],
"onlyMainContent": true,
"excludeTags": ["span"],
"waitFor": 2000
}
}'
```
## Crawl a site
Instead of scraping a single page, you can collect content across multiple pages using the `/crawl` endpoint. You can use the same `sessionOptions` and `scrapeOptions` as in `/scrape`, along with additional crawl-specific options below.
### Crawl Options
The URL of the page to crawl.
Maximum number of pages to crawl before stopping (minimum: 1).
When `true`, follow links discovered on pages to expand the crawl.
When `true`, skip pre-generating URLs from sitemaps at the target origin.
Regex or wildcard patterns for URL paths to exclude from the crawl.
Regex or wildcard patterns for URL paths to include (only matching pages will be crawled).
Session configuration used during the crawl. See [Session Parameters](/docs/sessions/parameters).
Scrape options used during the crawl. See [Scrape Options](#scrape-options).
#### Example with options
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const crawlResult = await client.crawl.startAndWait({
url: "https://hyperbrowser.ai",
maxPages: 5,
includePatterns: ["/blog/*"],
scrapeOptions: {
formats: ["markdown"],
onlyMainContent: true,
excludeTags: ["span"],
},
});
console.log("Crawl result:", crawlResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
crawl_result = client.crawl.start_and_wait(
{
"url": "https://hyperbrowser.ai",
"max_pages": 5,
"include_patterns": ["/blog/*"],
"scrape_options": {
"formats": ["markdown"],
"only_main_content": True,
"exclude_tags": ["span"],
},
}
)
print("Crawl result:\n", crawl_result.model_dump_json(indent=2))
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartCrawlJobParams, ScrapeOptions
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
crawl_result = client.crawl.start_and_wait(
StartCrawlJobParams(
url="https://hyperbrowser.ai",
max_pages=5,
include_patterns=["/blog/*"],
scrape_options=ScrapeOptions(
formats=["markdown"],
only_main_content=True,
exclude_tags=["span"],
),
)
)
print("Crawl result:\n", crawl_result.model_dump_json(indent=2))
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/crawl \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://hyperbrowser.ai",
"maxPages": 5,
"includePatterns": ["/blog/*"],
"scrapeOptions": {
"formats": ["markdown"],
"onlyMainContent": true,
"excludeTags": ["span"]
}
}'
```
## Structured extraction
The Extract API fetches data in a well-defined structure from any set of pages. Provide a list of URLs, and Hyperbrowser will collect relevant content (including optional crawling) and return data that fits your schema or prompt.
### Extract Options
List of page URLs. To crawl an origin for a URL, append `/*` (e.g., `https://example.com/*`) to follow relevant links up to `maxLinks`.
JSON Schema for the desired output.
Instructional prompt describing how to structure the extracted data. If no `schema` is provided, we will try to generate a schema based on the prompt.
Additional instructions to guide extraction behavior.
When crawling for any given `/*` URL, the maximum number of links to follow.
Milliseconds to wait after page load before extraction (useful for dynamic content and CAPTCHA detection when `sessionOptions.solveCaptchas` is enabled).
Session configuration used during extraction. See [Session Parameters](/docs/sessions/parameters).
You can provide a **schema**, or a **prompt**, or both. For best results, provide both a **schema** and a **prompt**. The **schema** should define exactly how you want the extracted data formatted, and the **prompt** should include any information that can help guide the extraction. If no **schema** is provided, we will try to automatically generate a **schema** based on the **prompt**.
# Crawl
Source: https://hyperbrowser.ai/docs/web-scraping/crawl
Crawl websites and get formatted data from multiple pages
The Crawl API allows you to crawl websites and get data from multiple pages in a single request. Starting from a URL, it can navigate through the site and extract content from linked pages.
For detailed usage, checkout the [Crawl API Reference](/docs/api-reference/start-a-crawl-job).
Hyperbrowser exposes endpoints for starting a crawl request and for getting its status and results. By default, crawling is handled in an asynchronous manner of first starting the job and then checking its status until it is completed. However, with our SDKs, we provide a simple function that handles the whole flow and returns the data once the job is completed.
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Usage
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
// Handles both starting and waiting for crawl job response
const crawlResult = await client.crawl.startAndWait({
url: "https://example.com",
maxPages: 10,
followLinks: true,
});
console.log("Crawl result:", crawlResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
# Start crawling and wait for completion
crawl_result = client.crawl.start_and_wait(
{
"url": "https://example.com",
"max_pages": 10,
"follow_links": True,
}
)
print("Crawl result:\n", crawl_result.model_dump_json(indent=2))
main()
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartCrawlJobParams
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
# Start crawling and wait for completion
crawl_result = client.crawl.start_and_wait(
StartCrawlJobParams(
url="https://example.com",
max_pages=10,
follow_links=True,
)
)
print("Crawl result:\n", crawl_result.model_dump_json(indent=2))
main()
```
```bash cURL theme={null}
# Start Crawl Job
curl -X POST https://api.hyperbrowser.ai/api/crawl \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"maxPages": 10,
"followLinks": true
}'
# Get Crawl Job Status
curl https://api.hyperbrowser.ai/api/crawl/{jobId}/status \
-H 'x-api-key: '
# Get Crawl Job Status and Data
curl https://api.hyperbrowser.ai/api/crawl/{jobId} \
-H 'x-api-key: '
```
## Response
The Start Crawl Job `POST /crawl` endpoint will return a `jobId` in the response which can be used to get information about the job in subsequent requests.
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c"
}
```
The Get Crawl Job Status `GET /crawl/{jobId}/status` will return the following data:
```json theme={null}
{
"status": "completed"
}
```
The Get Crawl Job `GET /crawl/{jobId}` will return the following data:
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c",
"status": "completed",
"totalCrawledPages": 10,
"data": [
{
"metadata": {
"title": "Example Page",
"description": "A sample webpage",
"url": "https://example.com"
},
"markdown": "# Example Page\nThis is content..."
}
]
}
```
The status of a crawl job can be one of `pending`, `running`, `completed`, `failed`. The results will be an array of scraped pages in the `data` field.
Each crawled page has it's own status of completed or failed and can have it's own error field, so be cautious of that.
To see the full schema, checkout the [API Reference](/docs/api-reference/start-a-crawl-job).
## Crawl Options
You can configure various options for the crawl job:
* **maxPages**: Maximum number of pages to crawl (default: 10, max: 100)
* **followLinks**: Whether to follow links on the crawled pages (default: true)
* **ignoreSitemap**: Whether to ignore the sitemap (default: false)
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const crawlResult = await client.crawl.startAndWait({
url: "https://example.com",
maxPages: 50,
followLinks: true,
ignoreSitemap: false,
});
console.log("Crawl result:", crawlResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Start crawling and wait for completion
crawl_result = client.crawl.start_and_wait(
{
"url": "https://example.com",
"max_pages": 50,
"follow_links": True,
"ignore_sitemap": False,
}
)
print("Crawl result:\n", crawl_result.model_dump_json(indent=2))
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartCrawlJobParams
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Start crawling and wait for completion
crawl_result = client.crawl.start_and_wait(
StartCrawlJobParams(
url="https://example.com", max_pages=50, follow_links=True, ignore_sitemap=False
)
)
print("Crawl result:\n", crawl_result.model_dump_json(indent=2))
```
## Session Configurations
You can also provide configurations for the session that will be used to execute the crawl job, such as using a proxy or solving CAPTCHAs. To see all the different available session parameters, checkout the [API Reference](/docs/api-reference/start-a-crawl-job#body-session-options) or [Session Parameters](/docs/sessions/parameters).
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const crawlResult = await client.crawl.startAndWait({
url: "https://example.com",
maxPages: 10,
followLinks: true,
sessionOptions: {
useProxy: true,
solveCaptchas: true,
proxyCountry: "US",
},
});
console.log("Crawl result:", crawlResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
crawl_result = client.crawl.start_and_wait(
{
"url": "https://example.com",
"max_pages": 10,
"follow_links": True,
"session_options": {"use_proxy": True, "solve_captchas": True},
}
)
print("Crawl result:", crawl_result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartCrawlJobParams, CreateSessionParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
crawl_result = client.crawl.start_and_wait(
StartCrawlJobParams(
url="https://example.com",
max_pages=10,
follow_links=True,
session_options=CreateSessionParams(use_proxy=True, solve_captchas=True),
)
)
print("Crawl result:", crawl_result)
```
Using proxy and solving CAPTCHAs will slow down the crawl so use it only if
necessary.
## Scrape Configurations
You can also provide optional scrape options for the crawl job such as the formats to return, only returning the main content of the page, setting the maximum timeout for navigating to a page, etc.
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const crawlResult = await client.crawl.startAndWait({
url: "https://example.com",
scrapeOptions: {
formats: ["markdown", "html", "links"],
onlyMainContent: false,
timeout: 10000,
},
});
console.log("Crawl result:", crawlResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Start crawling and wait for completion
crawl_result = client.crawl.start_and_wait(
{
"url": "https://example.com",
"scrape_options": {
"formats": ["html", "links", "markdown"],
"only_main_content": False,
"timeout": 10000,
},
}
)
print("Crawl result:", crawl_result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import ScrapeOptions, StartCrawlJobParams
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Start crawling and wait for completion
crawl_result = client.crawl.start_and_wait(
StartCrawlJobParams(
url="https://example.com",
scrape_options=ScrapeOptions(
formats=["html", "links", "markdown"],
only_main_content=False,
timeout=10000,
),
)
)
print("Crawl result:", crawl_result)
```
Hyperbrowser's CAPTCHA solving and proxy usage features require being on a `PAID` plan.
For a full reference on the crawl endpoint, checkout the [API Reference](/docs/api-reference/start-a-crawl-job).
# Extract
Source: https://hyperbrowser.ai/docs/web-scraping/extract
Extract structured data from web pages using AI
The Extract API allows you to extract structured data from web pages using AI. You can define a schema and prompt, and Hyperbrowser will extract the data matching your requirements.
For detailed usage, checkout the [Extract API Reference](/docs/api-reference/start-an-extract-job).
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Usage
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const extractResult = await client.extract.startAndWait({
urls: ["https://example.com"],
prompt: "Extract the main heading and description from the page",
schema: {
type: "object",
properties: {
heading: { type: "string" },
description: { type: "string" },
},
required: ["heading", "description"],
},
});
console.log("Extract result:", extractResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
# Start extraction and wait for completion
extract_result = client.extract.start_and_wait(
{
"urls": ["https://example.com"],
"prompt": "Extract the main heading and description from the page",
"schema": {
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"},
},
"required": ["heading", "description"],
},
}
)
print("Extract result:\n", extract_result.model_dump_json(indent=2))
main()
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartExtractJobParams
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
# Start extraction and wait for completion
extract_result = client.extract.start_and_wait(
StartExtractJobParams(
urls=["https://example.com"],
prompt="Extract the main heading and description from the page",
schema={
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"},
},
"required": ["heading", "description"],
},
)
)
print("Extract result:\n", extract_result.model_dump_json(indent=2))
main()
```
```bash cURL theme={null}
# Start Extract Job
curl -X POST https://api.hyperbrowser.ai/api/extract \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"urls": ["https://example.com"],
"prompt": "Extract the main heading and description from the page",
"schema": {
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"}
},
"required": ["heading", "description"]
}
}'
# Get Extract Job Status
curl https://api.hyperbrowser.ai/api/extract/{jobId}/status \
-H 'x-api-key: '
# Get Extract Job Status and Data
curl https://api.hyperbrowser.ai/api/extract/{jobId} \
-H 'x-api-key: '
```
## Response
The Start Extract Job `POST /extract` endpoint will return a `jobId` in the response which can be used to get information about the job in subsequent requests.
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c"
}
```
The Get Extract Job Status `GET /extract/{jobId}/status` will return the following data:
```json theme={null}
{
"status": "completed"
}
```
The Get Extract Job `GET /extract/{jobId}` will return the following data:
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c",
"status": "completed",
"data": {
"heading": "Example Domain",
"description": "This domain is for use in documentation examples without needing permission. Avoid use in operations."
}
}
```
The status of an extract job can be one of `pending`, `running`, `completed`, `failed`.
To see the full schema, checkout the [API Reference](/docs/api-reference/start-an-extract-job).
## Schema Definition
You can define a JSON schema to specify the structure of the data you want to extract. The schema should follow the JSON Schema specification.
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const extractResult = await client.extract.startAndWait({
urls: ["https://news.ycombinator.com"],
prompt: "Extract all article titles and their URLs from the front page",
schema: {
type: "object",
properties: {
articles: {
type: "array",
items: {
type: "object",
properties: {
title: { type: "string" },
url: { type: "string" },
score: { type: "number" },
},
required: ["title", "url"],
},
},
},
required: ["articles"],
},
});
console.log("Extract result:", extractResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
extract_result = client.extract.start_and_wait(
{
"urls": ["https://news.ycombinator.com"],
"prompt": "Extract all article titles and their URLs from the front page",
"schema": {
"type": "object",
"properties": {
"articles": {
"type": "array",
"items": {
"type": "object",
"properties": {
"title": {"type": "string"},
"url": {"type": "string"},
"score": {"type": "number"},
},
"required": ["title", "url"],
},
}
},
"required": ["articles"],
},
}
)
print("Extract result:\n", extract_result.model_dump_json(indent=2))
main()
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartExtractJobParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
extract_result = client.extract.start_and_wait(
StartExtractJobParams(
urls=["https://news.ycombinator.com"],
prompt="Extract all article titles and their URLs from the front page",
schema={
"type": "object",
"properties": {
"articles": {
"type": "array",
"items": {
"type": "object",
"properties": {
"title": {"type": "string"},
"url": {"type": "string"},
"score": {"type": "number"},
},
"required": ["title", "url"],
},
}
},
"required": ["articles"],
},
)
)
print("Extract result:\n", extract_result.model_dump_json(indent=2))
main()
```
For best results, provide both a schema and a prompt. The schema should define exactly how you want the extract data formatted and the prompt should have any information that can help guide the extraction. If no schema is provided, then we will try to automatically generate a schema based on the prompt.
## Session Configurations
You can also provide configurations for the session that will be used to execute the extract job, such as using a proxy or solving CAPTCHAs. To see all the different available session parameters, checkout the [API Reference](/docs/api-reference/start-an-extract-job#body-session-options) or [Session Parameters](/docs/sessions/parameters).
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const extractResult = await client.extract.startAndWait({
urls: ["https://example.com"],
prompt: "Extract the main heading and description",
schema: {
type: "object",
properties: {
heading: { type: "string" },
description: { type: "string" },
},
},
sessionOptions: {
useProxy: true,
solveCaptchas: true,
proxyCountry: "US",
},
});
console.log("Extract result:", extractResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
extract_result = client.extract.start_and_wait(
{
"urls": ["https://example.com"],
"prompt": "Extract the main heading and description",
"schema": {
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"},
},
},
"session_options": {"use_proxy": True, "solve_captchas": True},
}
)
print("Extract result:", extract_result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartExtractJobParams, CreateSessionParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
extract_result = client.extract.start_and_wait(
StartExtractJobParams(
urls=["https://example.com"],
prompt="Extract the main heading and description",
schema={
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"},
},
},
session_options=CreateSessionParams(use_proxy=True, solve_captchas=True),
)
)
print("Extract result:", extract_result)
```
Hyperbrowser's CAPTCHA solving and proxy usage features require being on a PAID plan.
Using proxy and solving CAPTCHAs will slow down the page scraping in the extract job so use it only if necessary.
For a full reference on the extract endpoint, checkout the [API Reference](/docs/api-reference/start-an-extract-job).
# Scrape
Source: https://hyperbrowser.ai/docs/web-scraping/scrape
Scrape any page and get formatted data
The Scrape API allows you to get the data you want from web pages with a single call. You can scrape page content and capture its data in various formats like markdown or html.
For detailed usage, checkout the [Scrape API Reference](/docs/api-reference/create-new-scrape-job).
Hyperbrowser exposes endpoints for starting a scrape request and for getting its status and results. By default, scraping is handled in an asynchronous manner of first starting the job and then checking its status until it is completed. However, with our SDKs, we provide a simple function that handles the whole flow and returns the data once the job is completed.
## Installation
```bash npm theme={null}
npm install @hyperbrowser/sdk dotenv
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk dotenv
```
```bash pip theme={null}
pip install hyperbrowser python-dotenv
```
```bash uv theme={null}
uv add hyperbrowser python-dotenv
```
## Usage
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
// Handles both starting and waiting for scrape job response
const scrapeResult = await client.scrape.startAndWait({
url: "https://example.com",
});
console.log("Scrape result:", scrapeResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
# Start scraping and wait for completion
scrape_result = client.scrape.start_and_wait({"url": "https://example.com"})
print("Scrape result:\n", scrape_result.model_dump_json(indent=2))
main()
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartScrapeJobParams
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
# Start scraping and wait for completion
scrape_result = client.scrape.start_and_wait(
StartScrapeJobParams(url="https://example.com")
)
print("Scrape result:\n", scrape_result.model_dump_json(indent=2))
main()
```
```bash cURL theme={null}
# Start Scrape Job
curl -X POST https://api.hyperbrowser.ai/api/scrape \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com"
}'
# Get Scrape Job Status
curl https://api.hyperbrowser.ai/api/scrape/{jobId}/status \
-H 'x-api-key: '
# Get Scrape Job Status and Data
curl https://api.hyperbrowser.ai/api/scrape/{jobId} \
-H 'x-api-key: '
```
## Response
The Start Scrape Job `POST /scrape` endpoint will return a `jobId` in the response which can be used to get information about the job in subsequent requests.
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c"
}
```
The Get Scrape Job Status `GET /scrape/{jobId}/status` will return the following data:
```json theme={null}
{
"status": "completed"
}
```
The Get Scrape Job `GET /scrape/{jobId}` will return the following data:
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c",
"status": "completed",
"data": {
"metadata": {
"title": "Example Page",
"description": "A sample webpage"
},
"markdown": "# Example Page\nThis is content..."
}
}
```
The status of a scrape job can be one of `pending`, `running`, `completed`, `failed`. There can also be other optional fields like `error` with an error message if an error was encountered, and `html` and `links` in the data object depending on which formats are requested for the request.
To see the full schema, checkout the [API Reference](/docs/api-reference/create-new-scrape-job).
## Session Configurations
You can also provide configurations for the session that will be used to execute the scrape job just as you would when creating a new session itself. These could include using a proxy or solving CAPTCHAs. To see all the different available session parameters, checkout the [API Reference](/docs/api-reference/create-new-scrape-job#body-session-options) or [Session Parameters](/docs/sessions/parameters).
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const scrapeResult = await client.scrape.startAndWait({
url: "https://example.com",
sessionOptions: {
useProxy: true,
solveCaptchas: true,
proxyCountry: "US",
},
});
console.log("Scrape result:", scrapeResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
scrape_result = client.scrape.start_and_wait(
{
"url": "https://example.com",
"session_options": {"use_proxy": True, "solve_captchas": True},
}
)
print("Scrape result:\n", scrape_result.model_dump_json(indent=2))
main()
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartScrapeJobParams, CreateSessionParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
def main():
scrape_result = client.scrape.start_and_wait(
StartScrapeJobParams(
url="https://example.com",
session_options=CreateSessionParams(use_proxy=True, solve_captchas=True),
)
)
print("Scrape result:\n", scrape_result.model_dump_json(indent=2))
main()
```
Proxy Usage and CAPTCHA solving are only available on `PAID` plans.
Using proxy and solving CAPTCHAs will slow down the scrape so use it if necessary.
## Scrape Configurations
You can also provide optional parameters for the scrape job itself such as the formats to return, only returning the main content of the page, setting the maximum timeout for navigating to a page, etc.
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const scrapeResult = await client.scrape.startAndWait({
url: "https://example.com",
scrapeOptions: {
formats: ["markdown", "html", "links"],
onlyMainContent: false,
timeout: 15000,
},
});
console.log("Scrape result:", scrapeResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Start scraping and wait for completion
scrape_result = client.scrape.start_and_wait(
{
"url": "https://example.com",
"scrape_options": {
"formats": ["html", "links", "markdown"],
"only_main_content": False,
"timeout": 5000,
},
}
)
print("Scrape result:", scrape_result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import ScrapeOptions, StartScrapeJobParams
# Load environment variables from .env file
load_dotenv()
# Initialize Hyperbrowser client
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
# Start scraping and wait for completion
scrape_result = client.scrape.start_and_wait(
StartScrapeJobParams(
url="https://example.com",
scrape_options=ScrapeOptions(
formats=["html", "links", "markdown"], only_main_content=False, timeout=5000
),
)
)
print("Scrape result:", scrape_result)
```
For a full reference on the scrape endpoint, checkout the [API Reference](/docs/api-reference/create-new-scrape-job).
## Batch Scrape
Batch Scrape works the same as regular scrape, except instead of a single URL, you can provide a list of up to 1,000 URLs to scrape at once.
Batch Scrape is currently only available on the `Scale` plan or higher.
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const main = async () => {
const scrapeResult = await client.scrape.batch.startAndWait({
urls: ["https://example.com", "https://hyperbrowser.ai"],
scrapeOptions: {
formats: ["markdown", "html", "links"],
},
});
console.log("Scrape result:", scrapeResult);
};
main();
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
scrape_result = client.scrape.batch.start_and_wait(
{
"urls": ["https://example.com", "https://hyperbrowser.ai"],
"scrape_options": {"formats": ["html", "links", "markdown"]},
}
)
print("Scrape result:", scrape_result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import ScrapeOptions, StartBatchScrapeJobParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
scrape_result = client.scrape.batch.start_and_wait(
StartBatchScrapeJobParams(
urls=["https://example.com", "https://hyperbrowser.ai"],
scrape_options=ScrapeOptions(formats=["html", "links", "markdown"]),
)
)
print("Scrape result:", scrape_result)
```
### Response
The Start Batch Scrape Job `POST /scrape/batch` endpoint will return a `jobId` in the response which can be used to get information about the job in subsequent requests.
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c"
}
```
The Get Batch Scrape Job Status `GET /scrape/batch/{jobId}/status` will return the following data:
```json theme={null}
{
"status": "completed"
}
```
The Get Batch Scrape Job `GET /scrape/batch/{jobId}` will return the following data:
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c",
"status": "completed",
"totalScrapedPages": 2,
"totalPageBatches": 1,
"currentPageBatch": 1,
"batchSize": 20,
"data": [
{
"markdown": "Hyperbrowser\n\n[Home](https://hyperbrowser.ai/)...",
"metadata": {
"url": "https://www.hyperbrowser.ai/",
"title": "Hyperbrowser",
"viewport": "width=device-width, initial-scale=1",
"link:icon": "https://www.hyperbrowser.ai/favicon.ico",
"sourceURL": "https://hyperbrowser.ai",
"description": "Infinite Browsers"
},
"url": "hyperbrowser.ai",
"status": "completed",
"error": null
},
{
"markdown": "Example Domain\n\n# Example Domain...",
"metadata": {
"url": "https://www.example.com/",
"title": "Example Domain",
"viewport": "width=device-width, initial-scale=1",
"sourceURL": "https://example.com"
},
"url": "example.com",
"status": "completed",
"error": null
}
]
}
```
Hyperbrowser's CAPTCHA solving and proxy usage features require being on a `PAID` plan.
The status of a batch scrape job can be one of `pending`, `running`, `completed`, `failed`. The results of all the scrapes will be an array in the `data` field of the response. Each scraped page will be returned in the order of the initial provided urls, and each one will have its own status and information.
To see the full schema, checkout the [API Reference](/docs/api-reference/start-a-batch-scrape-job).
As with the single scrape, by default, batch scraping is handled in an asynchronous manner of first starting the job and then checking its status until it is completed. However, with our SDKs, we provide a simple function (`client.scrape.batch.startAndWait`) that handles the whole flow and returns the data once the job is completed.
# Crawl
Source: https://hyperbrowser.ai/docs/web/crawl
Crawl a website and get structured data from multiple pages
Crawl starts from a URL and follows links across the site, returning content from each page in the formats you choose—markdown, HTML, links, screenshots, structured JSON, or a branding profile. It shares the same output options as [Fetch](/docs/web/fetch), applied to every page it visits.
For the full API schema, see the [Crawl API Reference](/docs/api-reference/start-a-web-crawl-job).
***
## Quick Start
```bash npm theme={null}
npm install @hyperbrowser/sdk
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk
```
```bash pip theme={null}
pip install hyperbrowser
```
```bash uv theme={null}
uv add hyperbrowser
```
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.crawl.startAndWait({
url: "https://example.com",
crawlOptions: {
maxPages: 10,
followLinks: true,
},
});
console.log(result);
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.crawl.start_and_wait(
{
"url": "https://example.com",
"crawl_options": {
"max_pages": 10,
"follow_links": True,
},
}
)
print(result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartWebCrawlJobParams, WebCrawlOptions
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.crawl.start_and_wait(
StartWebCrawlJobParams(
url="https://example.com",
crawl_options=WebCrawlOptions(
max_pages=10,
follow_links=True,
),
)
)
print(result)
```
```bash cURL theme={null}
# Start a crawl job
curl -X POST https://api.hyperbrowser.ai/api/web/crawl \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"crawlOptions": {
"maxPages": 10,
"followLinks": true
}
}'
# Get crawl job results
curl https://api.hyperbrowser.ai/api/web/crawl/{jobId} \
-H 'x-api-key: '
```
Crawling is asynchronous—start the job and then poll for results. The SDKs provide a `startAndWait` / `start_and_wait` convenience method that handles polling for you.
***
## Response
Starting a crawl job returns a `jobId`:
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c"
}
```
Once complete, the full response includes an array of page results under `data`:
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c",
"status": "completed",
"data": [
{
"url": "https://example.com",
"status": "completed",
"metadata": {
"title": "Example Domain",
"sourceURL": "https://example.com"
},
"markdown": "# Example Domain\n\nThis domain is for use in illustrative examples..."
},
{
"url": "https://example.com/about",
"status": "completed",
"metadata": {
"title": "About - Example Domain",
"sourceURL": "https://example.com/about"
},
"markdown": "# About\n\nMore information about this example..."
}
],
"totalPages": 2,
"totalPageBatches": 1,
"currentPageBatch": 0,
"batchSize": 10
}
```
The `status` field on the job can be `pending`, `running`, `completed`, or `failed`. Each page in `data` also has its own `status` and may include an `error` field if that page failed.
***
## Crawl Options
Control how the crawler traverses the site with `crawlOptions`:
| Field | Type | Default | Description |
| - | - | - | - |
| `crawlOptions.maxPages` | `number` | `10` | Maximum number of pages to crawl (max: 100) |
| `crawlOptions.followLinks` | `boolean` | `true` | Whether to follow links found on crawled pages |
| `crawlOptions.ignoreSitemap` | `boolean` | `false` | Whether to ignore the site's sitemap |
| `crawlOptions.includePatterns` | `string[]` | `[]` | URL patterns to include (only matching URLs are crawled) |
| `crawlOptions.excludePatterns` | `string[]` | `[]` | URL patterns to exclude from crawling |
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.crawl.startAndWait({
url: "https://example.com",
crawlOptions: {
maxPages: 50,
followLinks: true,
includePatterns: ["/docs/*", "/blog/*"],
excludePatterns: ["/docs/archive/*"],
},
});
console.log(result);
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.crawl.start_and_wait(
{
"url": "https://example.com",
"crawl_options": {
"max_pages": 50,
"follow_links": True,
"include_patterns": ["/docs/*", "/blog/*"],
"exclude_patterns": ["/docs/archive/*"],
},
}
)
print(result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import StartWebCrawlJobParams, WebCrawlOptions
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.crawl.start_and_wait(
StartWebCrawlJobParams(
url="https://example.com",
crawl_options=WebCrawlOptions(
max_pages=50,
follow_links=True,
include_patterns=["/docs/*", "/blog/*"],
exclude_patterns=["/docs/archive/*"],
),
)
)
print(result)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/crawl \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"crawlOptions": {
"maxPages": 50,
"followLinks": true,
"includePatterns": ["/docs/*", "/blog/*"],
"excludePatterns": ["/docs/archive/*"]
}
}'
```
***
## Outputs
Use `outputs.formats` to control what data is returned for each crawled page. This works the same as [Fetch outputs](/docs/web/fetch#outputs)—you can request markdown, HTML, links, screenshots, structured JSON, or a branding profile.
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.crawl.startAndWait({
url: "https://example.com",
outputs: {
formats: ["markdown", "links"],
},
crawlOptions: {
maxPages: 10,
},
});
for (const page of result.data) {
console.log(page.url, page.markdown?.slice(0, 100));
}
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.crawl.start_and_wait(
{
"url": "https://example.com",
"outputs": {"formats": ["markdown", "links"]},
"crawl_options": {"max_pages": 10},
}
)
for page in result.data:
print(page.url, page.markdown[:100] if page.markdown else "")
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import (
StartWebCrawlJobParams,
WebCrawlOptions,
FetchOutputOptions,
)
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.crawl.start_and_wait(
StartWebCrawlJobParams(
url="https://example.com",
outputs=FetchOutputOptions(formats=["markdown", "links"]),
crawl_options=WebCrawlOptions(max_pages=10),
)
)
for page in result.data:
print(page.url, page.markdown[:100] if page.markdown else "")
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/crawl \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"outputs": {
"formats": ["markdown", "links"]
},
"crawlOptions": {
"maxPages": 10
}
}'
```
Pass a JSON Schema to extract structured data from each crawled page. You can use a raw JSON Schema object, a Zod schema (Node), or a Pydantic model (Python).
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
import { z } from "zod";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const PageSchema = z.object({
heading: z.string(),
description: z.string(),
});
const result = await client.web.crawl.startAndWait({
url: "https://example.com",
outputs: {
formats: [
"markdown",
{
type: "json",
schema: PageSchema,
},
],
},
crawlOptions: {
maxPages: 5,
},
});
for (const page of result.data) {
console.log(page.url, page.json);
}
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from pydantic import BaseModel
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class PageData(BaseModel):
heading: str
description: str
result = client.web.crawl.start_and_wait(
{
"url": "https://example.com",
"outputs": {
"formats": [
"markdown",
{"type": "json", "schema": PageData},
]
},
"crawl_options": {"max_pages": 5},
}
)
for page in result.data:
print(page.url, page.json_)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import (
StartWebCrawlJobParams,
WebCrawlOptions,
FetchOutputOptions,
FetchOutputJson,
)
from pydantic import BaseModel
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class PageData(BaseModel):
heading: str
description: str
result = client.web.crawl.start_and_wait(
StartWebCrawlJobParams(
url="https://example.com",
outputs=FetchOutputOptions(
formats=[
"markdown",
FetchOutputJson(type="json", schema=PageData),
]
),
crawl_options=WebCrawlOptions(max_pages=5),
)
)
for page in result.data:
print(page.url, page.json_)
```
For the full list of output formats and options (screenshots, sanitization, selectors, storage state), see the [Fetch outputs documentation](/docs/web/fetch#outputs).
***
## Output Controls
Control what gets extracted from each crawled page. These work the same as [Fetch output controls](/docs/web/fetch#output-controls):
| Field | Type | Default | Description |
| - | - | - | - |
| `outputs.sanitize` | `string` | `"none"` | Sanitize mode: `"none"`, `"basic"`, or `"advanced"` |
| `outputs.includeSelectors` | `string[]` | `[]` | CSS selectors to include (only matching elements returned) |
| `outputs.excludeSelectors` | `string[]` | `[]` | CSS selectors to exclude from output |
| `outputs.storageState` | `object` | — | Pre-seed localStorage/sessionStorage before fetching |
***
## Browser & Stealth
Configure how the cloud browser runs. These options apply to all pages in the crawl:
| Field | Type | Default | Description |
| - | - | - | - |
| `stealth` | `string` | `"auto"` | Stealth mode: `"none"`, `"auto"`, or `"ultra"` |
| `browser.profileId` | `string` | — | Reuse an existing browser profile |
| `browser.solveCaptchas` | `boolean` | `false` | Enable CAPTCHA solving |
| `browser.screen` | `object` | `{ width: 1280, height: 720 }` | Set viewport dimensions (`width`, `height`) |
| `browser.location` | `object` | — | Localize via proxy location (`country`, `state`, `city`). If set, proxy is enabled automatically |
## Navigation Controls
Control page load behavior and timing for each crawled page:
| Field | Type | Default | Description |
| - | - | - | - |
| `navigation.waitUntil` | `string` | `"domcontentloaded"` | Load condition: `"load"`, `"domcontentloaded"`, or `"networkidle"` |
| `navigation.waitFor` | `number` | `0` | Milliseconds to wait after navigation completes before collecting outputs (0–30000) |
| `navigation.timeoutMs` | `number` | `30000` | Max time (ms) to wait for navigation (1–60000) |
## Cache Controls
Control caching behavior for crawl results:
| Field | Type | Default | Description |
| - | - | - | - |
| `cache.maxAgeSeconds` | `number` | — | Cache control—cached results older than this are treated as stale. Set to `0` to bypass cache reads |
***
## Pagination
Crawl results are returned in batches. You can control pagination when retrieving results:
| Parameter | Type | Default | Description |
| - | - | - | - |
| `page` | `number` | `0` | Page batch index to retrieve |
| `batchSize` | `number` | `10` | Number of page results per batch |
The response includes pagination metadata:
| Field | Description |
| - | - |
| `totalPages` | Total number of crawled pages |
| `totalPageBatches` | Total number of result batches |
| `currentPageBatch` | Current batch index |
| `batchSize` | Number of results in each batch |
When using `startAndWait` / `start_and_wait` with `returnAllPages` set to `true` (the default), the SDK automatically fetches all paginated results and combines them into a single response.
# Fetch
Source: https://hyperbrowser.ai/docs/web/fetch
Fetch any page and get data in the formats you choose
Fetch loads a URL in a cloud browser and returns the outputs you specify—markdown, HTML, links, screenshots, structured JSON, or a branding profile. It's the core building block for web scraping with Hyperbrowser.
For the full API schema, see the [Fetch API Reference](/docs/api-reference/fetch-a-web-page).
***
## Quick Start
```bash npm theme={null}
npm install @hyperbrowser/sdk
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk
```
```bash pip theme={null}
pip install hyperbrowser
```
```bash uv theme={null}
uv add hyperbrowser
```
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.fetch({
url: "https://example.com",
});
console.log(result);
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch({"url": "https://example.com"})
print(result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(FetchParams(url="https://example.com"))
print(result)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/fetch \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"outputs": {
"formats": ["markdown", "links"]
}
}'
```
***
## Response
The response includes a `jobId`, overall `status`, and the outputs under `data`:
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c",
"status": "completed",
"data": {
"metadata": {
"title": "Example Domain",
"sourceURL": "https://example.com"
},
"markdown": "# Example Domain\n\nThis domain is for use in illustrative examples...",
"links": [
"https://www.iana.org/domains/example"
]
}
}
```
***
## Outputs
Use `outputs.formats` to specify what data you want returned.
| Output | Description |
| - | - |
| `markdown` | Page content converted to Markdown |
| `html` | Raw HTML of the page |
| `links` | All links found on the page |
| `screenshot` | Screenshot image (configure with options) |
| `json` | Structured data extracted using a JSON Schema, a prompt, or both |
| `branding` | Visual brand profile—colors, fonts, logo, button styles, personality |
`outputs.formats` cannot contain duplicate output types (e.g. two screenshots).
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.fetch({
url: "https://example.com",
outputs: {
formats: ["markdown", "links"],
},
});
console.log(result.data.markdown);
console.log(result.data.links);
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
{
"url": "https://example.com",
"outputs": {"formats": ["markdown", "links"]},
}
)
print(result.data.markdown)
print(result.data.links)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams, FetchOutputOptions
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
FetchParams(
url="https://example.com",
outputs=FetchOutputOptions(formats=["markdown", "links"]),
)
)
print(result.data.markdown)
print(result.data.links)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/fetch \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"outputs": {
"formats": ["markdown", "links"]
}
}'
```
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.fetch({
url: "https://hackernews.com",
outputs: {
formats: [
"markdown",
{
type: "screenshot",
fullPage: true,
format: "png",
},
],
},
});
console.log(result.data.screenshot);
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
{
"url": "https://hackernews.com",
"outputs": {
"formats": [
"markdown",
{
"type": "screenshot",
"full_page": True,
"format": "png",
},
]
},
}
)
print(result.data.screenshot)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams, FetchOutputOptions, FetchOutputScreenshot
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
FetchParams(
url="https://hackernews.com",
outputs=FetchOutputOptions(
formats=[
"markdown",
FetchOutputScreenshot(
type="screenshot",
full_page=True,
format="png",
),
]
),
)
)
print(result.data.screenshot)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/fetch \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://hackernews.com",
"outputs": {
"formats": [
"markdown",
{
"type": "screenshot",
"fullPage": true,
"format": "png"
}
]
}
}'
```
Extract structured data from the page using `prompt`, `schema`, or both:
* **`prompt` only** — Describe what you want in natural language. A schema is auto-generated from the prompt.
* **`schema` only** — Provide a JSON Schema (or Zod/Pydantic equivalent) defining the exact output structure.
* **`prompt` and `schema`** — The schema defines the output structure, while the prompt provides additional guidance for the extraction.
For best results, provide both a `schema` and a `prompt`. The schema defines exactly how you want the data formatted, and the prompt provides context to guide the extraction.
**Prompt only (schema auto-generated):**
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.fetch({
url: "https://example.com",
outputs: {
formats: [
{
type: "json",
prompt: "Extract the main heading and a brief description of the page",
},
],
},
});
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
{
"url": "https://example.com",
"outputs": {
"formats": [
{
"type": "json",
"prompt": "Extract the main heading and a brief description of the page",
}
]
},
}
)
print(result.data.json_)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams, FetchOutputOptions, FetchOutputJson
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
FetchParams(
url="https://example.com",
outputs=FetchOutputOptions(
formats=[
FetchOutputJson(
type="json",
prompt="Extract the main heading and a brief description of the page",
)
]
),
)
)
print(result.data.json_)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/fetch \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"outputs": {
"formats": [
{
"type": "json",
"prompt": "Extract the main heading and a brief description of the page"
}
]
}
}'
```
**Schema only:**
```typescript Node (JSON Schema) theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.fetch({
url: "https://example.com",
outputs: {
formats: [
{
type: "json",
schema: {
type: "object",
properties: {
heading: { type: "string" },
description: { type: "string" },
},
required: ["heading", "description"],
additionalProperties: false,
},
},
],
},
});
```
```typescript Node (Zod) theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
import { z } from "zod";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const PageSchema = z.object({
heading: z.string(),
description: z.string(),
});
const result = await client.web.fetch({
url: "https://example.com",
outputs: {
formats: [
{
type: "json",
schema: PageSchema,
},
],
},
});
```
```python Python 1.0+ (JSON Schema) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
{
"url": "https://example.com",
"outputs": {
"formats": [
{
"type": "json",
"schema": {
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"},
},
"required": ["heading", "description"],
"additionalProperties": False,
},
}
]
},
}
)
print(result.data.json_)
```
```python Python (legacy, JSON Schema) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams, FetchOutputOptions, FetchOutputJson
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
FetchParams(
url="https://example.com",
outputs=FetchOutputOptions(
formats=[
FetchOutputJson(
type="json",
schema={
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"},
},
"required": ["heading", "description"],
"additionalProperties": False,
},
)
]
),
)
)
print(result.data.json_)
```
```python Python 1.0+ (Pydantic schema) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from pydantic import BaseModel
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class PageData(BaseModel):
heading: str
description: str
result = client.web.fetch(
{
"url": "https://example.com",
"outputs": {"formats": [{"type": "json", "schema": PageData}]},
}
)
print(result.data.json_)
```
```python Python (legacy, Pydantic schema) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams, FetchOutputOptions, FetchOutputJson
from pydantic import BaseModel
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class PageData(BaseModel):
heading: str
description: str
result = client.web.fetch(
FetchParams(
url="https://example.com",
outputs=FetchOutputOptions(
formats=[FetchOutputJson(type="json", schema=PageData)]
),
)
)
print(result.data.json_)
```
**Both prompt and schema (recommended):**
```typescript Node (JSON Schema) theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.fetch({
url: "https://example.com",
outputs: {
formats: [
{
type: "json",
prompt: "Extract the main heading and a brief description of the page",
schema: {
type: "object",
properties: {
heading: { type: "string" },
description: { type: "string" },
},
required: ["heading", "description"],
additionalProperties: false,
},
},
],
},
});
```
```typescript Node (Zod) theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
import { z } from "zod";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const PageSchema = z.object({
heading: z.string(),
description: z.string(),
});
const result = await client.web.fetch({
url: "https://example.com",
outputs: {
formats: [
{
type: "json",
prompt: "Extract the main heading and a brief description of the page",
schema: PageSchema,
},
],
},
});
```
```python Python 1.0+ (JSON Schema) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
{
"url": "https://example.com",
"outputs": {
"formats": [
{
"type": "json",
"prompt": "Extract the main heading and a brief description of the page",
"schema": {
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"},
},
"required": ["heading", "description"],
"additionalProperties": False,
},
}
]
},
}
)
print(result.data.json_)
```
```python Python (legacy, JSON Schema) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams, FetchOutputOptions, FetchOutputJson
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
FetchParams(
url="https://example.com",
outputs=FetchOutputOptions(
formats=[
FetchOutputJson(
type="json",
prompt="Extract the main heading and a brief description of the page",
schema={
"type": "object",
"properties": {
"heading": {"type": "string"},
"description": {"type": "string"},
},
"required": ["heading", "description"],
"additionalProperties": False,
},
)
]
),
)
)
print(result.data.json_)
```
```python Python 1.0+ (Pydantic schema) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from pydantic import BaseModel
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class PageData(BaseModel):
heading: str
description: str
result = client.web.fetch(
{
"url": "https://example.com",
"outputs": {
"formats": [
{
"type": "json",
"prompt": "Extract the main heading and a brief description of the page",
"schema": PageData,
}
]
},
}
)
print(result.data.json_)
```
```python Python (legacy, Pydantic schema) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams, FetchOutputOptions, FetchOutputJson
from pydantic import BaseModel
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
class PageData(BaseModel):
heading: str
description: str
result = client.web.fetch(
FetchParams(
url="https://example.com",
outputs=FetchOutputOptions(
formats=[
FetchOutputJson(
type="json",
prompt="Extract the main heading and a brief description of the page",
schema=PageData,
)
]
),
)
)
print(result.data.json_)
```
**Node SDK**: You can pass a Zod schema directly to the `schema` field, and it will be automatically converted to JSON Schema.
**Python SDK**: You can pass a Pydantic model class directly to the `schema` field, and it will be automatically converted to JSON Schema.
Screenshot cropping rules:
* `fullPage` and `cropToContent` are mutually exclusive.
* If both `cropToContentMaxHeight` and `cropToContentMinHeight` are set, max must be ≥ min.
* Crop dimensions must be between **100** and **8000** pixels.
Extract a structured brand profile for the page: color roles (primary, secondary, accent, background, text), typography, logo + favicon, primary/secondary button styles, personality, and confidence scores. Combines in-browser DOM analysis with an LLM pass over the detected elements.
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
async function main() {
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.fetch({
url: "https://stripe.com",
outputs: {
formats: ["branding"],
},
});
console.log(result.data?.branding?.colors?.primary);
console.log(result.data?.branding?.images?.logo);
console.log(result.data?.branding?.personality?.tone);
}
main().catch(console.error);
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
{
"url": "https://stripe.com",
"outputs": {"formats": ["branding"]},
}
)
branding = result.data.branding if result.data else None
if branding:
if branding.colors:
print(branding.colors.primary)
if branding.images:
print(branding.images.logo)
if branding.personality:
print(branding.personality.tone)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import FetchParams, FetchOutputOptions
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
FetchParams(
url="https://stripe.com",
outputs=FetchOutputOptions(formats=["branding"]),
)
)
branding = result.data.branding if result.data else None
if branding:
if branding.colors:
print(branding.colors.primary)
if branding.images:
print(branding.images.logo)
if branding.personality:
print(branding.personality.tone)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/fetch \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://stripe.com",
"outputs": {
"formats": ["branding"]
}
}'
```
The response `data.branding` object includes:
| Field | Description |
| - | - |
| `colorScheme` | `"light"` or `"dark"` |
| `colors` | `primary`, `secondary`, `accent`, `background`, `textPrimary`, … |
| `fonts` | Cleaned brand fonts with roles (`heading`, `body`, `monospace`, …) |
| `typography` | `fontFamilies`, `fontStacks`, `fontSizes` |
| `spacing` | `baseUnit`, `borderRadius` |
| `components` | `buttonPrimary`, `buttonSecondary`, `input` — full CSS style objects |
| `images` | `logo`, `logoHref`, `logoAlt`, `favicon`, `ogImage` |
| `personality` | `tone`, `energy`, `targetAudience` |
| `designSystem` | `framework` (tailwind, bootstrap, …), `componentLibrary` |
| `confidence` | `buttons`, `colors`, `overall` (0–1) |
Branding works identically in [Crawl](/docs/web/crawl) — each crawled page gets its own `branding` profile.
***
## Output Controls
Control what gets extracted and returned:
| Field | Type | Default | Description |
| - | - | - | - |
| `outputs.sanitize` | `string` | `"none"` | Sanitize mode: `"none"`, `"basic"`, or `"advanced"` |
| `outputs.includeSelectors` | `string[]` | `[]` | CSS selectors to include (only matching elements returned) |
| `outputs.excludeSelectors` | `string[]` | `[]` | CSS selectors to exclude from output |
| `outputs.storageState` | `object` | — | Pre-seed localStorage/sessionStorage before fetching |
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.fetch({
url: "https://example.com",
outputs: {
formats: ["html"],
excludeSelectors: ["nav", "footer", "aside"],
storageState: {
localStorage: {
"example:key": "example:value",
},
},
},
navigation: {
waitUntil: "load",
waitFor: 2000,
},
cache: {
maxAgeSeconds: 3600,
},
});
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
{
"url": "https://example.com",
"outputs": {
"formats": ["html"],
"exclude_selectors": ["nav", "footer", "aside"],
"storage_state": {"local_storage": {"example:key": "example:value"}},
},
"navigation": {"wait_until": "load", "wait_for": 2000},
"cache": {"max_age_seconds": 3600},
}
)
print(result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import (
FetchParams,
FetchOutputOptions,
FetchNavigationOptions,
FetchCacheOptions,
FetchStorageStateOptions,
)
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.fetch(
FetchParams(
url="https://example.com",
outputs=FetchOutputOptions(
formats=["html"],
exclude_selectors=["nav", "footer", "aside"],
storage_state=FetchStorageStateOptions(
local_storage={"example:key": "example:value"}
),
),
navigation=FetchNavigationOptions(wait_until="load", wait_for=2000),
cache=FetchCacheOptions(max_age_seconds=3600),
)
)
print(result)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/fetch \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"url": "https://example.com",
"outputs": {
"formats": ["html"],
"excludeSelectors": ["nav", "footer", "aside"],
"storageState": {
"localStorage": {
"example:key": "example:value"
}
}
},
"navigation": {
"waitUntil": "load",
"waitFor": 2000
},
"cache": {
"maxAgeSeconds": 3600
}
}'
```
## Browser & Stealth
Configure how the cloud browser runs:
| Field | Type | Default | Description |
| - | - | - | - |
| `stealth` | `string` | `"auto"` | Stealth mode: `"none"`, `"auto"`, or `"ultra"` (recommended: `"auto"` or `"ultra"`) |
| `browser.profileId` | `string` | — | Reuse an existing browser profile |
| `browser.solveCaptchas` | `boolean` | `false` | Enable CAPTCHA solving |
| `browser.screen` | `object` | `{ width: 1280, height: 720 }` | Set viewport dimensions (`width`, `height`) |
| `browser.location` | `object` | — | Localize via proxy location (`country`, `state`, `city`). If set, proxy is enabled automatically |
## Navigation Controls
Control page load behavior and timing:
| Field | Type | Default | Description |
| - | - | - | - |
| `navigation.waitUntil` | `string` | `"domcontentloaded"` | Load condition: `"load"`, `"domcontentloaded"`, or `"networkidle"` |
| `navigation.waitFor` | `number` | `0` | Milliseconds to wait after navigation completes before collecting outputs (0–30000) |
| `navigation.timeoutMs` | `number` | `30000` | Max time (ms) to wait for navigation (1–60000) |
## Cache Controls
Control caching behavior for fetch results:
| Field | Type | Default | Description |
| - | - | - | - |
| `cache.maxAgeSeconds` | `number` | — | Cache control—cached results older than this are treated as stale. Set to `0` to bypass cache reads |
Using cache can improve response times for frequently accessed pages. Set `maxAgeSeconds` based on how fresh you need the data to be.
***
## X402 Support
For agentic workflows, Hyperbrowser supports x402 payment for the Fetch API, enabling agents to programmatically pay for browser sessions at request time using USDC, with payment handled inline via HTTP 402 responses. See the [X402 Integration](/docs/integrations/x402) page for details.
# Overview
Source: https://hyperbrowser.ai/docs/web/overview
Fetch pages, crawl sites, and search the web
The Web API is Hyperbrowser's unified interface for web data extraction. Fetch pages, crawl entire sites, and search the web through a consistent set of endpoints.
***
## API Overview
Fetch a single URL and return markdown, HTML, links, screenshots, or structured JSON.
Crawl a website and get structured data from multiple pages.
Search the web and get clean, structured results.
***
# Search
Source: https://hyperbrowser.ai/docs/web/search
Query the web and get clean, structured results
Search queries the web and returns structured results—titles, URLs, and snippets—ready to use in your application. Combine it with [Fetch](/docs/web/fetch) to retrieve full page content from search results.
For the full API schema, see the [Search API Reference](/docs/api-reference/search-the-web).
***
## Quick Start
```bash npm theme={null}
npm install @hyperbrowser/sdk
```
```bash yarn theme={null}
yarn add @hyperbrowser/sdk
```
```bash pip theme={null}
pip install hyperbrowser
```
```bash uv theme={null}
uv add hyperbrowser
```
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.search({
query: "hyperbrowser browser automation",
});
console.log(result);
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.search({"query": "hyperbrowser browser automation"})
print(result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import WebSearchParams
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.search(WebSearchParams(query="hyperbrowser browser automation"))
print(result)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/search \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"query": "hyperbrowser browser automation"
}'
```
***
## Response
The response includes search results with titles, URLs, and snippets:
```json theme={null}
{
"jobId": "962372c4-a140-400b-8c26-4ffe21d9fb9c",
"status": "completed",
"data": {
"query": "hyperbrowser browser automation",
"results": [
{
"title": "Hyperbrowser - Browser Automation Platform",
"url": "https://hyperbrowser.ai",
"description": "Hyperbrowser provides cloud browsers for AI agents and web scraping..."
},
{
"title": "Getting Started with Hyperbrowser",
"url": "https://docs.hyperbrowser.ai/quickstart",
"description": "Learn how to use Hyperbrowser for browser automation..."
}
]
}
}
```
## Parameters
| Parameter | Type | Required | Description |
| - | - | - | - |
| `query` | `string` | Yes | The search query (max 500 characters) |
| `page` | `number` | No | Page number for pagination (1-100, default: 1) |
| `maxAgeSeconds` | `number` | No | Cache control—cached results older than this are treated as stale. Set to `0` to bypass cache reads. |
| `location` | `object` | No | Location hint for localized search results |
| `filters` | `object` | No | Advanced search filters |
### Location Object
| Field | Type | Required | Description |
| - | - | - | - |
| `country` | `string` | Yes | ISO-2 country code (e.g., `"US"`, `"GB"`, `"DE"`) |
| `state` | `string` | No | State code (for supported countries) |
| `city` | `string` | No | City name (max 200 characters) |
### Filters Object
| Field | Type | Description |
| - | - | - |
| `exactPhrase` | `boolean` | Wrap query in quotes for exact match |
| `semanticPhrase` | `boolean` | Use semantic search |
| `excludeTerms` | `string[]` | Terms to exclude from results |
| `boostTerms` | `string[]` | Terms to prioritize in results |
| `filetype` | `string` | Filter by file type: `pdf`, `doc`, `docx`, `xls`, `xlsx`, `ppt`, `pptx`, `html` |
| `site` | `string` | Limit results to a specific site |
| `excludeSite` | `string` | Exclude results from a specific site |
| `intitle` | `string` | Search term must appear in page title |
| `inurl` | `string` | Search term must appear in URL |
`exactPhrase` and `semanticPhrase` cannot both be `true`.
#### Example with filters
```typescript Node theme={null}
import { Hyperbrowser } from "@hyperbrowser/sdk";
import { config } from "dotenv";
config();
const client = new Hyperbrowser({
apiKey: process.env.HYPERBROWSER_API_KEY,
});
const result = await client.web.search({
query: "machine learning tutorials",
page: 1,
location: {
country: "US",
},
filters: {
excludeTerms: ["beginner"],
filetype: "pdf",
site: "arxiv.org",
},
});
```
```python Python 1.0+ theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.search(
{
"query": "machine learning tutorials",
"page": 1,
"location": {"country": "US"},
"filters": {
"exclude_terms": ["beginner"],
"filetype": "pdf",
"site": "arxiv.org",
},
}
)
print(result)
```
```python Python (legacy) theme={null}
import os
from dotenv import load_dotenv
from hyperbrowser import Hyperbrowser
from hyperbrowser.models import WebSearchParams, WebSearchLocation, WebSearchFilters
load_dotenv()
client = Hyperbrowser(api_key=os.getenv("HYPERBROWSER_API_KEY"))
result = client.web.search(
WebSearchParams(
query="machine learning tutorials",
page=1,
location=WebSearchLocation(country="US"),
filters=WebSearchFilters(
exclude_terms=["beginner"],
filetype="pdf",
site="arxiv.org",
),
)
)
print(result)
```
```bash cURL theme={null}
curl -X POST https://api.hyperbrowser.ai/api/web/search \
-H 'Content-Type: application/json' \
-H 'x-api-key: ' \
-d '{
"query": "machine learning tutorials",
"page": 1,
"location": {
"country": "US"
},
"filters": {
"excludeTerms": ["beginner"],
"filetype": "pdf",
"site": "arxiv.org"
}
}'
```
***
## X402 Support
For agentic workflows, Hyperbrowser supports x402 payment for the Search API, enabling agents to programmatically pay for browser sessions at request time using USDC, with payment handled inline via HTTP 402 responses. See the [X402 Integration](/docs/integrations/x402) page for details.