From fcdf3c3de77e21fdf73e4e151dd3a0a531e4d46d Mon Sep 17 00:00:00 2001 From: andrewleesteele <8799863+andrewleesteele@users.noreply.github.com> Date: Tue, 6 Oct 2026 22:03:03 +0000 Subject: [PATCH] Restructure How it works: Configure, Control, Scale, Observe, Manage Co-Authored-By: Claude Opus 5.5 --- apps/deploy.mdx | 85 ++++- apps/develop.mdx | 4 +- apps/invoke.mdx | 67 +++- apps/logs.mdx | 4 +- apps/secrets.mdx | 104 ------ apps/status.mdx | 4 +- apps/stop.mdx | 63 ---- auth/configuration.mdx | 490 ++++++++++++++++++++++++++- auth/connection-lifecycle.mdx | 4 +- auth/credentials.mdx | 474 -------------------------- auth/faq.mdx | 2 +- auth/fill-from-vault.mdx | 2 +- auth/hosted-ui.mdx | 12 +- auth/managed-auth.mdx | 20 +- auth/overview.mdx | 3 +- auth/react.mdx | 4 +- browsers/bot-detection/hcaptcha.mdx | 17 - browsers/bot-detection/overview.mdx | 25 +- browsers/bot-detection/stealth.mdx | 10 +- browsers/computer-controls.mdx | 10 + browsers/concurrency-and-limits.mdx | 68 ++++ browsers/curl.mdx | 1 + browsers/live-view.mdx | 2 +- browsers/payments.mdx | 3 +- browsers/performance.mdx | 2 +- browsers/playwright-execution.mdx | 14 +- browsers/pools.mdx | 6 +- browsers/profiles.mdx | 2 +- browsers/regions.mdx | 2 +- browsers/telemetry/categories.mdx | 1 + browsers/telemetry/export.mdx | 1 + browsers/telemetry/overview.mdx | 1 + browsers/telemetry/streaming.mdx | 1 + changelog.mdx | 10 +- docs.json | 210 ++++++------ info/network-access.mdx | 3 +- info/projects.mdx | 16 +- integrations/wallets/agentcard.mdx | 2 +- integrations/wallets/overview.mdx | 2 +- integrations/wallets/stripe-link.mdx | 2 +- introduction/configure.mdx | 122 +++++++ introduction/control.mdx | 139 ++++---- introduction/create.mdx | 4 +- introduction/manage.mdx | 35 ++ introduction/observe.mdx | 3 +- introduction/scale.mdx | 86 +++-- proxies/custom.mdx | 2 +- proxies/datacenter.mdx | 4 +- proxies/isp.mdx | 2 +- proxies/overview.mdx | 5 +- proxies/residential.mdx | 8 +- reference/cli/managed-auth.mdx | 2 +- vaults/overview.mdx | 16 +- 53 files changed, 1230 insertions(+), 951 deletions(-) delete mode 100644 apps/secrets.mdx delete mode 100644 apps/stop.mdx delete mode 100644 auth/credentials.mdx delete mode 100644 browsers/bot-detection/hcaptcha.mdx create mode 100644 browsers/concurrency-and-limits.mdx create mode 100644 introduction/configure.mdx create mode 100644 introduction/manage.mdx diff --git a/apps/deploy.mdx b/apps/deploy.mdx index 1e411963..e574db30 100644 --- a/apps/deploy.mdx +++ b/apps/deploy.mdx @@ -1,5 +1,7 @@ --- -title: "Deploying" +title: "Deploy an App" +sidebarTitle: "Deploy" +description: "Deploy your app to KERNEL, set environment variables, and pass secrets" --- Kernel's app deployment process is as simple as it is fast. There are no configuration files to manage or complex CI/CD pipelines. @@ -94,6 +96,87 @@ const client = new Kernel({ apiKey: process.env.MY_KERNEL_API_KEY }); Now the API calls your app makes go out as your key. The deployment key stays in place for Kernel's own use — running the invocation and reporting its result — so your key only needs permissions for the calls you actually make. +## Secrets + +Pass API keys and other secrets as [environment variables](#environment-variables) when you deploy, with `--env` or `--env-file`. Then read them in your app: + + +```typescript TypeScript +import Anthropic from "@anthropic-ai/sdk"; +import OpenAI from "openai"; + +app.action('ai-action', async (ctx: KernelContext) => { + // Access API keys from environment variables + const anthropic = new Anthropic({ + apiKey: process.env.ANTHROPIC_API_KEY, + }); + + const openai = new OpenAI({ + apiKey: process.env.OPENAI_API_KEY, + }); + + // Use the clients... +}); +``` + +```python Python +import os +from anthropic import Anthropic +from openai import OpenAI + +@app.action("ai-action") +async def ai_action(ctx: KernelContext): + # Access API keys from environment variables + anthropic = Anthropic( + api_key=os.environ.get("ANTHROPIC_API_KEY"), + ) + + openai = OpenAI( + api_key=os.environ.get("OPENAI_API_KEY"), + ) + + # Use the clients... +``` + + +### Per-invocation secrets + +For use cases where different API keys are needed per invocation (such as platforms using end-user keys), pass the secrets at runtime using the [payload parameter](/apps/invoke#payload-parameter). + +Use encryption standards in your app to protect sensitive data. + + +```typescript TypeScript +import OpenAI from "openai"; + +app.action('ai-action', async (ctx: KernelContext, payload) => { + // Decrypt the API key passed at runtime + const apiKey = decrypt(payload.encryptedApiKey); + + const openai = new OpenAI({ + apiKey: apiKey, + }); + + // Use the client with the user's API key... +}); +``` + +```python Python +from openai import OpenAI + +@app.action("ai-action") +async def ai_action(ctx: KernelContext, payload): + # Decrypt the API key passed at runtime + api_key = decrypt(payload["encryptedApiKey"]) + + openai = OpenAI( + api_key=api_key, + ) + + # Use the client with the user's API key... +``` + + ## Deployment notes - **The dependency manifest (`package.json` for JS/TS, `pyproject.toml` for Python) must be present in the root directory of your project.** diff --git a/apps/develop.mdx b/apps/develop.mdx index f572fac1..a77cdf3e 100644 --- a/apps/develop.mdx +++ b/apps/develop.mdx @@ -1,5 +1,7 @@ --- -title: "Developing" +title: "Develop an App" +sidebarTitle: "Develop" +description: "Build an app with actions that run co-located with KERNEL browsers" --- In addition to our browser API, Kernel provides a code execution platform for deploying and invoking code. Typically, Kernel's code execution platform is used for deploying and invoking browser automations or web agents. diff --git a/apps/invoke.mdx b/apps/invoke.mdx index 6be6b018..424740ba 100644 --- a/apps/invoke.mdx +++ b/apps/invoke.mdx @@ -1,5 +1,7 @@ --- -title: "Invoking" +title: "Invoke an App" +sidebarTitle: "Invoke" +description: "Run an app action from the API or CLI, pass a payload, and stop a running invocation" --- ## Via API @@ -154,4 +156,67 @@ See [here](/apps/develop#parameters) to learn how to access the payload in your If your action specifies a [return value](/apps/develop#return-values), the invocation returns its value once it completes. (The Kernel CLI uses asynchronous invocations under the hood) + +## Stop an invocation + +You can terminate an invocation that's running. This is useful for stopping automations or agents stuck in an infinite loop. + + +Terminating an invocation also destroys any browsers associated with it. + + +### Via API +You can stop an invocation by setting its status to `failed`. This will cancel the invocation and mark it as terminated. + + +```typescript Typescript/Javascript +import Kernel from '@onkernel/sdk'; + +const kernel = new Kernel(); + +const invocation = await kernel.invocations.update('invocation_id', { + status: 'failed', + output: JSON.stringify({ error: 'Invocation cancelled by user' }), +}); +``` + +```python Python +from kernel import Kernel + +kernel = Kernel() +invocation = kernel.invocations.update( + id="invocation_id", + status="failed", + output='{"error":"Invocation cancelled by user"}', +) +``` + +```go Go +package main + +import ( + "context" + + "github.com/kernel/kernel-go-sdk" +) + +func main() { + ctx := context.Background() + client := kernel.NewClient() + + invocation, err := client.Invocations.Update(ctx, "invocation_id", kernel.InvocationUpdateParams{ + Status: kernel.InvocationUpdateParamsStatusFailed, + Output: kernel.String(`{"error":"Invocation cancelled by user"}`), + }) + if err != nil { + panic(err) + } + _ = invocation +} +``` + + +### Via CLI +Use `ctrl-c` in the terminal tab where you launched the invocation. + App invocations accrue compute usage separately from any browsers they create. See the [pricing FAQ](/info/pricing#faq) for how app invocations are charged. diff --git a/apps/logs.mdx b/apps/logs.mdx index 13454391..4995db06 100644 --- a/apps/logs.mdx +++ b/apps/logs.mdx @@ -1,5 +1,7 @@ --- -title: "Logs" +title: "Invocation Logs" +sidebarTitle: "Logs" +description: "Stream an invocation's logs from the API or CLI" --- ## Via API diff --git a/apps/secrets.mdx b/apps/secrets.mdx deleted file mode 100644 index 6b20afbf..00000000 --- a/apps/secrets.mdx +++ /dev/null @@ -1,104 +0,0 @@ ---- -title: "Secrets" ---- - -There are multiple ways to pass secrets and API keys to your Kernel app: - -## 1. Deployment environment variables - -Deploy your app with secrets as [environment variables](/apps/deploy#environment-variables). Your app can then access them at runtime. - -You can set environment variables in two ways: - -- **`--env` flag**: Pass individual key-value pairs directly in the command -- **`--env-file` flag**: Load variables from a `.env` file - -```bash -# Using --env flag for individual variables -kernel deploy my_app.ts --env OPENAI_API_KEY=sk-... --env ANTHROPIC_API_KEY=sk-ant-... - -# Using --env-file to load from a file -kernel deploy my_app.ts --env-file .env - -# Combine both approaches -kernel deploy my_app.ts --env-file .env --env OPENAI_API_KEY=sk-... -``` - -Then access the variables in your app: - - -```typescript TypeScript -import Anthropic from "@anthropic-ai/sdk"; -import OpenAI from "openai"; - -app.action('ai-action', async (ctx: KernelContext) => { - // Access API keys from environment variables - const anthropic = new Anthropic({ - apiKey: process.env.ANTHROPIC_API_KEY, - }); - - const openai = new OpenAI({ - apiKey: process.env.OPENAI_API_KEY, - }); - - // Use the clients... -}); -``` - -```python Python -import os -from anthropic import Anthropic -from openai import OpenAI - -@app.action("ai-action") -async def ai_action(ctx: KernelContext): - # Access API keys from environment variables - anthropic = Anthropic( - api_key=os.environ.get("ANTHROPIC_API_KEY"), - ) - - openai = OpenAI( - api_key=os.environ.get("OPENAI_API_KEY"), - ) - - # Use the clients... -``` - - -## 2. Runtime variables - -For use cases where different API keys are needed per invocation (such as platforms using end-user keys), pass the secrets at runtime using the [payload parameter](/apps/invoke#payload-parameter). - -Use encryption standards in your app to protect sensitive data. - - -```typescript TypeScript -import OpenAI from "openai"; - -app.action('ai-action', async (ctx: KernelContext, payload) => { - // Decrypt the API key passed at runtime - const apiKey = decrypt(payload.encryptedApiKey); - - const openai = new OpenAI({ - apiKey: apiKey, - }); - - // Use the client with the user's API key... -}); -``` - -```python Python -from openai import OpenAI - -@app.action("ai-action") -async def ai_action(ctx: KernelContext, payload): - # Decrypt the API key passed at runtime - api_key = decrypt(payload["encryptedApiKey"]) - - openai = OpenAI( - api_key=api_key, - ) - - # Use the client with the user's API key... -``` - \ No newline at end of file diff --git a/apps/status.mdx b/apps/status.mdx index 21a81988..658bfb10 100644 --- a/apps/status.mdx +++ b/apps/status.mdx @@ -1,5 +1,7 @@ --- -title: "Status" +title: "Invocation Status" +sidebarTitle: "Status" +description: "Follow an invocation's status by streaming or polling" --- Once you've [deployed](/apps/deploy) an app and invoked it, you can monitor its status using streaming for real-time updates or polling for periodic checks. diff --git a/apps/stop.mdx b/apps/stop.mdx deleted file mode 100644 index da53a560..00000000 --- a/apps/stop.mdx +++ /dev/null @@ -1,63 +0,0 @@ ---- -title: "Stopping" ---- - -You can terminate an invocation that's running. This is useful for stopping automations or agents stuck in an infinite loop. - - -Terminating an invocation also destroys any browsers associated with it. - - -## Via API -You can stop an invocation by setting its status to `failed`. This will cancel the invocation and mark it as terminated. - - -```typescript Typescript/Javascript -import Kernel from '@onkernel/sdk'; - -const kernel = new Kernel(); - -const invocation = await kernel.invocations.update('invocation_id', { - status: 'failed', - output: JSON.stringify({ error: 'Invocation cancelled by user' }), -}); -``` - -```python Python -from kernel import Kernel - -kernel = Kernel() -invocation = kernel.invocations.update( - id="invocation_id", - status="failed", - output='{"error":"Invocation cancelled by user"}', -) -``` - -```go Go -package main - -import ( - "context" - - "github.com/kernel/kernel-go-sdk" -) - -func main() { - ctx := context.Background() - client := kernel.NewClient() - - invocation, err := client.Invocations.Update(ctx, "invocation_id", kernel.InvocationUpdateParams{ - Status: kernel.InvocationUpdateParamsStatusFailed, - Output: kernel.String(`{"error":"Invocation cancelled by user"}`), - }) - if err != nil { - panic(err) - } - _ = invocation -} -``` - - -## Via CLI -Use `ctrl-c` in the terminal tab where you launched the invocation. diff --git a/auth/configuration.mdx b/auth/configuration.mdx index e90ad981..e23f28af 100644 --- a/auth/configuration.mdx +++ b/auth/configuration.mdx @@ -6,11 +6,483 @@ description: "Shared options for managed auth connections, regardless of integra Managed Auth connections use the same configuration whether you collect credentials through the [Hosted UI](/auth/hosted-ui), the [React component](/auth/react), or the [programmatic flow](/auth/programmatic). These options apply to the initial login, every background health check, and each automatic reauthentication attempt. -## Credentials and Auto-Reauth +## Credentials and auto-reauth by default, KERNEL saves durable credential fields after a successful login. these can support eligible automatic reauthentication attempts, including totp codes generated from an available secret. submitted one-time codes aren't saved and don't provide access to future codes. if a later login requires user input, your application must start a new interactive login. -To opt out of credential saving, set `save_credentials: false` when creating the connection. See [Credentials](/auth/credentials) for configuration examples. +To opt out of credential saving, set `save_credentials: false` when creating the connection. + +credentials let you store login information securely. KERNEL can attempt automatic reauthentication for eligible flows using stored credentials, including totp codes generated from an available secret. saving credentials or completing an interactive login doesn't guarantee unattended reauthentication. supplying a one-time code doesn't give KERNEL the ability to obtain future codes. if a site requires user input, start a new [interactive login](/auth/connection-lifecycle#flows-that-need-input-a-choice-or-approval). + +There are three ways to provide credentials: +- **Automatically save during login** — Capture credentials directly from the user when they log in via [Hosted UI](/auth/hosted-ui) or [Programmatic](/auth/programmatic) +- **Pre-store in Kernel** — Create credentials before login for supported headless authentication flows +- **Connect 1Password** — Use credentials from your existing 1Password vaults + + + Connect your 1Password vaults to automatically use existing credentials with Managed Auth. Credentials are automatically matched by domain. + + +### Save credentials during login + +By default, Kernel saves durable credential fields entered during login so they can be used for eligible reauthentication attempts. No extra parameters are needed: + + +```typescript TypeScript +const login = await kernel.auth.connections.login(auth.id); +``` + +```python Python +login = await kernel.auth.connections.login(auth.id) +``` + +```go Go +login, err := client.Auth.Connections.Login(ctx, auth.ID, kernel.AuthConnectionLoginParams{}) +if err != nil { + panic(err) +} +_ = login +``` + + +Once saved, the browser profile reuses its authenticated session until the site expires it. For supported credential-based flows, Kernel can then reauthenticate with the stored values. Credentials are updated after every successful login. Submitted one-time codes aren't saved; Kernel generates TOTP codes from a stored `totp_secret`. + +To opt out of credential saving, set `save_credentials: false` when creating the connection: + + +```typescript TypeScript +const auth = await kernel.auth.connections.create({ + domain: 'example.com', + profile_name: 'my-profile', + save_credentials: false, +}); +``` + +```python Python +auth = await kernel.auth.connections.create( + domain="example.com", + profile_name="my-profile", + save_credentials=False, +) +``` + +```go Go +auth, err := client.Auth.Connections.New(ctx, kernel.AuthConnectionNewParams{ + ManagedAuthCreateRequest: kernel.ManagedAuthCreateRequestParam{ + Domain: "example.com", + ProfileName: "my-profile", + SaveCredentials: kernel.Bool(false), + }, +}) +if err != nil { + panic(err) +} +_ = auth +``` + + +### Pre-store credentials + +For credential-based flows that you want to run without user input, create credentials upfront: + + +```typescript TypeScript +const credential = await kernel.credentials.create({ + name: 'my-netflix-login', + domain: 'netflix.com', + values: { + email: 'user@netflix.com', + password: 'secretpassword123', + }, +}); +``` + +```python Python +credential = await kernel.credentials.create( + name="my-netflix-login", + domain="netflix.com", + values={ + "email": "user@netflix.com", + "password": "secretpassword123", + }, +) +``` + +```go Go +credential, err := client.Credentials.New(ctx, kernel.CredentialNewParams{ + CreateCredentialRequest: kernel.CreateCredentialRequestParam{ + Name: "my-netflix-login", + Domain: "netflix.com", + Values: map[string]string{ + "email": "user@netflix.com", + "password": "secretpassword123", + }, + }, +}) +if err != nil { + panic(err) +} +_ = credential +``` + + +Then link the credential when creating a connection: + + +```typescript TypeScript +const auth = await kernel.auth.connections.create({ + domain: 'netflix.com', + profile_name: 'my-profile', + credential: { name: credential.name }, +}); + +// Start login with stored credentials +const login = await kernel.auth.connections.login(auth.id); +``` + +```python Python +auth = await kernel.auth.connections.create( + domain="netflix.com", + profile_name="my-profile", + credential={"name": credential.name}, +) + +# Start login with stored credentials +login = await kernel.auth.connections.login(auth.id) +``` + +```go Go +auth, err := client.Auth.Connections.New(ctx, kernel.AuthConnectionNewParams{ + ManagedAuthCreateRequest: kernel.ManagedAuthCreateRequestParam{ + Domain: "netflix.com", + ProfileName: "my-profile", + Credential: kernel.ManagedAuthCreateRequestCredentialParam{ + Name: kernel.String(credential.Name), + }, + }, +}) +if err != nil { + panic(err) +} + +// Start login with stored credentials +login, err := client.Auth.Connections.Login(ctx, auth.ID, kernel.AuthConnectionLoginParams{}) +if err != nil { + panic(err) +} +_ = login +``` + + +#### 2FA with TOTP + +For sites with authenticator app 2FA, include `totp_secret` so KERNEL can generate a fresh code during automatic login and reauthentication. Supply a base32 secret of 16–128 characters or an `otpauth://totp/` provisioning URI. The default is SHA1, 6 digits, and a 30-second period. If the authenticator uses different settings, provide `totp_algorithm` (`SHA1`, `SHA256`, or `SHA512`), `totp_digits` (6–9), and `totp_period` (15–300 seconds): + + +```typescript TypeScript +const credential = await kernel.credentials.create({ + name: 'my-login', + domain: 'github.com', + values: { + username: 'my-username', + password: 'my-password', + }, + totp_secret: 'JBSWY3DPEHPK3PXP', + totp_algorithm: 'SHA512', + totp_digits: 8, + totp_period: 60, +}); +``` + +```python Python +credential = await kernel.credentials.create( + name="my-login", + domain="github.com", + values={ + "username": "my-username", + "password": "my-password", + }, + totp_secret="JBSWY3DPEHPK3PXP", + totp_algorithm="SHA512", + totp_digits=8, + totp_period=60, +) +``` + +```go Go +credential, err := client.Credentials.New(ctx, kernel.CredentialNewParams{ + CreateCredentialRequest: kernel.CreateCredentialRequestParam{ + Name: "my-login", + Domain: "github.com", + Values: map[string]string{ + "username": "my-username", + "password": "my-password", + }, + TotpSecret: kernel.String("JBSWY3DPEHPK3PXP"), + TotpAlgorithm: kernel.CreateCredentialRequestTotpAlgorithmSha512, + TotpDigits: kernel.Int(8), + TotpPeriod: kernel.Int(60), + }, +}) +if err != nil { + panic(err) +} +_ = credential +``` + + +The examples use typed fields available in TypeScript, Python, and Go SDK v0.116.0 or later. You can also pass an `otpauth://totp/` provisioning URI as `totp_secret`. + +- URI parameters override explicit settings. If a parameter is missing, the API uses its explicit field, then the default. +- Replacing a URI resets omitted settings to defaults. Rotating a raw secret preserves stored settings unless you send new values. +- The API stores only the normalized seed, never the URI label or issuer. +- A code's length follows `totp_digits`; don't assume six digits when reading `totp_code` or calling `totpCode()`. + +#### SSO / OAuth + +For sites with "Sign in with Google/GitHub/Microsoft", set `sso_provider` so Kernel can select the matching SSO route. Automatic completion depends on the provider's login requirements. + +Common SSO provider domains (Google, Microsoft, Okta, Auth0, GitHub, etc.) are allowed by default, so you don't need to add them to `allowed_domains`: + + +```typescript TypeScript +const credential = await kernel.credentials.create({ + name: 'my-google-login', + domain: 'accounts.google.com', + sso_provider: 'google', + values: { + email: 'user@gmail.com', + password: 'password', + }, +}); + +const auth = await kernel.auth.connections.create({ + domain: 'target-site.com', + profile_name: 'my-profile', + credential: { name: credential.name }, +}); +``` + +```python Python +credential = await kernel.credentials.create( + name="my-google-login", + domain="accounts.google.com", + sso_provider="google", + values={ + "email": "user@gmail.com", + "password": "password", + }, +) + +auth = await kernel.auth.connections.create( + domain="target-site.com", + profile_name="my-profile", + credential={"name": credential.name}, +) +``` + +```go Go +credential, err := client.Credentials.New(ctx, kernel.CredentialNewParams{ + CreateCredentialRequest: kernel.CreateCredentialRequestParam{ + Name: "my-google-login", + Domain: "accounts.google.com", + SSOProvider: kernel.String("google"), + Values: map[string]string{ + "email": "user@gmail.com", + "password": "password", + }, + }, +}) +if err != nil { + panic(err) +} + +auth, err := client.Auth.Connections.New(ctx, kernel.AuthConnectionNewParams{ + ManagedAuthCreateRequest: kernel.ManagedAuthCreateRequestParam{ + Domain: "target-site.com", + ProfileName: "my-profile", + Credential: kernel.ManagedAuthCreateRequestCredentialParam{ + Name: kernel.String(credential.Name), + }, + }, +}) +if err != nil { + panic(err) +} +_ = auth +``` + + +### Partial credentials + +Credentials don't need to contain every field required by the login form. You can store what you have and collect the necessary fields from the user. `auth.connections.login()` pauses for missing values. + +As an example, the below credential has email + TOTP secret stored (and automatically handled), but no password. The password is dynamically collected from the user using Kernel's Hosted UI or your Programmatic flow: + + +```typescript TypeScript +const credential = await kernel.credentials.create({ + name: 'my-login', + domain: 'example.com', + values: { email: 'user@example.com' }, // No password + totp_secret: 'JBSWY3DPEHPK3PXP', +}); + +const auth = await kernel.auth.connections.create({ + domain: 'example.com', + profile_name: 'my-profile', + credential: { name: credential.name }, +}); + +const login = await kernel.auth.connections.login(auth.id); + +// Stream state changes and submit the missing password +const authEvents = await kernel.auth.connections.follow(auth.id); +for await (const event of authEvents) { + const passwordField = event.fields?.find(field => field.ref === 'password'); + if ( + event.event === 'managed_auth_state' && + event.flow_step === 'AWAITING_INPUT' && + event.interaction_id && + passwordField + ) { + // Only password is pending; email is filled from the stored credential. + await kernel.auth.connections.submit(auth.id, { + interaction_id: event.interaction_id, + field_values: { [passwordField.id]: 'user-provided-password' }, + }); + } +} +// TOTP auto-submitted from credential → SUCCESS +``` + +```python Python +credential = await kernel.credentials.create( + name="my-login", + domain="example.com", + values={"email": "user@example.com"}, # No password + totp_secret="JBSWY3DPEHPK3PXP", +) + +auth = await kernel.auth.connections.create( + domain="example.com", + profile_name="my-profile", + credential={"name": credential.name}, +) + +login = await kernel.auth.connections.login(auth.id) + +# Stream state changes and submit the missing password +auth_events = await kernel.auth.connections.follow(auth.id) +async for event in auth_events: + password_field = next( + (field for field in (event.fields or []) if field.ref == "password"), + None, + ) + if ( + event.event == "managed_auth_state" + and event.flow_step == "AWAITING_INPUT" + and event.interaction_id + and password_field + ): + # Only password is pending; email is filled from the stored credential. + await kernel.auth.connections.submit( + auth.id, + interaction_id=event.interaction_id, + field_values={password_field.id: "user-provided-password"}, + ) +# TOTP auto-submitted from credential → SUCCESS +``` + +```go Go +credential, err := client.Credentials.New(ctx, kernel.CredentialNewParams{ + CreateCredentialRequest: kernel.CreateCredentialRequestParam{ + Name: "my-login", + Domain: "example.com", + Values: map[string]string{ + "email": "user@example.com", // No password + }, + TotpSecret: kernel.String("JBSWY3DPEHPK3PXP"), + }, +}) +if err != nil { + panic(err) +} + +auth, err := client.Auth.Connections.New(ctx, kernel.AuthConnectionNewParams{ + ManagedAuthCreateRequest: kernel.ManagedAuthCreateRequestParam{ + Domain: "example.com", + ProfileName: "my-profile", + Credential: kernel.ManagedAuthCreateRequestCredentialParam{ + Name: kernel.String(credential.Name), + }, + }, +}) +if err != nil { + panic(err) +} + +login, err := client.Auth.Connections.Login(ctx, auth.ID, kernel.AuthConnectionLoginParams{}) +if err != nil { + panic(err) +} +_ = login + +// Stream state changes and submit the missing password +authEvents := client.Auth.Connections.FollowStreaming(ctx, auth.ID) +for authEvents.Next() { + event := authEvents.Current() + if event.Event != "managed_auth_state" || event.FlowStep != "AWAITING_INPUT" || event.InteractionID == "" { + continue + } + for _, field := range event.Fields { + if field.Ref != "password" { + continue + } + // Only password is pending; email is filled from the stored credential. + _, err := client.Auth.Connections.Submit(ctx, auth.ID, kernel.AuthConnectionSubmitParams{ + SubmitFieldsRequest: kernel.SubmitFieldsRequestParam{ + InteractionID: kernel.String(event.InteractionID), + FieldValues: map[string]string{ + field.ID: "user-provided-password", + }, + }, + }) + if err != nil { + panic(err) + } + break + } +} +if err := authEvents.Err(); err != nil { + panic(err) +} +// TOTP auto-submitted from credential → SUCCESS +``` + + +This is useful when you want to: +- Store TOTP secrets but have users enter their password each time +- Pre-fill username/email but collect password at runtime +- Merge user-provided values into an existing credential automatically on successful login + +### Credential security + +| Feature | Description | +|---------|-------------| +| **Encrypted at rest** | Values encrypted using per-organization keys | +| **Write-only** | Values cannot be retrieved via API after creation | +| **Never logged** | Values are never written to logs | +| **Never shared** | Values are never passed to LLMs | +| **Isolated execution** | Authentication runs in isolated browser environments | + +### Credential notes + +- The `values` object is flexible and can be used to store whatever fields the login form needs (`email`, `username`, `company_id`, etc.) +- Deleting a credential unlinks it from associated connections so they can no longer auto-authenticate +- Use one credential per account. We recommend creating separate credentials for different user accounts + +### Automatic reauthentication Automatic re-authentication is gated by two boolean flags that both default to `true`: @@ -60,7 +532,7 @@ Automatic reauthentication requires a previously successful login and saved cred If Kernel can't complete an automatic attempt, the connection transitions to `NEEDS_AUTH` so you can start a new login. -## Custom Login URL +## Custom login URL If the site's login page isn't at the default location, specify it when creating the connection: @@ -96,7 +568,7 @@ _ = auth ``` -## Browser Region +## Browser region Set `browser.region` to choose where Managed Auth runs the connection's initial login, health checks, and automatic reauthentication. Choose from `us-east`, `eu-west`, and `ap-southeast`. Region selection is available on [Start-Up and Enterprise plans](/info/pricing); omitted values default to `us-east`. @@ -167,7 +639,7 @@ _ = login Browser placement and proxy location are independent. `browser.region` chooses where the browser runs; the connection's [proxy](/proxies/overview) controls the exit IP that websites see. Regional browsers don't provide a data residency guarantee. See [Regional Browsers](/browsers/regions) for storage and processing details. -## SSO/OAuth Support +## SSO/OAuth support Managed Auth supports common "Sign in with Google/GitHub/Microsoft" flows. The user completes the OAuth flow with the provider, and Kernel saves the authenticated session to the profile. Automatic reauthentication depends on the provider's login requirements. See [Can this connection auto-reauth?](/auth/connection-lifecycle#can-this-connection-auto-reauth) for how Kernel determines eligibility. @@ -207,7 +679,7 @@ _ = auth ``` -## Custom Proxy +## Custom proxy Pin the auth flow to a specific [proxy](/proxies/overview) so logins, health checks, and automatic re-authentications all egress through that proxy. This is useful for sites that allowlist IPs, geo-pin sessions, or treat IP changes as a fraud signal. @@ -328,7 +800,7 @@ _ = login ``` -## Record Sessions for Debugging +## Record sessions for debugging Set `record_session: true` to capture a [replay](/browsers/replays) of every browser session tied to the connection — initial logins, background health checks, and automatic re-authentications. The entire browser session is recorded. @@ -393,7 +865,7 @@ _ = login Managed auth recordings are subject to the same retention rules as other session replay recordings. Each managed auth session row stores its own `replay_id` for the recording captured during that session. -## Post-Login URL +## Post-login URL After successful authentication, `post_login_url` will be set to the page where the login landed. Use this to start your automation from the right place: @@ -433,7 +905,7 @@ if managedAuth.PostLoginURL != "" { ``` -## Updating a Connection +## Updating a connection After creating a connection, you can update its configuration with `auth.connections.update`: diff --git a/auth/connection-lifecycle.mdx b/auth/connection-lifecycle.mdx index 5cb4904a..3b194279 100644 --- a/auth/connection-lifecycle.mdx +++ b/auth/connection-lifecycle.mdx @@ -226,7 +226,7 @@ See the [API reference](https://kernel.sh/docs/api-reference/managed-auth/start- ### Recovering -- **`credentials_invalid`** — Update the linked [credential](/auth/credentials) and call `.login()` to re-run the flow. When the site identifies which field it rejected during an interactive login, Kernel asks for a corrected value in place — see [replacing a rejected credential](/auth/programmatic#replacing-a-rejected-credential). +- **`credentials_invalid`** — Update the linked [credential](/auth/configuration#credentials-and-auto-reauth) and call `.login()` to re-run the flow. When the site identifies which field it rejected during an interactive login, Kernel asks for a corrected value in place — see [replacing a rejected credential](/auth/programmatic#replacing-a-rejected-credential). - **`totp_code_rejected`** — Retry with a code from a new TOTP window. If independently generated codes keep failing, reconnect the account and update its TOTP secret. One rejected code does not prove that the saved secret is stale. - **`totp_required` / `sms_code_required` / `email_code_required`** — Start an interactive login and provide the requested code. Add a TOTP secret to the linked credential to make future authenticator-code challenges automatic. - **`account_choice_required` / `customer_input_required` / `external_action_required`** — Start an interactive login and complete the choice, field, or external approval. Kernel does not guess an identity or trigger notification-producing steps during unattended reauth. @@ -274,5 +274,5 @@ To record every auth session on the connection (logins, health checks, and reaut ## See also - [Connection Configuration](/auth/configuration) — `health_check_interval`, `proxy`, `record_session`, and other shared options -- [Credentials](/auth/credentials) — what gets stored and how it powers auto-reauth +- [Credentials](/auth/configuration#credentials-and-auto-reauth) — what gets stored and how it powers auto-reauth - [FAQ](/auth/faq) — quick answers to common questions diff --git a/auth/credentials.mdx b/auth/credentials.mdx deleted file mode 100644 index 2f253f3a..00000000 --- a/auth/credentials.mdx +++ /dev/null @@ -1,474 +0,0 @@ ---- -title: "Managed Auth Credentials" -description: "Use stored credentials for login and eligible automatic reauthentication attempts" ---- - -credentials let you store login information securely. KERNEL can attempt automatic reauthentication for eligible flows using stored credentials, including totp codes generated from an available secret. saving credentials or completing an interactive login doesn't guarantee unattended reauthentication. supplying a one-time code doesn't give KERNEL the ability to obtain future codes. if a site requires user input, start a new [interactive login](/auth/connection-lifecycle#flows-that-need-input-a-choice-or-approval). - -**There are three ways to provide credentials:** -- **Automatically save during login** — Capture credentials directly from the user when they log in via [Hosted UI](/auth/hosted-ui) or [Programmatic](/auth/programmatic) -- **Pre-store in Kernel** — Create credentials before login for supported headless authentication flows -- **Connect 1Password** — Use credentials from your existing 1Password vaults - - - Connect your 1Password vaults to automatically use existing credentials with Managed Auth. Credentials are automatically matched by domain. - - -## Save credentials during login - -By default, Kernel saves durable credential fields entered during login so they can be used for eligible reauthentication attempts. No extra parameters are needed: - - -```typescript TypeScript -const login = await kernel.auth.connections.login(auth.id); -``` - -```python Python -login = await kernel.auth.connections.login(auth.id) -``` - -```go Go -login, err := client.Auth.Connections.Login(ctx, auth.ID, kernel.AuthConnectionLoginParams{}) -if err != nil { - panic(err) -} -_ = login -``` - - -Once saved, the browser profile reuses its authenticated session until the site expires it. For supported credential-based flows, Kernel can then reauthenticate with the stored values. Credentials are updated after every successful login. Submitted one-time codes aren't saved; Kernel generates TOTP codes from a stored `totp_secret`. - -To opt out of credential saving, set `save_credentials: false` when creating the connection: - - -```typescript TypeScript -const auth = await kernel.auth.connections.create({ - domain: 'example.com', - profile_name: 'my-profile', - save_credentials: false, -}); -``` - -```python Python -auth = await kernel.auth.connections.create( - domain="example.com", - profile_name="my-profile", - save_credentials=False, -) -``` - -```go Go -auth, err := client.Auth.Connections.New(ctx, kernel.AuthConnectionNewParams{ - ManagedAuthCreateRequest: kernel.ManagedAuthCreateRequestParam{ - Domain: "example.com", - ProfileName: "my-profile", - SaveCredentials: kernel.Bool(false), - }, -}) -if err != nil { - panic(err) -} -_ = auth -``` - - -## Pre-store credentials - -For credential-based flows that you want to run without user input, create credentials upfront: - - -```typescript TypeScript -const credential = await kernel.credentials.create({ - name: 'my-netflix-login', - domain: 'netflix.com', - values: { - email: 'user@netflix.com', - password: 'secretpassword123', - }, -}); -``` - -```python Python -credential = await kernel.credentials.create( - name="my-netflix-login", - domain="netflix.com", - values={ - "email": "user@netflix.com", - "password": "secretpassword123", - }, -) -``` - -```go Go -credential, err := client.Credentials.New(ctx, kernel.CredentialNewParams{ - CreateCredentialRequest: kernel.CreateCredentialRequestParam{ - Name: "my-netflix-login", - Domain: "netflix.com", - Values: map[string]string{ - "email": "user@netflix.com", - "password": "secretpassword123", - }, - }, -}) -if err != nil { - panic(err) -} -_ = credential -``` - - -Then link the credential when creating a connection: - - -```typescript TypeScript -const auth = await kernel.auth.connections.create({ - domain: 'netflix.com', - profile_name: 'my-profile', - credential: { name: credential.name }, -}); - -// Start login with stored credentials -const login = await kernel.auth.connections.login(auth.id); -``` - -```python Python -auth = await kernel.auth.connections.create( - domain="netflix.com", - profile_name="my-profile", - credential={"name": credential.name}, -) - -# Start login with stored credentials -login = await kernel.auth.connections.login(auth.id) -``` - -```go Go -auth, err := client.Auth.Connections.New(ctx, kernel.AuthConnectionNewParams{ - ManagedAuthCreateRequest: kernel.ManagedAuthCreateRequestParam{ - Domain: "netflix.com", - ProfileName: "my-profile", - Credential: kernel.ManagedAuthCreateRequestCredentialParam{ - Name: kernel.String(credential.Name), - }, - }, -}) -if err != nil { - panic(err) -} - -// Start login with stored credentials -login, err := client.Auth.Connections.Login(ctx, auth.ID, kernel.AuthConnectionLoginParams{}) -if err != nil { - panic(err) -} -_ = login -``` - - -### 2FA with TOTP - -For sites with authenticator app 2FA, include `totp_secret` so KERNEL can generate a fresh code during automatic login and reauthentication. Supply a base32 secret of 16–128 characters or an `otpauth://totp/` provisioning URI. The default is SHA1, 6 digits, and a 30-second period. If the authenticator uses different settings, provide `totp_algorithm` (`SHA1`, `SHA256`, or `SHA512`), `totp_digits` (6–9), and `totp_period` (15–300 seconds): - - -```typescript TypeScript -const credential = await kernel.credentials.create({ - name: 'my-login', - domain: 'github.com', - values: { - username: 'my-username', - password: 'my-password', - }, - totp_secret: 'JBSWY3DPEHPK3PXP', - totp_algorithm: 'SHA512', - totp_digits: 8, - totp_period: 60, -}); -``` - -```python Python -credential = await kernel.credentials.create( - name="my-login", - domain="github.com", - values={ - "username": "my-username", - "password": "my-password", - }, - totp_secret="JBSWY3DPEHPK3PXP", - totp_algorithm="SHA512", - totp_digits=8, - totp_period=60, -) -``` - -```go Go -credential, err := client.Credentials.New(ctx, kernel.CredentialNewParams{ - CreateCredentialRequest: kernel.CreateCredentialRequestParam{ - Name: "my-login", - Domain: "github.com", - Values: map[string]string{ - "username": "my-username", - "password": "my-password", - }, - TotpSecret: kernel.String("JBSWY3DPEHPK3PXP"), - TotpAlgorithm: kernel.CreateCredentialRequestTotpAlgorithmSha512, - TotpDigits: kernel.Int(8), - TotpPeriod: kernel.Int(60), - }, -}) -if err != nil { - panic(err) -} -_ = credential -``` - - -The examples use typed fields available in TypeScript, Python, and Go SDK v0.116.0 or later. You can also pass an `otpauth://totp/` provisioning URI as `totp_secret`. - -- URI parameters override explicit settings. If a parameter is missing, the API uses its explicit field, then the default. -- Replacing a URI resets omitted settings to defaults. Rotating a raw secret preserves stored settings unless you send new values. -- The API stores only the normalized seed, never the URI label or issuer. -- A code's length follows `totp_digits`; don't assume six digits when reading `totp_code` or calling `totpCode()`. - -### SSO / OAuth - -For sites with "Sign in with Google/GitHub/Microsoft", set `sso_provider` so Kernel can select the matching SSO route. Automatic completion depends on the provider's login requirements. - -Common SSO provider domains (Google, Microsoft, Okta, Auth0, GitHub, etc.) are allowed by default, so you don't need to add them to `allowed_domains`: - - -```typescript TypeScript -const credential = await kernel.credentials.create({ - name: 'my-google-login', - domain: 'accounts.google.com', - sso_provider: 'google', - values: { - email: 'user@gmail.com', - password: 'password', - }, -}); - -const auth = await kernel.auth.connections.create({ - domain: 'target-site.com', - profile_name: 'my-profile', - credential: { name: credential.name }, -}); -``` - -```python Python -credential = await kernel.credentials.create( - name="my-google-login", - domain="accounts.google.com", - sso_provider="google", - values={ - "email": "user@gmail.com", - "password": "password", - }, -) - -auth = await kernel.auth.connections.create( - domain="target-site.com", - profile_name="my-profile", - credential={"name": credential.name}, -) -``` - -```go Go -credential, err := client.Credentials.New(ctx, kernel.CredentialNewParams{ - CreateCredentialRequest: kernel.CreateCredentialRequestParam{ - Name: "my-google-login", - Domain: "accounts.google.com", - SSOProvider: kernel.String("google"), - Values: map[string]string{ - "email": "user@gmail.com", - "password": "password", - }, - }, -}) -if err != nil { - panic(err) -} - -auth, err := client.Auth.Connections.New(ctx, kernel.AuthConnectionNewParams{ - ManagedAuthCreateRequest: kernel.ManagedAuthCreateRequestParam{ - Domain: "target-site.com", - ProfileName: "my-profile", - Credential: kernel.ManagedAuthCreateRequestCredentialParam{ - Name: kernel.String(credential.Name), - }, - }, -}) -if err != nil { - panic(err) -} -_ = auth -``` - - -## Partial Credentials - -Credentials don't need to contain every field required by the login form. You can store what you have and collect the necessary fields from the user. `auth.connections.login()` pauses for missing values. - -As an example, the below credential has email + TOTP secret stored (and automatically handled), but no password. The password is dynamically collected from the user using Kernel's Hosted UI or your Programmatic flow: - - -```typescript TypeScript -const credential = await kernel.credentials.create({ - name: 'my-login', - domain: 'example.com', - values: { email: 'user@example.com' }, // No password - totp_secret: 'JBSWY3DPEHPK3PXP', -}); - -const auth = await kernel.auth.connections.create({ - domain: 'example.com', - profile_name: 'my-profile', - credential: { name: credential.name }, -}); - -const login = await kernel.auth.connections.login(auth.id); - -// Stream state changes and submit the missing password -const authEvents = await kernel.auth.connections.follow(auth.id); -for await (const event of authEvents) { - const passwordField = event.fields?.find(field => field.ref === 'password'); - if ( - event.event === 'managed_auth_state' && - event.flow_step === 'AWAITING_INPUT' && - event.interaction_id && - passwordField - ) { - // Only password is pending; email is filled from the stored credential. - await kernel.auth.connections.submit(auth.id, { - interaction_id: event.interaction_id, - field_values: { [passwordField.id]: 'user-provided-password' }, - }); - } -} -// TOTP auto-submitted from credential → SUCCESS -``` - -```python Python -credential = await kernel.credentials.create( - name="my-login", - domain="example.com", - values={"email": "user@example.com"}, # No password - totp_secret="JBSWY3DPEHPK3PXP", -) - -auth = await kernel.auth.connections.create( - domain="example.com", - profile_name="my-profile", - credential={"name": credential.name}, -) - -login = await kernel.auth.connections.login(auth.id) - -# Stream state changes and submit the missing password -auth_events = await kernel.auth.connections.follow(auth.id) -async for event in auth_events: - password_field = next( - (field for field in (event.fields or []) if field.ref == "password"), - None, - ) - if ( - event.event == "managed_auth_state" - and event.flow_step == "AWAITING_INPUT" - and event.interaction_id - and password_field - ): - # Only password is pending; email is filled from the stored credential. - await kernel.auth.connections.submit( - auth.id, - interaction_id=event.interaction_id, - field_values={password_field.id: "user-provided-password"}, - ) -# TOTP auto-submitted from credential → SUCCESS -``` - -```go Go -credential, err := client.Credentials.New(ctx, kernel.CredentialNewParams{ - CreateCredentialRequest: kernel.CreateCredentialRequestParam{ - Name: "my-login", - Domain: "example.com", - Values: map[string]string{ - "email": "user@example.com", // No password - }, - TotpSecret: kernel.String("JBSWY3DPEHPK3PXP"), - }, -}) -if err != nil { - panic(err) -} - -auth, err := client.Auth.Connections.New(ctx, kernel.AuthConnectionNewParams{ - ManagedAuthCreateRequest: kernel.ManagedAuthCreateRequestParam{ - Domain: "example.com", - ProfileName: "my-profile", - Credential: kernel.ManagedAuthCreateRequestCredentialParam{ - Name: kernel.String(credential.Name), - }, - }, -}) -if err != nil { - panic(err) -} - -login, err := client.Auth.Connections.Login(ctx, auth.ID, kernel.AuthConnectionLoginParams{}) -if err != nil { - panic(err) -} -_ = login - -// Stream state changes and submit the missing password -authEvents := client.Auth.Connections.FollowStreaming(ctx, auth.ID) -for authEvents.Next() { - event := authEvents.Current() - if event.Event != "managed_auth_state" || event.FlowStep != "AWAITING_INPUT" || event.InteractionID == "" { - continue - } - for _, field := range event.Fields { - if field.Ref != "password" { - continue - } - // Only password is pending; email is filled from the stored credential. - _, err := client.Auth.Connections.Submit(ctx, auth.ID, kernel.AuthConnectionSubmitParams{ - SubmitFieldsRequest: kernel.SubmitFieldsRequestParam{ - InteractionID: kernel.String(event.InteractionID), - FieldValues: map[string]string{ - field.ID: "user-provided-password", - }, - }, - }) - if err != nil { - panic(err) - } - break - } -} -if err := authEvents.Err(); err != nil { - panic(err) -} -// TOTP auto-submitted from credential → SUCCESS -``` - - -This is useful when you want to: -- Store TOTP secrets but have users enter their password each time -- Pre-fill username/email but collect password at runtime -- Merge user-provided values into an existing credential automatically on successful login - -## Security - -| Feature | Description | -|---------|-------------| -| **Encrypted at rest** | Values encrypted using per-organization keys | -| **Write-only** | Values cannot be retrieved via API after creation | -| **Never logged** | Values are never written to logs | -| **Never shared** | Values are never passed to LLMs | -| **Isolated execution** | Authentication runs in isolated browser environments | - -## Notes - -- The `values` object is flexible and can be used to store whatever fields the login form needs (`email`, `username`, `company_id`, etc.) -- Deleting a credential unlinks it from associated connections so they can no longer auto-authenticate -- Use one credential per account. We recommend creating separate credentials for different user accounts diff --git a/auth/faq.mdx b/auth/faq.mdx index 525ba9ce..360696c4 100644 --- a/auth/faq.mdx +++ b/auth/faq.mdx @@ -61,4 +61,4 @@ See [Reuse one identity across sites](/browsers/profiles/agent-patterns#reuse-on Managed Auth is included on all plans with no per-connection fees. It uses browser sessions for login, health checks, and eligible reauthentication attempts. These count toward your browser usage like any other browser session. -Auth sessions are fast (typically 5-30 seconds each). Kernel monitors session health and can automatically reauthenticate eligible credential-based flows when sessions expire. Most sessions stay valid for days. For example, monitoring 100 auth connections typically costs less than $5/month in browser usage. See [Pricing & Limits](/info/pricing#managed-auth) for details. +Auth sessions are fast (typically 5-30 seconds each). Kernel monitors session health and can automatically reauthenticate eligible credential-based flows when sessions expire. Most sessions stay valid for days. For example, monitoring 100 auth connections typically costs less than $5/month in browser usage. See [Pricing](/info/pricing#faq) for details. diff --git a/auth/fill-from-vault.mdx b/auth/fill-from-vault.mdx index f81bbf4c..675d208b 100644 --- a/auth/fill-from-vault.mdx +++ b/auth/fill-from-vault.mdx @@ -1,5 +1,5 @@ --- -title: "Overview" +title: "Fill from Vault" description: "Collect end-user credentials and inject them into browser forms while controlling the login workflow" --- diff --git a/auth/hosted-ui.mdx b/auth/hosted-ui.mdx index 5b510123..bcb0e6f8 100644 --- a/auth/hosted-ui.mdx +++ b/auth/hosted-ui.mdx @@ -12,7 +12,7 @@ Use the Hosted UI when: ## Getting started -### 1. Create a Connection +### 1. Create a connection A Managed Auth connection saves a domain's authentication state to a [profile](/browsers/profiles) so future browsers can reuse it. You can attach multiple auth connections to the same profile, one per domain. @@ -45,7 +45,7 @@ _ = auth ``` -### 2. Start a Login Session +### 2. Start a login session Start a Managed Auth Session to get the hosted login URL. @@ -67,7 +67,7 @@ _ = login ``` -### 3. Collect Credentials +### 3. Collect credentials Send the user to the hosted login page: @@ -152,7 +152,7 @@ if authenticated { The SSE stream closes automatically when the flow succeeds, fails, expires, or is canceled. The session expires after 20 minutes if not completed, and the flow times out after 10 minutes of waiting for user input. -### 5. Use the Profile +### 5. Use the profile Create browsers with the profile and navigate to the site. The browser loads the authentication state saved during login: @@ -203,7 +203,7 @@ Managed Auth Connections are generated using Kernel's [stealth](/browsers/bot-de -## Complete Example +## Complete example ```typescript TypeScript @@ -420,6 +420,6 @@ func main() { The hosted page redirects to whatever URL you pass. Only set these from your own trusted backend — never let an end user supply them directly. -## Connection Configuration +## Connection configuration Connection-level options — custom login URL, SSO/OAuth, custom proxy, session recording, post-login URL, and updates — apply equally to all integration flows and are documented in [Connection Configuration](/auth/configuration). diff --git a/auth/managed-auth.mdx b/auth/managed-auth.mdx index 8d1a0f4d..b4126407 100644 --- a/auth/managed-auth.mdx +++ b/auth/managed-auth.mdx @@ -1,5 +1,6 @@ --- -title: "Overview" +title: "Managed Auth Overview" +sidebarTitle: "Overview" description: "Handle website login, reuse session state, and recover eligible connections automatically" --- @@ -51,7 +52,7 @@ _ = auth A **Managed Auth Session** is the corresponding login flow for the specified connection. Users provide credentials via a KERNEL-hosted page or your own UI. - link a [credential](/auth/credentials) so KERNEL can attempt reauthentication when the connection is eligible. stored credentials alone don't make every flow eligible. + link a [credential](/auth/configuration#credentials-and-auto-reauth) so KERNEL can attempt reauthentication when the connection is eligible. stored credentials alone don't make every flow eligible. ```typescript TypeScript @@ -195,6 +196,21 @@ these steps establish the initial connection. your integration must also handle +## Connection options + +Every connection uses the same options, whichever integration you choose. See [configuration](/auth/configuration) for examples of each. + +| Option | What it does | +| --- | --- | +| [Credentials](/auth/configuration#credentials-and-auto-reauth) | Save login values during the first login, or store them ahead of time, so KERNEL can attempt eligible reauthentication. | +| [Automatic reauthentication](/auth/configuration#automatic-reauthentication) | Run health checks and attempt to reauthenticate eligible connections when a session expires, controlled by `health_checks` and `auto_reauth`. | +| [Custom login URL](/auth/configuration#custom-login-url) | Start the login from a specific page instead of the domain's default. | +| [Browser region](/auth/configuration#browser-region) | Run the login and health checks in the region your browsers use. | +| [SSO and OAuth](/auth/configuration#ssooauth-support) | Support "Sign in with Google, GitHub, or Microsoft" flows. Common identity providers are allowed by default. | +| [Custom proxy](/auth/configuration#custom-proxy) | Send login traffic through a proxy you choose. | +| [Session recording](/auth/configuration#record-sessions-for-debugging) | Record login attempts as replays for debugging. | +| [Post-login URL](/auth/configuration#post-login-url) | Read the page the login landed on, so your automation starts from the right place. | + ## Why Managed Auth? Managed Auth runs **login flows** by navigating login pages, filling credentials, following SSO redirects, and guiding users through additional authentication steps. It saves the resulting session state to a reusable profile. diff --git a/auth/overview.mdx b/auth/overview.mdx index 99893cd2..177a9f78 100644 --- a/auth/overview.mdx +++ b/auth/overview.mdx @@ -1,5 +1,6 @@ --- -title: "Overview" +title: "Authentication Overview" +sidebarTitle: "Overview" description: "Choose how your browser agents authenticate and reuse signed-in sessions" --- diff --git a/auth/react.mdx b/auth/react.mdx index 6beaee7f..beecc847 100644 --- a/auth/react.mdx +++ b/auth/react.mdx @@ -19,7 +19,7 @@ bun add @onkernel/managed-auth-react ## Getting started -### 1. Start a Login Session on your backend +### 1. Start a login session on your backend Same as the Hosted UI flow — create a connection and start a login. The login response returns the connection `id` and a one-time `handoff_code`; those are the two values you'll hand to the component on the frontend. @@ -270,7 +270,7 @@ import { Wrap them in `` and `` to inherit the same styling/localization plumbing as the all-in-one component. -## Connection Configuration +## Connection configuration Connection-level options — custom login URL, SSO/OAuth, custom proxy, session recording, post-login URL, and updates — are set on `auth.connections.create` (or later via `auth.connections.update`) and apply equally regardless of which integration flow you use. See [Connection Configuration](/auth/configuration). diff --git a/browsers/bot-detection/hcaptcha.mdx b/browsers/bot-detection/hcaptcha.mdx deleted file mode 100644 index f1763b96..00000000 --- a/browsers/bot-detection/hcaptcha.mdx +++ /dev/null @@ -1,17 +0,0 @@ ---- -title: "hCaptcha" ---- - -Kernel's hCaptcha solver is a beta feature for teams that need help handling hCaptcha challenges in browser automations. - -When enabled for your organization, Kernel can attempt to solve supported hCaptcha challenges automatically from Kernel browsers. This is useful for permitted automation where hCaptcha appears as part of a normal browser workflow, such as QA, account operations, or user-authorized agent tasks. - - -The hCaptcha solver is in beta and isn't enabled for all organizations by default. - - -## Get access - -To use the hCaptcha solver, [contact Kernel support](https://www.kernel.sh/docs/info/support) and ask to have the hCaptcha beta enabled for your organization. - -Include the website or workflow you're testing, your expected volume, and whether you're already using [stealth mode](/browsers/bot-detection/stealth), [profiles](/browsers/profiles), or custom [proxies](/proxies/overview). This helps us confirm the right setup for your use case. diff --git a/browsers/bot-detection/overview.mdx b/browsers/bot-detection/overview.mdx index 44c23a27..100c4ce0 100644 --- a/browsers/bot-detection/overview.mdx +++ b/browsers/bot-detection/overview.mdx @@ -1,5 +1,6 @@ --- -title: "Overview" +title: "Stealth Overview" +sidebarTitle: "Overview" description: "Help your browser agents access websites with anti-detection defaults, stealth mode, proxies, and opt-in Web Bot Auth (WBA)." --- @@ -12,7 +13,7 @@ Under the hood, our browsers are optimized for realistic environments. **Everyth This guide explains how bot detection works at a high level, common pitfalls to avoid, and how Kernel's features can help your automations run reliably. -## How Bot Detection Works +## How bot detection works Most detection systems look for inconsistencies between how a real user's browser behaves and how an automated one does. Common giveaways include: - **IP addresses**: IPs from data centers (AWS, GCP, Azure) @@ -24,11 +25,11 @@ Most detection systems look for inconsistencies between how a real user's browse These systems are heuristic and probabilistic — small mismatches can still trigger blocks. The goal isn't to “beat” detection but rather emulate the real-world conditions of a normal browser session. -## Kernel Features That Help +## KERNEL features that help ### Anti-detection defaults Every Kernel browser launches with anti-detection chrome configuration applied. No setup required. -### [Stealth Mode](/browsers/bot-detection/stealth) +### [Stealth mode](/browsers/bot-detection/stealth) On top of the defaults, stealth mode adds a default ISP proxy and an automatic CAPTCHA solver. Both are opt-out so you can BYO proxy and/or CAPTCHA tooling. ### [Web Bot Auth (WBA)](/browsers/bot-detection/web-bot-auth) @@ -39,27 +40,27 @@ WBA lets your agent sign requests with a verifiable identity. Participating webs ### [Config Registry](/config-registry) Kernel recommended browser and proxy configurations for websites. -### [Configurable Proxies](/proxies/overview) +### [Configurable proxies](/proxies/overview) Bring your own proxy network or use Kernel's managed proxy pool (selectable down to ZIP-code level). If needed, use the same IP to reduce detection and allow for regional testing or QA. ### [Profiles](/browsers/profiles) Profiles persist cookies, local storage, and session data between runs. Combined with a fixed proxy, this mimics a returning user. We recommend using them to persist authenticated states and reduce CAPTCHAs. -### [Browser Pools](/browsers/pools) +### [Browser pools](/browsers/pools) Browser pools let you reuse browsers across multiple visits to the same website, which introduces consistency with respect to the IP address. Since IP addresses are one of the main components of fingerprinting used by modern bot detection systems, browser pools drastically increase your chances of avoiding detection. -### [Playwright Execution API](/browsers/playwright-execution) +### [Playwright execution API](/browsers/playwright-execution) Executes Playwright scripts in the same VM as the browser, ensuring headers, user-agent strings, and environment match. Kernel automatically applies Patchright to remove automation fingerprints, including headless indicators. -### [Computer Controls API](/browsers/computer-controls) +### [Computer controls API](/browsers/computer-controls) Controls the browser without using the Chrome DevTools Protocol (CDP), which can reduce bot detection signals. Emulates native keyboard and mouse input directly at the OS level and includes human-like [bezier curves](/browsers/computer-controls#move-the-mouse) by default. -### [GPU Acceleration](/browsers/gpu-acceleration) +### [GPU acceleration](/browsers/gpu-acceleration) Many detection systems fingerprint canvas and WebGL rendering output and cross-check it against the claimed GPU. Software-rendered browsers produce pixel hashes that don't match any real consumer GPU, which is a strong bot signal on sites with rendering-based fingerprinting. GPU-enabled Kernel browsers render through real hardware, producing output consistent with a normal user's device. -## Getting Started +## Getting started Before you start automating your workflow, we recommend that you manually test your website to understand how it behaves with Kernel's browsers. Here's how to do that: @@ -73,7 +74,7 @@ Before you start automating your workflow, we recommend that you manually test y Once you have a stable baseline, replicate those conditions in your automations. -## Recommended Practices +## Recommended practices | Category | Recommendation | |-----------|----------------| @@ -88,7 +89,7 @@ Once you have a stable baseline, replicate those conditions in your automations. | **Network Identity** | Use stable IP addresses, especially if logging in. See [Choosing a proxy type](#choosing-a-proxy-type) below. | | **Extensions** | Use the [Extensions API](/browsers/extensions) carefully — each adds its own fingerprint, which can be detected. | -## Choosing a Proxy Type +## Choosing a proxy type IP address is one of the strongest signals bot detection systems use. Kernel offers several [proxy types](/proxies/overview), each with different trade-offs for detection avoidance. diff --git a/browsers/bot-detection/stealth.mdx b/browsers/bot-detection/stealth.mdx index 2613bab8..216316da 100644 --- a/browsers/bot-detection/stealth.mdx +++ b/browsers/bot-detection/stealth.mdx @@ -108,20 +108,24 @@ _ = kernelBrowser If you're looking for proxy-level configuration with Kernel browsers, see [Proxies](/proxies/overview). -## CAPTCHA Handling Behavior +## CAPTCHA handling behavior Below are tips for working with Kernel's Stealth Mode auto-CAPTCHA solver across different challenge types and automation frameworks. -### Anthropic Computer Use +### Anthropic computer use Anthropic Computer Use stops when it encounters a CAPTCHA. Use Kernel's auto-CAPTCHA solver by adding this to your prompt: `"If you see a CAPTCHA or similar test, just wait for it to get solved automatically by the browser."` -### Cloudflare Challenge +### Cloudflare challenge When encountering a Cloudflare challenge, our auto-CAPTCHA solver will attempt to handle it. Once the "Ready" message appears on the screen, continue with your intended browser actions (e.g., entering credentials and submitting a login attempt). After the "Ready" message appears, don't click the Cloudflare CAPTCHA checkbox — this can interfere with the solver. + +### hCaptcha (beta) + +KERNEL can also attempt to solve supported hCaptcha challenges automatically. The hCaptcha solver is in beta and isn't enabled for every organization by default. To turn it on, [contact support](/info/support) with the website or workflow you're testing, your expected volume, and whether you already use stealth mode, [profiles](/browsers/profiles), or custom [proxies](/proxies/overview). diff --git a/browsers/computer-controls.mdx b/browsers/computer-controls.mdx index 98471df0..07b6d994 100644 --- a/browsers/computer-controls.mdx +++ b/browsers/computer-controls.mdx @@ -5,6 +5,16 @@ description: "Control the computer's mouse, keyboard, and screen" Use OS-level controls to move and click the mouse, type and press keys, scroll, drag, and capture screenshots from a running browser session. Both `moveMouse` and `dragMouse` use human-like [Bézier curves](https://en.wikipedia.org/wiki/B%C3%A9zier_curve) by default. +## Why computer use for agents + +Kernel's computer controls are built to match how computer-use models were trained — the same primitives the model emits (screenshot, click at coords, type, key, scroll, drag) map 1:1 onto the API. There's no harness translating model output into framework calls. + +- **Native fit.** Screenshot, click, type, key, scroll, drag — the primitives the model already speaks. +- **Faster screenshots.** Captures bypass CDP, which removes the largest source of latency in a vision loop. +- **Better against bot detection.** No CDP connection means no CDP fingerprint to leak. Pairs naturally with [stealth mode](/browsers/bot-detection/stealth) and [residential proxies](/proxies/residential). +- **Human-like input.** OS-level events with Bézier-curve mouse paths, variable typing speed, and configurable mistype rate. +- **Not DOM-limited.** Screenshots capture the full VM, so the agent can see and interact with native dialogs, canvas elements, iframes, and PDFs — not just things you can address with a selector. + ## Click the mouse Simulate mouse clicks at specific coordinates. You can select the button, click type (down, up, click), number of clicks, and optional modifier keys to hold. diff --git a/browsers/concurrency-and-limits.mdx b/browsers/concurrency-and-limits.mdx new file mode 100644 index 00000000..d83c71b0 --- /dev/null +++ b/browsers/concurrency-and-limits.mdx @@ -0,0 +1,68 @@ +--- +title: "Concurrency and Limits" +description: "How many browsers you can run, how fast you can create them, and what each one gets" +--- + +Three separate limits shape a scaled workload, and they're easy to confuse. Concurrency caps how many browsers exist at once. The create rate caps how fast you can ask for new ones. Per-browser resources cap what one browser can do. + +## Concurrency + +One org-wide limit covers every browser you're running, whether created on demand with `browsers.create()` or reserved in a [browser pool](/browsers/pools). The full limit is available to either API in any mix. + +| Limit | Developer | Hobbyist | Start-Up | Enterprise | +| --- | --- | --- | --- | --- | +| Concurrent browsers | 5 | 10 | 150 | Custom | +| App invocations | 5 | 10 | 50 | Custom | +| App invocations (per app) | 5 | 10 | 20 | Custom | +| Managed auth health check interval | 6 hours minimum | 1 hour minimum | 20 minutes minimum | Custom | + +Limits are org-wide unless stated otherwise. + +Two things count against the concurrent browser limit that people don't expect: + +- **Reserved pool capacity counts whether or not it's acquired.** A pool sized to 40 browsers uses 40 of your limit for as long as it exists. +- **Browsers in [standby](/browsers/standby) count.** Standby stops usage charges, not the concurrency slot. Delete the browser to release it. + +Set per-project caps if you're splitting one org limit across teams or environments — see [project concurrency limits](/info/projects#concurrency-limits). + +## Rate limits + +Kernel enforces per-organization rate limits on API requests. Browser creation is rate limited separately from concurrency: the create rate caps how fast you can create browsers, not how many you may run. + +{/* TODO: add the per-plan browser-create rate table once the numbers are confirmed. */} + +Acquiring from a [browser pool](/browsers/pools) isn't subject to the create rate — the pool's browsers already exist. If your traffic arrives in bursts, that's the reason to use a pool even when your concurrency headroom is fine. + +### What happens at the limit + +Exceeding a rate limit returns `429 Too Many Requests`. Rate-limited endpoints include these headers on every response: + +| Header | Description | +| --- | --- | +| `X-RateLimit-Limit` | Maximum requests allowed per minute | +| `X-RateLimit-Remaining` | Requests remaining in the current window | +| `Retry-After` | Seconds to wait before retrying (only on `429` responses) | + +All Kernel SDKs retry a `429` up to 2 times, honoring `Retry-After`. If retries are exhausted, the SDK raises a typed `RateLimitError` carrying the response headers, so you can apply your own backoff. Queue on your side rather than tightening the retry loop: a `429` means the org is over budget for the minute, so retrying faster doesn't help. + +If you're hitting the ceiling in normal operation, [contact us](https://calendly.com/d/d3tn-5kp-5yt) — the limit is raisable. + +## Per-browser resources + +| Resource | Headful | Headless | +| --- | --- | --- | +| Default memory | 8 GB | 1 GB | + +Memory is the practical ceiling on how many tabs and how heavy a page one browser handles. A [headless](/browsers/headless) browser at 1 GB is sized for short-lived, single-page, high-concurrency automation; open a dozen heavy tabs in one and Chromium starts killing renderers. If your workload needs many concurrent pages, spread it across more browsers rather than more tabs in one. Headful browsers support up to 16 GB: set `memory` on a [browser pool](/browsers/pools) when a workload needs more than the default. + +[GPU acceleration](/browsers/gpu-acceleration) is a separate browser type with its own [usage rate](/info/pricing#usage-rates), available on Start-Up and Enterprise. + +## Other limits worth knowing + +| Limit | Where | +| --- | --- | +| Browser `timeout_seconds` (default 60, max 259200 / 72h) | [Termination](/browsers/termination) | +| Pool `timeout_seconds` (default 600) and fill rate | [Browser pools](/browsers/pools) | +| How managed auth health checks run | [Connection lifecycle](/auth/connection-lifecycle) | +| Replay retention, extensions, projects, per plan | [Pricing](/info/pricing#managed-infrastructure) | +| Monthly spend | [Spending caps](/info/spending-caps) | diff --git a/browsers/curl.mdx b/browsers/curl.mdx index e7f372d0..7bc0e680 100644 --- a/browsers/curl.mdx +++ b/browsers/curl.mdx @@ -1,5 +1,6 @@ --- title: "Curl" +sidebarTitle: "Browser Curl" description: "Send HTTP requests through Kernel browsers" --- diff --git a/browsers/live-view.mdx b/browsers/live-view.mdx index 62862928..2682dff7 100644 --- a/browsers/live-view.mdx +++ b/browsers/live-view.mdx @@ -76,7 +76,7 @@ If your environment restricts outbound traffic, allow the [Live View domains and To enable clipboard sharing, add `allow="autoplay; clipboard-read; clipboard-write"` to the iframe element. -If your application uses a **Content Security Policy (CSP)**, you must add the following directives to allow the live view iframe and its WebSocket connection. See [Network access](/info/network-access#content-security-policy) for the complete firewall and CSP requirements. +If your application uses a **Content Security Policy (CSP)**, you must add the following directives to allow the live view iframe and its WebSocket connection. See [firewall allowlist](/info/network-access#content-security-policy) for the complete firewall and CSP requirements. ``` frame-src https://*.onkernel.com:8443 diff --git a/browsers/payments.mdx b/browsers/payments.mdx index eab30059..b5cdcee2 100644 --- a/browsers/payments.mdx +++ b/browsers/payments.mdx @@ -1,5 +1,6 @@ --- -title: "Payments" +title: "Payments Overview" +sidebarTitle: "Overview" description: "Let browser agents complete purchases without handling raw payment details" --- diff --git a/browsers/performance.mdx b/browsers/performance.mdx index bc6583e0..7b6be4d3 100644 --- a/browsers/performance.mdx +++ b/browsers/performance.mdx @@ -19,7 +19,7 @@ Kernel browsers run in `us-east`. Use our [app platform](/apps/develop) to coloc 2. Create browser rate limit -Kernel enforces [rate limits](/info/pricing#rate-limiting) on browser creation based on your plan. Our SDKs automatically retry, respecting the `Retry-After` header for delay timing. If retries are exhausted, the SDK throws a typed `RateLimitError` with the response headers accessible for custom backoff logic. +Kernel enforces [rate limits](/browsers/concurrency-and-limits#rate-limits) on browser creation based on your plan. Our SDKs automatically retry, respecting the `Retry-After` header for delay timing. If retries are exhausted, the SDK throws a typed `RateLimitError` with the response headers accessible for custom backoff logic. 3. Non-default browser configurations diff --git a/browsers/playwright-execution.mdx b/browsers/playwright-execution.mdx index 92991154..0bb66a33 100644 --- a/browsers/playwright-execution.mdx +++ b/browsers/playwright-execution.mdx @@ -416,14 +416,16 @@ fs.writeFileSync('screenshot.png', buffer); For OS-level screenshots using coordinates and regions, see [Computer Controls](/browsers/computer-controls#take-screenshots). -## Performance benefits + +## Why playwright execution over a direct CDP connection -Compared to connecting over CDP: -- **Lower latency** - Code runs in the same VM as the browser -- **Higher throughput** - No websocket overhead for commands -- **Simpler code** - No need to manage CDP connections +If you're reaching for Playwright, prefer the execution API over `connectOverCDP`. Same Playwright API you already know, none of the setup. -This makes it ideal for one-off operations where you need maximum speed. +- **Run from anywhere.** No `playwright` package to version-pin, no Chromium download, no CDP connection to manage. Send the code, get the result. +- **Co-located with the browser.** Code runs in the same VM as the browser — no network hop between your script and the page, fewer flakes. +- **Patchright by default.** Hardened against bot detection out of the box. +- **Full Playwright API.** `page`, `context`, and `browser` are all in scope. Anything Playwright can do — DOM queries, file uploads, full-page screenshots — works here. +- **Returns values.** `return` from your code and the result comes back in the response. Easy to use as an agent tool. ## MCP server integration diff --git a/browsers/pools.mdx b/browsers/pools.mdx index d83958ee..ceccaf35 100644 --- a/browsers/pools.mdx +++ b/browsers/pools.mdx @@ -5,9 +5,9 @@ description: "Configure a pool of ready-to-use browsers for instant acquisition" A browser pool is a fixed set of identical browsers that Kernel keeps running for you. Configure it once — stealth, proxies, [private networking](/browsers/private-networking), extensions, viewport, a [profile](#profiles-with-browser-pools) — then acquire a browser whenever a task needs one and release it when you're done. -Acquiring is faster than creating an on-demand browser because the browser is already running: you skip start-up, including the [Chromium restart](/browsers/performance#troubleshooting-latency) that some settings trigger, and you aren't subject to the [rate limit](/info/pricing#rate-limiting) on browser creation. +Acquiring is faster than creating an on-demand browser because the browser is already running: you skip start-up, including the [Chromium restart](/browsers/performance#troubleshooting-latency) that some settings trigger, and you aren't subject to the [rate limit](/browsers/concurrency-and-limits#rate-limits) on browser creation. -Idle browsers in a pool aren't billed, but the pool's capacity counts against your [concurrency limit](/info/pricing#concurrency-limits). See [Scale](/introduction/scale) for how browser pools fit into production architecture. +Idle browsers in a pool aren't billed, but the pool's capacity counts against your [concurrency limit](/browsers/concurrency-and-limits#concurrency). See [Scale](/introduction/scale) for how browser pools fit into production architecture. ## How browser pools work @@ -30,7 +30,7 @@ A few constraints to weigh before moving a workload onto a browser pool: - **No GPU browsers.** GPU-accelerated browsers are on-demand only. Use `browsers.create()` for WebGL, video, or canvas-heavy work. - **One fixed configuration per browser pool**, with `start_url` the only setting you can override per acquisition — see [Create a browser pool](#create-a-browser-pool). - **A profile set on the browser pool loads read-only**, and a browser pool holds one at a time — see [Per-user profiles with browser pools](#per-user-profiles-with-browser-pools) for how to persist state per user. -- **Browser pool capacity counts against your [concurrency limit](/info/pricing#concurrency-limits)** whether or not its browsers are acquired, though idle pooled browsers aren't billed. +- **Browser pool capacity counts against your [concurrency limit](/browsers/concurrency-and-limits#concurrency)** whether or not its browsers are acquired, though idle pooled browsers aren't billed. - **Plan-gated.** Browser pools are available on the Start-Up and Enterprise plans. ## Create a browser pool diff --git a/browsers/profiles.mdx b/browsers/profiles.mdx index de1198be..71e4137b 100644 --- a/browsers/profiles.mdx +++ b/browsers/profiles.mdx @@ -1,5 +1,5 @@ --- -title: "Browser Profiles" +title: "Profiles Overview" sidebarTitle: "Overview" description: "Persist and reuse browser state across browser sessions" --- diff --git a/browsers/regions.mdx b/browsers/regions.mdx index db348d63..726753f3 100644 --- a/browsers/regions.mdx +++ b/browsers/regions.mdx @@ -175,7 +175,7 @@ Browser pool lists support the same filter. Omit it to list resources across all - **Profiles and extensions** aren't tied to a region. You can reuse your existing [profiles](/browsers/profiles) and [extensions](/browsers/extensions) with browsers in any region within the same project. - **Proxy configurations** can be reused across regions. Browser region chooses where the browser runs; [proxy location](/proxies/overview) controls the exit IP websites see. -- **Concurrency and rate limits** apply across all regions combined, not separately in each region. Browser pool capacity counts toward the same [concurrency limit](/info/pricing#concurrency-limits). +- **Concurrency and rate limits** apply across all regions combined, not separately in each region. Browser pool capacity counts toward the same [concurrency limit](/browsers/concurrency-and-limits#concurrency). Regional browsers reduce interaction latency; they don't provide a data residency guarantee. Profiles, replays, and session metadata aren't confined to the browser's selected region and may be stored or processed in the US. diff --git a/browsers/telemetry/categories.mdx b/browsers/telemetry/categories.mdx index fede1c6a..0a20b1cf 100644 --- a/browsers/telemetry/categories.mdx +++ b/browsers/telemetry/categories.mdx @@ -1,5 +1,6 @@ --- title: "Telemetry Categories" +sidebarTitle: "Categories" description: "The categories a browser session can capture, what each contains, and their cost" --- diff --git a/browsers/telemetry/export.mdx b/browsers/telemetry/export.mdx index 6cc11018..39c11bf2 100644 --- a/browsers/telemetry/export.mdx +++ b/browsers/telemetry/export.mdx @@ -1,5 +1,6 @@ --- title: "Export Telemetry" +sidebarTitle: "Export" description: "Send a session's captured events to your own observability backend over OTLP" --- diff --git a/browsers/telemetry/overview.mdx b/browsers/telemetry/overview.mdx index f98f5b3c..94ba1a04 100644 --- a/browsers/telemetry/overview.mdx +++ b/browsers/telemetry/overview.mdx @@ -1,5 +1,6 @@ --- title: "Telemetry Overview" +sidebarTitle: "Overview" description: "Capture what happens inside a browser session" --- diff --git a/browsers/telemetry/streaming.mdx b/browsers/telemetry/streaming.mdx index 5f064ade..f096f856 100644 --- a/browsers/telemetry/streaming.mdx +++ b/browsers/telemetry/streaming.mdx @@ -1,5 +1,6 @@ --- title: "Stream Telemetry" +sidebarTitle: "Streaming" description: "Consume a session's live telemetry stream from the SDK or CLI" --- diff --git a/changelog.mdx b/changelog.mdx index 57a2837e..0fd233cf 100644 --- a/changelog.mdx +++ b/changelog.mdx @@ -168,7 +168,7 @@ For API library updates, see the [Node SDK](https://github.com/onkernel/kernel-n ## Documentation updates -- Added a [Foreman integration guide](/integrations/vercel/foreman) to the Vercel section. +- Added a [Foreman integration guide](/cookbooks/eve-foreman) to the Vercel section. - Documented [replay iframe embedding](/browsers/replays) for embedding session replays inside your own dashboards. - Documented the 30-day retention window for browser telemetry events. - Documented automatic [proxy](/proxies/overview) cleanup behavior. @@ -396,7 +396,7 @@ For API library updates, see the [Node SDK](https://github.com/onkernel/kernel-n ## Documentation updates -- Added a new [hCaptcha](/browsers/bot-detection/hcaptcha) page documenting beta support for hCaptcha solving. +- Added a new [hCaptcha](/browsers/bot-detection/stealth#hcaptcha-beta) page documenting beta support for hCaptcha solving. - Refreshed [managed auth](/auth/managed-auth) documentation for May 2026: new dedicated [connection lifecycle](/auth/connection-lifecycle) page covering health checks and re-authentication, a shared [connection configuration](/auth/configuration) reference, documented `success_url` / `error_url` query parameters for the [hosted UI](/auth/hosted-ui), [`start_url`](/browsers/create-a-browser) references across browser and pool docs, a reorganized sidebar, and new FAQ entries for short-session reauth and multi-step login forms. - Clarified that managed residential proxy IPs are stable within a session but are not guaranteed to persist across sessions. - Updated the [Yutori integration guide](/integrations/computer-use/yutori) to Navigator n1.5. @@ -446,7 +446,7 @@ For API library updates, see the [Node SDK](https://github.com/onkernel/kernel-n - Anti-detection features are now on by default for all Kernel browsers. [Stealth mode](/browsers/bot-detection/stealth) now specifically adds the managed ISP proxy and CAPTCHA solver (both opt-out), and non-stealth browsers fully support [custom proxies](/proxies/custom). - Revamped the [managed auth hosted login page](/auth/hosted-ui): all available sign-in options (password fields, SSO providers, MFA, alternate sign-in methods) now render together on a single page, so end users can pick the path they want, instead of being funneled through one at a time. - Managed auth input fields now display contextual helper text when the site surfaces hints, reducing user confusion on multi-step logins. -- The "Save credentials after login" option is now automatically disabled when [1Password](/integrations/1password) is selected as the [credential source](/auth/credentials), since those credentials are already managed externally. +- The "Save credentials after login" option is now automatically disabled when [1Password](/integrations/1password) is selected as the [credential source](/auth/configuration#credentials-and-auto-reauth), since those credentials are already managed externally. - Kernel now supports WebSocket connections through its API, enabling `process attach` and other long-lived streaming workflows. - Exceeding invocation concurrency limits now returns a 429 response immediately, for faster and more actionable feedback. @@ -501,7 +501,7 @@ For API library updates, see the [Node SDK](https://github.com/onkernel/kernel-n ## Documentation updates -- Added [API rate limiting](/info/pricing#rate-limiting) documentation. +- Added [API rate limiting](/browsers/concurrency-and-limits#rate-limits) documentation. - Documented the [`disable_default_proxy`](/browsers/bot-detection/stealth) option for stealth browsers. - Updated [live view embedding](/browsers/live-view) docs with iframe focus tips, clipboard sharing guidance, and CSP configuration. - Documented [managed auth re-authentication triggers](/auth/managed-auth). @@ -610,7 +610,7 @@ For API library updates, see the [Node SDK](https://github.com/onkernel/kernel-n - Fixed screen resize accuracy by removing unnecessary rounding in `ChangeScreenSize` to ensure pixel-perfect display dimensions. ## Documentation updates -- Enhanced [secrets](/apps/secrets) documentation with practical examples for LLM-powered applications and detailed guidance for deploying apps with environment file configurations. +- Enhanced [secrets](/apps/deploy#secrets) documentation with practical examples for LLM-powered applications and detailed guidance for deploying apps with environment file configurations. diff --git a/docs.json b/docs.json index d7aceec5..c80efd86 100644 --- a/docs.json +++ b/docs.json @@ -6,6 +6,9 @@ { "source": "/careers/backend-engineer", "destination": "https://jobs.ashbyhq.com/usekernel" }, { "source": "/careers/engineer-new-grad", "destination": "https://jobs.ashbyhq.com/usekernel" }, { "source": "/careers/customer-engineer", "destination": "https://jobs.ashbyhq.com/usekernel" }, + { "source": "/browsers/bot-detection/hcaptcha", "destination": "/browsers/bot-detection/stealth#hcaptcha-beta" }, + { "source": "/apps/secrets", "destination": "/apps/deploy#secrets" }, + { "source": "/apps/stop", "destination": "/apps/invoke#stop-an-invocation" }, { "source": "/auth/agent/overview", "destination": "/auth/managed-auth" }, { "source": "/auth/agent/hosted-ui", "destination": "/auth/hosted-ui" }, { "source": "/auth/agent/programmatic", "destination": "/auth/programmatic" }, @@ -19,8 +22,9 @@ { "source": "/profiles.md", "destination": "/browsers/profiles.md" }, { "source": "/profiles/overview", "destination": "/browsers/profiles" }, { "source": "/profiles/overview.md", "destination": "/browsers/profiles.md" }, - { "source": "/profiles/credentials", "destination": "/auth/credentials" }, - { "source": "/profiles/credentials.md", "destination": "/auth/credentials.md" }, + { "source": "/profiles/credentials", "destination": "/auth/configuration#credentials-and-auto-reauth" }, + { "source": "/profiles/credentials.md", "destination": "/auth/configuration.md" }, + { "source": "/auth/credentials", "destination": "/auth/configuration#credentials-and-auto-reauth" }, { "source": "/profiles/managed-auth", "destination": "/auth/managed-auth" }, { "source": "/profiles/managed-auth.md", "destination": "/auth/managed-auth.md" }, { "source": "/profiles/managed-auth/overview", "destination": "/auth/managed-auth" }, @@ -103,6 +107,9 @@ } ] }, + "seo": { + "indexing": "all" + }, "navigation": { "tabs": [ { @@ -111,25 +118,36 @@ { "group": "Overview", "pages": [ - "index", - "introduction/create", - "introduction/control", - "introduction/observe", - "introduction/scale" + "index" ] }, { - "group": "Working with your browser", + "group": "How it works", "pages": [ { - "group": "Basics", - "expanded": true, + "group": "Configure", "pages": [ - "browsers/live-view", - "browsers/termination", - "browsers/standby", - "browsers/headless", - "info/projects", + "introduction/configure", + { + "group": "Stealth", + "pages": [ + "browsers/bot-detection/overview", + "browsers/bot-detection/stealth", + "browsers/bot-detection/web-bot-auth" + ] + }, + { + "group": "Proxies", + "pages": [ + "proxies/overview", + "proxies/residential", + "proxies/isp", + "proxies/mobile", + "proxies/datacenter", + "proxies/custom", + "proxies/errors" + ] + }, { "group": "Profiles", "pages": [ @@ -138,45 +156,22 @@ "browsers/profiles/concurrency", "browsers/profiles/agent-patterns" ] - } - ] - }, - { - "group": "Intermediate", - "expanded": true, - "pages": [ + }, { - "group": "Bot Anti-Detection", + "group": "Vaults", "pages": [ - "browsers/bot-detection/overview", - "browsers/bot-detection/stealth", - "browsers/bot-detection/hcaptcha", - { - "group": "Proxies", - "pages": [ - "proxies/overview", - "proxies/custom", - "proxies/residential", - "proxies/mobile", - "proxies/isp", - "proxies/datacenter", - "proxies/errors" - ] - }, - "browsers/bot-detection/web-bot-auth", - "bots" + "vaults/overview", + "vaults/credentials", + "vaults/existing-credential-vault", + "vaults/fill", + "vaults/1password" ] }, { - "group": "Auth", + "group": "Authentication", "pages": [ "auth/overview", - { - "group": "Fill from Vault", - "pages": [ - "auth/fill-from-vault" - ] - }, + "auth/fill-from-vault", { "group": "Managed Auth", "pages": [ @@ -185,64 +180,96 @@ "auth/react", "auth/programmatic", "auth/configuration", - "auth/connection-lifecycle", - "auth/credentials", - "auth/faq" + "auth/connection-lifecycle" ] } ] }, { - "group": "Vaults", + "group": "Payments", "pages": [ - "vaults/overview", - "vaults/credentials", - "vaults/existing-credential-vault", - "vaults/fill", - "vaults/1password" + "browsers/payments", + "integrations/wallets/overview", + "integrations/wallets/stripe-link", + "integrations/wallets/agentcard" ] }, - "browsers/payments", - "browsers/replays", - "browsers/viewport", - "browsers/regions", - "browsers/gpu-acceleration", - "config-registry", - "info/api-keys", - "info/audit-logs", - "browsers/file-io", - "browsers/process-execution", - "browsers/curl", - "browsers/ssh", - "browsers/computer-controls", + "config-registry" + ] + }, + { + "group": "Control", + "pages": [ + "introduction/control", "browsers/playwright-execution", + "browsers/computer-controls", "browsers/webmcp", - "browsers/repl" + "browsers/repl", + "browsers/process-execution", + "browsers/file-io", + { + "group": "Code Execution Platform", + "pages": [ + "apps/develop", + "apps/deploy", + "apps/invoke", + "apps/status", + "apps/logs" + ] + } ] }, { - "group": "Advanced", - "expanded": true, + "group": "Scale", "pages": [ - "browsers/extensions", - "browsers/private-networking", - "browsers/chrome-policies", + "introduction/scale", + "browsers/concurrency-and-limits", + "browsers/performance", + "browsers/pools" + ] + }, + { + "group": "Observe", + "pages": [ + "introduction/observe", + "browsers/live-view", + "browsers/replays", { - "group": "Telemetry", + "group": "Browser Telemetry", "pages": [ "browsers/telemetry/overview", "browsers/telemetry/categories", "browsers/telemetry/streaming", "browsers/telemetry/export" ] - }, - "browsers/pools" + } ] }, { - "group": "FAQ", + "group": "Manage", + "pages": [ + "introduction/manage", + "info/projects", + "info/api-keys", + "info/audit-logs", + "info/network-access" + ] + } + ] + }, + { + "group": "Working with your browser", + "pages": [ + { + "group": "Intermediate", + "expanded": true, "pages": [ - "browsers/performance" + { + "group": "Bot Anti-Detection", + "pages": [ + "bots" + ] + } ] } ] @@ -275,15 +302,6 @@ "integrations/computer-use/yutori" ] }, - { - "group": "Wallets", - "icon": "/images/integration-icons/payments.svg", - "pages": [ - "integrations/wallets/overview", - "integrations/wallets/stripe-link", - "integrations/wallets/agentcard" - ] - }, "integrations/docker-sandboxes", "integrations/e2e", "integrations/hermes-agent", @@ -317,18 +335,6 @@ "integrations/1password" ] }, - { - "group": "deploying your agent", - "pages": [ - "apps/develop", - "apps/deploy", - "apps/invoke", - "apps/stop", - "apps/secrets", - "apps/status", - "apps/logs" - ] - }, { "group": "Agent Skills", "pages": [ @@ -350,8 +356,6 @@ { "group": "Info", "pages": [ - "info/network-access", - "browsers/faq", "info/concepts", "info/zero-data-retention", "info/pricing", diff --git a/info/network-access.mdx b/info/network-access.mdx index 64fe89cf..44ea7259 100644 --- a/info/network-access.mdx +++ b/info/network-access.mdx @@ -1,5 +1,6 @@ --- -title: "Network Access" +title: "Allowlist KERNEL Domains" +sidebarTitle: "Firewall Allowlist" description: "Domain and port allowlist for connecting to Kernel" --- diff --git a/info/projects.mdx b/info/projects.mdx index b9f37a88..9757b052 100644 --- a/info/projects.mdx +++ b/info/projects.mdx @@ -5,13 +5,13 @@ description: "Organize resources and isolate access within your Kernel organizat A **Project** is a named container for Kernel resources inside an organization. Use projects to separate environments (like `production` and `staging`), split resources between teams, or isolate customer workloads — each project has its own browsers, profiles, credentials, proxies, extensions, deployments, and browser pools. -## Why Projects? +## Why projects? - **Isolate environments** — keep `production` resources apart from `staging` or experiments. - **Scope access** — issue API keys that can only see resources in one project. - **Concurrency limits** — set an org-wide default cap for every project, or override it per project, so one team or environment can't exhaust your org quota. -## The Default Project +## The default project Every organization has at least one project. Resources that existed before projects were introduced have been moved into a project named **Default**, so your existing browsers, apps, profiles, and other resources continue to work without any changes on your end. @@ -23,7 +23,7 @@ Your organization must always have **at least one active project**. The API retu A project must also be empty before it can be deleted. If active resources remain, the API returns `409 Conflict` with code `project_not_empty`; delete or otherwise remove those resources and retry. Organizations without Projects enabled receive `404 Not Found` with code `projects_disabled` from project-management endpoints. -## Scoping Requests to a Project +## Scoping requests to a project Pass the `X-Kernel-Project-Id` header with a project ID on any API request to scope it to a specific project. Project names are not accepted in this header. Without the header (and without a project-scoped API key), requests act on your organization's **default project**: reads return the default project's resources, and writes create resources in it. @@ -98,7 +98,7 @@ func main() { ``` -## Authentication and Project Scope +## Authentication and project scope ### API keys @@ -112,7 +112,7 @@ API keys can be **org-wide** or **project-scoped**. OAuth tokens (used by the Kernel CLI and MCP server) are **always org-wide**. You cannot bind an OAuth session to a single project. To scope OAuth-authenticated requests, send the `X-Kernel-Project-Id` header with each request — or use the CLI's `--project` flag (see below). -## Using Projects from the CLI +## Using projects from the CLI The Kernel [CLI](/reference/cli/projects) has first-class project support: @@ -136,7 +136,7 @@ kernel projects limits set staging --max-concurrent-sessions 5 Under the hood, `--project` (or the env var) adds the `X-Kernel-Project-Id` header to every authenticated request. It's the recommended way to target a specific project when you're logged in with OAuth (`kernel login`), since OAuth itself is always org-wide. -## Managing Projects +## Managing projects Use the `/org/projects` REST endpoints (or the SDKs' `projects` resource) to manage projects. @@ -297,11 +297,11 @@ if err := client.Projects.Delete(ctx, "proj_abc123"); err != nil { Project deletion is a soft delete. A project that still owns active resources returns `project_not_empty`; the final active project returns `last_active_project`. -## Concurrency Limits +## Concurrency limits Kernel caps how many browsers can run at once, at two levels. A single limit covers both on-demand browsers (`browsers.create()`) and [browser pools](/browsers/pools) — standalone sessions and pool capacity count against the same cap. -- **Organization limit** — the total concurrent browsers allowed across your whole organization, determined by your plan. Every browser session and every browser in a browser pool counts against it. +- **Organization limit** — the total concurrent browsers allowed across your whole organization, determined by your plan. See [concurrency and limits](/browsers/concurrency-and-limits#concurrency) for each plan's limit. Every browser session and every browser in a browser pool counts against it. - **Per-project limits** — optional caps on individual projects, so one team or environment can't consume the entire org limit. Per-project caps come from two places: diff --git a/integrations/wallets/agentcard.mdx b/integrations/wallets/agentcard.mdx index 9c50e37a..b390875c 100644 --- a/integrations/wallets/agentcard.mdx +++ b/integrations/wallets/agentcard.mdx @@ -1,5 +1,5 @@ --- -title: "Agentcard" +title: "AgentCard" description: "Use Agentcard to approve browser checkouts against an enrolled payment method" --- diff --git a/integrations/wallets/overview.mdx b/integrations/wallets/overview.mdx index b23fd4cf..e48f15e9 100644 --- a/integrations/wallets/overview.mdx +++ b/integrations/wallets/overview.mdx @@ -1,5 +1,5 @@ --- -title: "Overview" +title: "Wallets" description: "Connect user wallets through KERNEL's native integrations" --- diff --git a/integrations/wallets/stripe-link.mdx b/integrations/wallets/stripe-link.mdx index 187332c1..69f7347c 100644 --- a/integrations/wallets/stripe-link.mdx +++ b/integrations/wallets/stripe-link.mdx @@ -1,5 +1,5 @@ --- -title: "link by stripe" +title: "Link by Stripe" description: "use link by stripe to approve a one-use payment credential for a browser checkout" --- diff --git a/introduction/configure.mdx b/introduction/configure.mdx new file mode 100644 index 00000000..32a6ea68 --- /dev/null +++ b/introduction/configure.mdx @@ -0,0 +1,122 @@ +--- +title: "Configure Overview" +sidebarTitle: "Overview" +description: "Create a browser, pick its shape, and choose what it carries onto a site" +mode: "wide" +--- + +A KERNEL browser works with no configuration: create one and drive it. This page covers the settings most agents touch at creation time. The rest of Configure covers the features that decide whether an agent gets through a real site: stealth, proxies, profiles, vaults, authentication, and payments. + +## Create a browser + + +```typescript TypeScript +import Kernel from '@onkernel/sdk'; + +const kernel = new Kernel(); + +const browser = await kernel.browsers.create(); +console.log(browser.session_id, browser.browser_live_view_url); +``` + +```python Python +from kernel import Kernel + +kernel = Kernel() + +browser = kernel.browsers.create() +print(browser.session_id, browser.browser_live_view_url) +``` + +```go Go +package main + +import ( + "context" + "fmt" + + "github.com/kernel/kernel-go-sdk" +) + +func main() { + ctx := context.Background() + client := kernel.NewClient() + + browser, err := client.Browsers.New(ctx, kernel.BrowserNewParams{}) + if err != nil { + panic(err) + } + fmt.Println(browser.SessionID, browser.BrowserLiveViewURL) +} +``` + +```bash CLI +kernel browsers create +``` + + +The response includes everything you need to drive the browser: `session_id`, `cdp_ws_url`, `webdriver_ws_url`, and `browser_live_view_url`. See [create a browser](/introduction/create) for the full walkthrough, including creating from a [browser pool](/browsers/pools). + +## Pick a browser type + +| Type | When to use it | How to set it | +| --- | --- | --- | +| **Headful** (default) | Agents on real sites. It has a real display, so [live view](/browsers/live-view) and [replays](/browsers/replays) work and bot detectors see a normal browser. 8 GB of memory by default. | Nothing to set | +| **[Headless](/browsers/headless)** | Short-lived or highly concurrent jobs that don't need to be watched. Lighter, at 1 GB by default, but some bot detectors notice it. | `headless: true` | +| **[GPU-accelerated](/browsers/gpu-acceleration)** | WebGL, video, and canvas-heavy sites. Headful only, doesn't support standby, and has its own usage rate. | `gpu: true` | + +## Set common options + + +```typescript TypeScript +const browser = await kernel.browsers.create({ + viewport: { width: 1280, height: 800 }, + timeout_seconds: 300, +}); +``` + +```python Python +browser = kernel.browsers.create( + viewport={"width": 1280, "height": 800}, + timeout_seconds=300, +) +``` + + +- **[Viewport](/browsers/viewport):** defaults to 1920x1080 at 25Hz. A custom viewport restarts Chromium on creation, so use a [browser pool](/browsers/pools) if you need it to be instant. +- **[Termination and timeouts](/browsers/termination):** `timeout_seconds` sets how long a browser can sit in standby before KERNEL deletes it. It defaults to 60 seconds and can be up to 72 hours. Delete browsers explicitly when you're done; Playwright's `browser.close()` doesn't delete them. + +## More browser settings + +Often agents don't require these, but they're there when you do: + +- **[Regions](/browsers/regions):** run browsers in `us-east`, `eu-west`, or `ap-southeast`, closer to your code and your users. +- **[Extensions](/browsers/extensions):** load unpacked Chrome extensions into a browser. +- **[Chrome policies](/browsers/chrome-policies):** apply Chrome enterprise policies, such as startup pages and bookmarks. +- **[Private networking](/browsers/private-networking):** reach services behind a VPN or tunnel from inside the browser session. + +## What to configure next + + + + Anti-detection defaults, stealth mode, CAPTCHA handling, and Web Bot Auth. + + + Route traffic through datacenter, ISP, residential, or mobile IPs, or bring your own. + + + Save cookies, storage, and logins from one session and load them into the next. + + + Store credentials and payment items that a browser fills into a page without your agent reading them. + + + Fill logins from a vault, or let managed auth handle the login and attempt to reauthenticate eligible connections. + + + Let agents complete checkouts through a wallet without handling raw card details. + + + Get browser and proxy settings that have already worked on the site you're automating. + + diff --git a/introduction/control.mdx b/introduction/control.mdx index 5ec902ce..3cd9caae 100644 --- a/introduction/control.mdx +++ b/introduction/control.mdx @@ -1,11 +1,36 @@ --- -title: "Control" -description: "Drive the browser with computer use, playwright execution, CDP, or WebDriver BiDi" +title: "Control Overview" +sidebarTitle: "Overview" +description: "Choose a control surface and where your agent loop runs" --- -Kernel browsers expose four ways to drive a session. For agents, we recommend starting with playwright execution and falling back to computer use, here's our guide: [playwright w/ computer use fallback](/browsers/playwright-computer-use-fallback). +You make two choices before you write any automation. They're independent, but the first constrains the second: -Both run co-located with the browser and avoid the bot-detection surface a direct CDP connection introduces. +1. **How you drive the browser** — the control surface your code or model uses to act on the page. +2. **Where the loop runs** — the machine your decision-making code runs on, relative to the browser. + + + +## 1. How you drive the browser + +KERNEL browsers accept several control surfaces. Pick by what's driving the page, not by what you already know. + +| Surface | Use it when | Trade-off | +| --- | --- | --- | +| [Playwright execution](/browsers/playwright-execution) | **Default.** You know what to do on the page — navigate, fill, extract, upload. | Needs a selector or DOM path that exists, and each call is stateless. | +| [Computer controls](/browsers/computer-controls) | **Recommended fallback.** A model is looking at pixels, or the page can't be driven programmatically. | Slower per step, and the model has to see the state to act. | +| [WebMCP](/browsers/webmcp) | The site exposes structured tools for the action you need. | Only works on sites that register tools. | +| [Browser REPL](/browsers/repl) | An agent writes its own helpers and reuses them across turns. State persists across calls until the REPL resets. | JavaScript only. | +| CDP | You have an existing Playwright, Puppeteer, or CDP codebase to point at Kernel. | Adds a protocol fingerprint and a network hop. | +| WebDriver BiDi | You need the W3C standard protocol. | Smaller client ecosystem. | + +For agents, start with [playwright execution with a computer use fallback](/browsers/playwright-computer-use-fallback): script the deterministic steps, and hand the page to a computer use model when a step doesn't respond to a selector. + +### Why the choice matters on hardened sites + +CDP is what Playwright and Puppeteer speak, and anti-bot vendors scan for its signatures. Computer controls carry no CDP connection, so there's no protocol fingerprint to leak. That makes them the stronger option on sites with aggressive detection. How much this matters is site-specific, so test before you commit — see [bot anti-detection](/browsers/bot-detection/overview). + +### Control surface examples @@ -210,85 +235,61 @@ fmt.Println(response.Result) -## Why computer use for agents +## 2. Where the loop runs + +Your loop is whatever decides the next action: a script, an agent, or a model. It can run in three places. + + + + Connect to `cdp_ws_url` or `webdriver_ws_url` from wherever your code already runs. Any CDP client works, and there's no lock-in. + + **Costs:** a network round trip per action, disconnects to handle, screenshot and DOM bandwidth, and the CDP fingerprint above. It's fine for low-frequency or deterministic work, and it hurts most in a vision loop. + + + Send code, not commands. Each call runs in the browser's VM against the live session, so an agent can drive the page turn by turn — one tool call per step, structured data back. + + **Costs:** each call is stateless, so the code you send has to be self-contained. There's nothing to install and no connection to manage. + + + Run JavaScript in a persistent runtime inside the browser's VM. Variables, helpers, and state carry across calls, so an agent can define what it needs once and build on it turn by turn. See [Browser REPL](/browsers/repl). + + **Costs:** JavaScript only, and state is lost when the REPL resets. + + + Deploy the whole agent next to the browser with the [code execution platform](/apps/develop), invoked on demand or on a schedule, with no infrastructure of your own. -Kernel's computer controls are built to match how computer-use models were trained — the same primitives the model emits (screenshot, click at coords, type, key, scroll, drag) map 1:1 onto the API. There's no harness translating model output into framework calls. + **Costs:** your agent has to be deployable as a Kernel app. It's worth it once the automation is long-running, stateful, or triggered by events rather than by a person. + + -- **Native fit.** Screenshot, click, type, key, scroll, drag — the primitives the model already speaks. -- **Faster screenshots.** Captures bypass CDP, which removes the largest source of latency in a vision loop. -- **Better against bot detection.** No CDP connection means no CDP fingerprint to leak. Pairs naturally with [stealth mode](/browsers/bot-detection/stealth) and [residential proxies](/proxies/residential). -- **Human-like input.** OS-level events with Bézier-curve mouse paths, variable typing speed, and configurable mistype rate. -- **Not DOM-limited.** Screenshots capture the full VM, so the agent can see and interact with native dialogs, canvas elements, iframes, and PDFs — not just things you can address with a selector. +### Where computer use fits -## Why playwright execution over a direct CDP connection +A computer use agent answers the first question, not the second — it still has to run its loop somewhere. Because every turn ships a screenshot instead of a small script, running that loop off-platform costs far more than it does for a Playwright-driven agent: you pay image bandwidth and a round trip on every step. That makes computer use the strongest case for running your loop next to the browser. Model inference stays with the model vendor either way. -If you're reaching for Playwright, prefer the execution API over `connectOverCDP`. Same Playwright API you already know, none of the setup. +### Putting it together -- **Run from anywhere.** No `playwright` package to version-pin, no Chromium download, no CDP connection to manage. Send the code, get the result. -- **Co-located with the browser.** Code runs in the same VM as the browser — no network hop between your script and the page, fewer flakes. -- **Patchright by default.** Hardened against bot detection out of the box. -- **Full Playwright API.** `page`, `context`, and `browser` are all in scope. Anything Playwright can do — DOM queries, file uploads, full-page screenshots — works here. -- **Returns values.** `return` from your code and the result comes back in the response. Easy to use as an agent tool. +| Your automation | Control surface | Where the loop runs | +| --- | --- | --- | +| Scheduled scrape of a known page | Playwright execution | Anywhere — one call, one result | +| Agent doing multi-step work on a normal site | Playwright execution, computer use fallback | Playwright execution API, or the code execution platform once it's long-running | +| Agent on a site with aggressive detection | Computer controls | Code execution platform | +| Existing Playwright suite you're migrating | CDP | Your own CI, then move hot paths to playwright execution | ## Computer use + playwright execution -Computer controls drive the browser the way a person would — they don't speak the programmatic API surface. Anything you'd reach for the DOM or Playwright client for (reading text and attributes, `page.goto`, file uploads, cookie or storage access, switching tabs) belongs on the [playwright execution](/browsers/playwright-execution) side. The recommended pattern for agents is computer controls for interaction, playwright execution as a tool the agent can call when it needs structured data or a programmatic action. - - -```typescript Typescript/Javascript -const response = await kernel.browsers.playwright.execute( - kernelBrowser.session_id, - { - code: ` - const rows = await page.$$eval('table tr', (trs) => - trs.map((tr) => Array.from(tr.querySelectorAll('td')).map((td) => td.textContent)) - ); - return rows; - `, - }, -); - -console.log(response.result); -``` - -```python Python -response = kernel.browsers.playwright.execute( - id=kernel_browser.session_id, - code=""" - const rows = await page.$$eval('table tr', (trs) => - trs.map((tr) => Array.from(tr.querySelectorAll('td')).map((td) => td.textContent)) - ); - return rows; - """, -) +Computer controls drive the browser the way a person would — they don't speak the programmatic API surface. Anything you'd reach for the DOM or Playwright client for (reading text and attributes, `page.goto`, file uploads, cookie or storage access, switching tabs) belongs on the [playwright execution](/browsers/playwright-execution) side. When computer use is driving, expose playwright execution to the agent as a tool it can call for structured data or a programmatic action. For the full pattern in the other direction — playwright execution first, computer use when a step doesn't respond to a selector — see [playwright with computer use fallback](/browsers/playwright-computer-use-fallback). -print(response.result) -``` +## Lower-level access -```go Go -response, err := client.Browsers.Playwright.Execute( - ctx, - kernelBrowser.SessionID, - kernel.BrowserPlaywrightExecuteParams{ - Code: ` - const rows = await page.$$eval('table tr', (trs) => - trs.map((tr) => Array.from(tr.querySelectorAll('td')).map((td) => td.textContent)) - ); - return rows; - `, - }, -) -if err != nil { - panic(err) -} +For work that isn't driving the page, you can also reach the browser's VM directly: -fmt.Println(response.Result) -``` - +- **[Browser curl](/browsers/curl):** send HTTP requests through the browser's network stack, with its cookies and proxy. +- **[SSH](/browsers/ssh):** open a shell in the browser's VM, or forward a local port into it. ## Going deeper - [Computer Controls reference](/browsers/computer-controls) — every mouse, keyboard, and screen primitive. - [Playwright Execution reference](/browsers/playwright-execution) — the full execution surface, return values, and timeouts. +- [Browser REPL reference](/browsers/repl) — persistence across calls, browser control helpers, and WebMCP helpers. - [Computer use integrations](/integrations/computer-use/anthropic) — drop-in examples for Anthropic, Gemini, OpenAI, and more. -- [Network access](/info/network-access) lists the domains and ports to allow for API, CDP, and WebDriver BiDi connections. +- [Firewall allowlist](/info/network-access) lists the domains and ports to allow for API, CDP, and WebDriver BiDi connections. diff --git a/introduction/create.mdx b/introduction/create.mdx index 4ee4d5af..8629f658 100644 --- a/introduction/create.mdx +++ b/introduction/create.mdx @@ -91,7 +91,7 @@ Most of what you'll tune at creation time falls into four buckets: `browsers.create()` boots a browser for you on the spot. That's the right call while you're building, and for workloads that run occasionally. -Once you're running the same task repeatedly — or more than a handful at a time — create a [browser pool](/browsers/pools) instead. A browser pool holds browsers that are already booted with your configuration applied, so `acquire` hands you one that's ready to drive rather than starting one from scratch. Two things get faster: configurations that restart Chromium on creation (custom viewports, extensions, kiosk mode) are already applied, and acquiring from a browser pool sidesteps the [rate limit](/info/pricing#rate-limiting) on browser creation that you'll otherwise hit at volume. +Once you're running the same task repeatedly — or more than a handful at a time — create a [browser pool](/browsers/pools) instead. A browser pool holds browsers that are already booted with your configuration applied, so `acquire` hands you one that's ready to drive rather than starting one from scratch. Two things get faster: configurations that restart Chromium on creation (custom viewports, extensions, kiosk mode) are already applied, and acquiring from a browser pool sidesteps the [rate limit](/browsers/concurrency-and-limits#rate-limits) on browser creation that you'll otherwise hit at volume. ```typescript Typescript/Javascript @@ -133,7 +133,7 @@ kernel browser-pools acquire checkout-pool ``` -An acquired browser returns the same fields as one you created directly, so the rest of your code is identical. Idle browsers in a browser pool aren't billed — you pay only while a browser is acquired and running — though browser pool capacity does count against your [concurrency limit](/info/pricing#concurrency-limits). +An acquired browser returns the same fields as one you created directly, so the rest of your code is identical. Idle browsers in a browser pool aren't billed — you pay only while a browser is acquired and running — though browser pool capacity does count against your [concurrency limit](/browsers/concurrency-and-limits#concurrency). ## Lifecycle diff --git a/introduction/manage.mdx b/introduction/manage.mdx new file mode 100644 index 00000000..3f567dbb --- /dev/null +++ b/introduction/manage.mdx @@ -0,0 +1,35 @@ +--- +title: "Manage Overview" +sidebarTitle: "Overview" +description: "Organize, secure, and govern how your team uses KERNEL" +mode: "wide" +--- + +Management settings apply across your organization rather than to a single browser: how resources are split between teams and environments, who and what can call the API, what happened and when, and how much you spend. + + + + Separate environments, teams, or customers, each with its own browsers, profiles, credentials, and limits. + + + Create, scope, rotate, and delete the keys your code and agents use. + + + Set a monthly spending guardrail for your organization, a project, or both. + + + A year of API request history across your organization, searchable and exportable on Start-Up and Enterprise. + + + The KERNEL domains and ports to allow if your firewall restricts outbound traffic. + + + +## Multi-tenant setups + +If you run KERNEL on behalf of your own customers, combine these pieces so each customer is isolated and capped: + +- **One [project](/info/projects) per customer.** Each project has its own browsers, profiles, credentials, proxies, and deployments. +- **A [project-scoped API key](/info/api-keys) per customer workload.** A project-scoped key can only reach resources in its project. +- **A [spending cap](/info/spending-caps) and a [concurrency limit](/info/projects#concurrency-limits) per project.** One customer's usage can't exhaust your budget or your organization's browser limit. +- **One [profile](/browsers/profiles) per end user.** Each user's logins and browser state stay separate, inside their customer's project. diff --git a/introduction/observe.mdx b/introduction/observe.mdx index 167e470f..97613025 100644 --- a/introduction/observe.mdx +++ b/introduction/observe.mdx @@ -1,5 +1,6 @@ --- -title: "Observe" +title: "Observe Overview" +sidebarTitle: "Overview" description: "Watch your agent work, debug what went wrong" --- diff --git a/introduction/scale.mdx b/introduction/scale.mdx index 53947ebb..4d09f66e 100644 --- a/introduction/scale.mdx +++ b/introduction/scale.mdx @@ -1,52 +1,75 @@ --- -title: "Scale" -description: "Recommended practices for scaling in production" +title: "Scale Overview" +sidebarTitle: "Overview" +description: "Plan around your limits, keep browsers fast, and decide whether a browser pool fits" --- -## Overview -This guide covers how to run Kernel in production at scale — which architecture to build around browser creation, and when to reach for a browser pool. It assumes you're comfortable [creating](/introduction/create) and [controlling](/introduction/control) browsers; for the mechanics of standing up a pool and acquiring from it, see [Browser Pools](/browsers/pools). +Scaling an agent on KERNEL comes down to three questions, in this order: how many browsers you can run and create, how fast each one starts, and whether your workload needs a browser pool. Most workloads scale on on-demand browsers alone. This guide assumes you're comfortable [creating](/introduction/configure) and [controlling](/introduction/control) browsers. -## Why a browser pool + +## 1. Know your limits -A [browser pool](/browsers/pools) keeps a set of identically-configured browsers ready for immediate use. Compared to creating browsers on demand, it gives you: +Three limits shape a workload at scale, and each is covered in [concurrency and limits](/browsers/concurrency-and-limits): -- **Low-latency acquisition** — the browser is already booted with your configuration applied (including settings like custom viewports, extensions, and kiosk-mode live view that otherwise [restart Chromium](/browsers/performance#troubleshooting-latency) on a fresh browser), so `acquire` hands you one that's ready to drive. -- **Reserved, pre-configured capacity** — a fixed set of browsers on your exact configuration, ready before traffic arrives. -- **Higher creation throughput** — acquiring from a pool isn't subject to the [rate limit](/info/pricing#rate-limiting) on `browsers.create()` that high-volume workloads hit. +- **Concurrency:** how many browsers can exist at once across your organization. Browsers in [standby](/browsers/standby) still count, so delete browsers when a task finishes to free the slot. +- **Create rate:** how fast you can create new browsers. Exceeding it returns `429 Too Many Requests`, which the SDKs retry automatically. +- **Per-browser resources:** how much memory each browser has, which caps how many tabs and how heavy a page one browser can handle. -The tradeoff: a browser pool counts against your concurrency limit whether or not its browsers are currently acquired — a pool sized to 40 holds 40 of your limit. Idle pooled browsers aren't billed, but they hold the slot. +Each plan has set limits, and they go up when you [upgrade your plan](/info/pricing). Enterprise limits are custom. Use [project concurrency limits](/info/projects#concurrency-limits) to split one organization's limit across teams or environments. -## When to use a pool vs on-demand +## 2. Keep browsers fast -Reach for a **browser pool** when: +A browser is created in about 30ms at P50 and 105ms at P99 ([performance](/browsers/performance)). Most slow starts come from configuration rather than load: custom viewports, extensions, and kiosk mode restart Chromium on creation and add seconds. Check [troubleshooting latency](/browsers/performance#troubleshooting-latency) before reaching for anything else, and run your code next to the browser with [Playwright execution](/browsers/playwright-execution) or the [code execution platform](/apps/develop) to cut the time each action takes. -- you're running the same workload repeatedly, in production -- acquisition latency matters — a cold start is unacceptable (for example, a synchronous, user-facing action) -- traffic is steady or high-frequency enough to keep the browser pool utilized -- you're hitting the `browsers.create()` rate limit at volume + +## 3. Decide between on-demand browsers and a browser pool + +We recommend defaulting to on-demand browsers, both when you're getting started and as you scale. Browser pools fit a specific type of workload, described below. Stick with **on-demand `browsers.create()`** when: -- volume is low, bursty, one-off, or you're still developing -- each session needs a different configuration (a pool is one fixed config) +- you're still building +- your configuration changes per user (a pool is one fixed config) - you need a GPU browser (not available in pools) -Concurrency and request patterns are how you *size* a pool once you've decided to use one — not a threshold that gates whether pools are worth it. Even a small pool pays off when acquisition latency matters and demand is steady. +Reach for a **browser pool** when: + +- you've built and scaled your workload, and every run uses the same workload attributes +- you're hitting the `browsers.create()` rate limit at volume +- you need the lowest possible acquisition latency (for example, a heavily customized browser config) +- traffic is steady or high-frequency enough to keep the browser pool utilized + + + If you're on an Enterprise plan, speak with your account manager about applicable rate limits for `browsers.create()` and what's best for your workloads. + + + +### What a browser pool gives you + +A [browser pool](/browsers/pools) keeps a set of identically-configured browsers ready for immediate use. Compared to creating browsers on demand, it gives you: + +- **Lowest-latency acquisition** — the browser is already booted with your configuration applied (including settings like custom viewports, extensions, and kiosk-mode live view that otherwise [restart Chromium](/browsers/performance#troubleshooting-latency) on a fresh browser), so `acquire` hands you one that's ready to drive. +- **Reserved, pre-configured capacity** — a fixed set of browsers on your exact configuration, ready before traffic arrives. +- **Higher creation throughput** — acquiring from a pool isn't subject to the [rate limit](/browsers/concurrency-and-limits#rate-limits) on `browsers.create()` that high-volume workloads hit. -## Sizing +The tradeoff: a browser pool counts against your concurrency limit whether or not its browsers are currently acquired — a pool sized to 40 holds 40 of your limit. Idle pooled browsers aren't billed, but they hold the slot. + + +### Sizing a pool Watch `available_count` and target 10–20% available under normal load, resizing before traffic peaks rather than during them. See [Sizing a browser pool](/browsers/pools#sizing-a-browser-pool) for the full guidance. ## Architecture patterns -### Direct browser creation (POC) + +### On-demand creation -For proof-of-concept work and early production systems with modest concurrency needs, creating browsers on-demand is the simplest approach. +Creating a browser per task is the simplest approach. It's the right fit while you're building and when each task needs its own configuration. **When to use:** -- Low or unpredictable volume -- Infrequent or one-off workloads - Early development and testing +- Configuration that changes per user or per task +- GPU browsers @@ -84,9 +107,10 @@ async function processTask(taskData: any) { ``` -### Single browser pool (scaling) + +### Single browser pool -For production systems with consistent, high-frequency workloads, a browser pool allows you to access higher concurrency plus predictable performance. +For production workloads that run on the same configuration every time, a browser pool hands you ready-to-drive browsers, and acquiring from it isn't subject to the `browsers.create()` rate limit. **When to use:** - Consistent, high-frequency workloads on a fixed configuration @@ -151,12 +175,13 @@ async function processTask(taskData: any) { - Always release browsers in a `finally` block to prevent browser pool exhaustion - Set `acquire_timeout_seconds` based on your SLA requirements -### Queue-based processing (high scale) + +### Queue-based processing -For systems exceeding browser pool capacity or with unpredictable bursts, implement a task queue to manage workloads gracefully. +When request volume exceeds your concurrency or traffic arrives in unpredictable bursts, put a task queue in front of your browsers. The example below acquires from a browser pool; the same pattern works on demand, with `browsers.create()` in place of `acquire` and `deleteByID` in place of `release`. **When to use:** -- Request volume exceeds a single browser pool's capacity +- Request volume exceeds your available concurrency - Highly variable traffic patterns - Need to prioritize certain tasks - Want to decouple request ingestion from processing @@ -164,8 +189,9 @@ For systems exceeding browser pool capacity or with unpredictable bursts, implem ```typescript -import { Queue } from 'bullmq'; // or any queue system +import { Queue, Worker } from 'bullmq'; // or any queue system import Kernel from '@onkernel/sdk'; +import { chromium } from 'playwright'; const kernel = new Kernel(); const POOL_NAME = 'production-pool'; diff --git a/proxies/custom.mdx b/proxies/custom.mdx index eeeec487..d5ceb320 100644 --- a/proxies/custom.mdx +++ b/proxies/custom.mdx @@ -107,7 +107,7 @@ func main() { ``` -## Configuration Parameters +## Configuration parameters - **`host`** (required) - Proxy server hostname or IP address - **`port`** (required) - Proxy server port (1-65535) diff --git a/proxies/datacenter.mdx b/proxies/datacenter.mdx index 46e623a3..27b54085 100644 --- a/proxies/datacenter.mdx +++ b/proxies/datacenter.mdx @@ -4,7 +4,7 @@ title: "Datacenter Proxies" Datacenter proxies use IP addresses assigned from datacenter servers to route your traffic and access locations around the world. With a shorter journey and simplified architecture, datacenter proxies are both the fastest and most cost-effective proxy option. -## IP Rotation Behavior +## IP rotation behavior Datacenter proxies use **rotating exit IPs** — a new exit IP is assigned per request, so different requests within the same browser session can exit through different IPs. @@ -86,7 +86,7 @@ func main() { ``` -## Configuration Parameters +## Configuration parameters - **`country`** (optional) - ISO 3166 country code (e.g., `US`, `GB`, `FR`) or `EU` for European Union exit nodes - **`bypass_hosts`** (optional) - Array of hostnames that bypass the proxy and connect directly (max 100 entries) diff --git a/proxies/isp.mdx b/proxies/isp.mdx index f9394c96..0d7be70b 100644 --- a/proxies/isp.mdx +++ b/proxies/isp.mdx @@ -4,7 +4,7 @@ title: "ISP Proxies" ISP (Internet Service Provider) proxies are hosted on datacenter infrastructure but use IP addresses assigned by real residential ISPs. Because the ASN belongs to a residential ISP, target sites see them as residential IPs — while the underlying datacenter hosting gives you the speed and stability you'd expect from a datacenter proxy. -## IP Rotation Behavior +## IP rotation behavior ISP proxies provide a **static exit IP that persists across sessions** — every tab, request, reconnection, and future browser session attached to this proxy exits through the same IP. The IP only changes in rare ISP-initiated replacement events. diff --git a/proxies/overview.mdx b/proxies/overview.mdx index 427db6a4..90d322b6 100644 --- a/proxies/overview.mdx +++ b/proxies/overview.mdx @@ -1,10 +1,11 @@ --- -title: "Overview" +title: "Proxies Overview" +sidebarTitle: "Overview" --- Kernel proxies enable you to route browser traffic through different types of proxy servers, providing enhanced privacy, flexibility, and bot detection avoidance. Proxies can be created once and reused across multiple browser sessions. -## Proxy Types +## Proxy types Kernel supports five types of proxies: diff --git a/proxies/residential.mdx b/proxies/residential.mdx index f9132c3c..f695333e 100644 --- a/proxies/residential.mdx +++ b/proxies/residential.mdx @@ -96,7 +96,7 @@ func main() { ``` -## Configuration Parameters +## Configuration parameters - **`country`** - ISO 3166 country code. Must be provided when providing other targeting options. - **`state`** - Two-letter state code. Only supported for US. @@ -105,11 +105,11 @@ func main() { - **`asn`** - Autonomous System Number. Conflicts with city and state. - **`bypass_hosts`** (optional) - Array of hostnames that bypass the proxy and connect directly (max 100 entries) -## Advanced Targeting Examples +## Advanced targeting examples Kernel recommends using the least-specific targeting configuration that works for your use case. The more specific a configuration, the less available IPs there are, increasing the chance of a slow connection or no available connection (`no_peer` connection error). -### Target by City +### Target by city Route traffic through a specific city: @@ -163,7 +163,7 @@ _ = proxy If the city name is not matched, the API will return the best 10 city names from the state to help you find the correct city identifier. -### Target by State +### Target by state Route traffic through a specific state: diff --git a/reference/cli/managed-auth.mdx b/reference/cli/managed-auth.mdx index 01407e9a..dc230865 100644 --- a/reference/cli/managed-auth.mdx +++ b/reference/cli/managed-auth.mdx @@ -132,7 +132,7 @@ Delete a managed auth connection. | `--yes`, `-y` | Skip the confirmation prompt. | ## Credentials -Store login field values, TOTP secrets, and SSO settings that managed auth connections use to authenticate. See [Credentials](/auth/credentials) for concepts. +Store login field values, TOTP secrets, and SSO settings that managed auth connections use to authenticate. See [Credentials](/auth/configuration#credentials-and-auto-reauth) for concepts. ### `kernel credentials create` Create a new credential. diff --git a/vaults/overview.mdx b/vaults/overview.mdx index 68dab206..a44bc138 100644 --- a/vaults/overview.mdx +++ b/vaults/overview.mdx @@ -1,5 +1,6 @@ --- -title: "Overview" +title: "Vaults Overview" +sidebarTitle: "Overview" description: "Group credentials and payment items, collect values, and control their use by attached browsers" --- @@ -27,6 +28,17 @@ navigation and submission, start with [Fill from Vault](/auth/fill-from-vault). see [wallet integrations](/integrations/wallets/overview). +## Choose where credentials come from + +a `credential` item can get its values from three places. all three end with the browser filling the login without your agent reading the values. + +| | [KERNEL-hosted collection](/vaults/credentials) | [an existing vault](/vaults/existing-credential-vault) | [1Password](/vaults/1password) | +| --- | --- | --- | --- | +| where the secret lives | encrypted in a KERNEL `credential` item | an encrypted copy in a KERNEL `credential` item | in the user's 1Password account | +| how it gets there | the user enters it in a KERNEL-hosted form, or your backend writes it | your backend copies it from the vault you already use and keeps it in sync | the user links their 1Password account once | +| who approves each use | your application or agent, when it calls `fill` | your application or agent, when it calls `fill` | the user, in the 1Password app | +| pick it when | you're collecting credentials from users for the first time | your credentials already live in another secrets manager | your users keep logins in a private 1Password vault | + ## How vaults work ### Sensitive values do not come back through the api @@ -72,7 +84,7 @@ availability. -### Inject values with KERNEL's fill api +### Inject values with KERNEL's fill API retrieve the item and require `fill` in `available_operations`. your controller authorizes the destination and supplies field names and selectors, not the stored values. KERNEL checks the browser attachment and item lifecycle, validates the target inputs, and writes values into the attached browser.