Compare commits

..

4 Commits

Author SHA1 Message Date
didericis-codex 031bc6301e Merge remote-tracking branch 'origin/main' into fix/quality-cleanup-consolidated
prd-number-check / require-numbered-prds (pull_request) Successful in 12s
tracker-policy-pr / check-pr (pull_request) Successful in 20s
test / image-input-builds (pull_request) Successful in 43s
test / integration-docker (pull_request) Successful in 1m5s
test / unit (pull_request) Successful in 52s
test / coverage (pull_request) Successful in 19s
2026-07-27 05:34:46 +00:00
didericis-codex aaf1e60abe fix: close consolidated quality review findings
tracker-policy-pr / check-pr (pull_request) Successful in 22s
lint / lint (push) Successful in 1m4s
test / unit (pull_request) Successful in 3m4s
test / image-input-builds (pull_request) Successful in 1m0s
test / integration-docker (pull_request) Successful in 1m10s
test / coverage (pull_request) Successful in 20s
prd-number-check / require-numbered-prds (pull_request) Failing after 10m52s
2026-07-27 05:33:44 +00:00
didericis-codex 82e1a353bd docs: analyze Sandbox Agent SDK architecture 2026-07-27 03:53:02 +00:00
didericis-codex a6e1aebda1 docs: update agent sandbox competitor landscape 2026-07-27 03:44:41 +00:00
8 changed files with 890 additions and 32 deletions
+14
View File
@@ -7,6 +7,7 @@ import asyncio
import math
import sys
from fastapi import FastAPI, HTTPException
from fastapi.exceptions import RequestValidationError
from fastapi.responses import JSONResponse
from pydantic import BaseModel, ConfigDict, StrictStr
from starlette.types import ASGIApp, Message, Receive, Scope, Send
@@ -208,6 +209,19 @@ def create_app(orch: OrchestratorCore, *, signing_key: str) -> FastAPI:
)
app.add_middleware(ControlPlaneBoundary, signing_key=key)
@app.exception_handler(RequestValidationError)
async def invalid_request(
_request: object, exc: RequestValidationError,
) -> JSONResponse:
errors = exc.errors()
location = errors[0].get("loc", ()) if errors else ()
field = str(location[1]) if len(location) > 1 else ""
suffix = f": {field}" if field else ""
return JSONResponse(
{"error": f"invalid request body{suffix}"},
status_code=400,
)
@app.get("/health")
def health() -> dict[str, str]:
return {"status": "ok"}
+29 -12
View File
@@ -29,8 +29,13 @@ class DbStore:
"""Create and verify the private parent directory."""
parent = self.db_path.parent
parent.mkdir(mode=0o700, parents=True, exist_ok=True)
if parent.is_symlink():
raise PermissionError(f"database directory must not be a symlink: {parent}")
parent.chmod(0o700)
if stat.S_IMODE(parent.stat().st_mode) != 0o700:
parent_stat = parent.lstat()
if not stat.S_ISDIR(parent_stat.st_mode):
raise PermissionError(f"database parent is not a directory: {parent}")
if stat.S_IMODE(parent_stat.st_mode) != 0o700:
raise PermissionError(f"database directory is not mode 0700: {parent}")
def _secure_db_file(self) -> None:
@@ -40,22 +45,34 @@ class DbStore:
This store contains control-plane identity tokens, so both creation and
repair are fail-closed rather than best-effort.
"""
flags = os.O_RDWR | os.O_CREAT | os.O_NOFOLLOW
fd = os.open(
self.db_path,
flags,
stat.S_IRUSR | stat.S_IWUSR,
)
try:
fd = os.open(
self.db_path,
os.O_WRONLY | os.O_CREAT | os.O_EXCL,
stat.S_IRUSR | stat.S_IWUSR,
)
except FileExistsError:
pass
else:
self._secure_open_file(fd)
finally:
os.close(fd)
self._chmod()
def _chmod(self) -> None:
"""Enforce and verify the private database mode after every write."""
self.db_path.chmod(0o600)
if stat.S_IMODE(self.db_path.stat().st_mode) != 0o600:
fd = os.open(self.db_path, os.O_RDWR | os.O_NOFOLLOW)
try:
self._secure_open_file(fd)
finally:
os.close(fd)
def _secure_open_file(self, fd: int) -> None:
"""Pin, validate, and secure an opened database filesystem object."""
file_stat = os.fstat(fd)
if not stat.S_ISREG(file_stat.st_mode):
raise PermissionError(
f"database must be a regular file: {self.db_path}"
)
os.fchmod(fd, stat.S_IRUSR | stat.S_IWUSR)
if stat.S_IMODE(os.fstat(fd).st_mode) != 0o600:
raise PermissionError(f"database is not mode 0600: {self.db_path}")
def _connect(self) -> sqlite3.Connection:
+293 -8
View File
@@ -32,9 +32,17 @@ not a principled scope exclusion: both are major hosted sandbox platforms and
belong in this landscape even though they target platform builders rather than
bot-bottle's local single-operator workflow.
Updated 2026-07-27 after a scan of recent Show HN launches: **Black LLAB,
Eve, CloudRouter, Nucleus, yolo-cage, and Sandbox Agent SDK** added as a
dated entrant cohort. They sharpen the comparison on three axes the original
table underweighted: the browser/preview loop, parallel-agent operator UX, and
a provider-neutral automation/session API.
## Summary
The main table compares bot-bottle against fifteen isolation/sandbox tools.
The main table compares bot-bottle against fifteen canonical
isolation/sandbox tools; a later section evaluates six recent HN entrants
without widening an already unwieldy table.
Governance/pre-action authorization and credential-only layers are covered
separately because they don't provide VM or container isolation. None
duplicate bot-bottle's combination of local
@@ -542,6 +550,199 @@ them.
framework runtime is not compromised.
- **Maturity**: Specification + reference implementation, 2026.
## Recent HN entrants (added 2026-07-27)
These are grouped by launch date rather than promoted into the main table.
Several are young or sparsely documented, and putting them beside mature
runtime platforms with false precision would obscure the useful comparison.
The HN launch posts are the evidence snapshot; feature claims should be
rechecked against their repositories before relying on them for a security
decision.
### Black LLAB
- **Source**: https://github.com/isaacdear/black-llab ;
HN launch https://news.ycombinator.com/item?id=47402394
- **Isolation/locality**: Local Docker environment, with an isolated container
created for each agent task. Shared host kernel; no stronger boundary is
claimed.
- **Agent integration**: General local/cloud model workspace. Its headline is
dynamic routing of simple prompts to local models and complex prompts to
hosted models, with code execution and web scraping inside the task
container.
- **Network/credentials**: No default-deny egress, payload inspection, or
host-side credential injection documented in the launch.
- **Competitive read**: Superficial overlap ("a container per agent task"),
but not a direct security-policy competitor. Its useful challenge is the
integrated model-selection UX, which bot-bottle intentionally leaves to the
selected agent provider.
- **Maturity**: Early solo project; HN launch received 1 point.
### Eve
- **Source**: https://eve.new/ ;
HN launch https://news.ycombinator.com/item?id=47721255
- **Isolation/locality**: Managed, hosted Linux sandbox per user/session
(claimed 2 vCPU, 4 GB RAM, 10 GB disk), with filesystem, code execution,
headless Chromium, and service connectors.
- **Agent integration**: End-user OpenClaw-style agent product. An orchestrator
routes subtasks to specialist models and can run parallel subagents that
coordinate through a shared filesystem. Web UI and iMessage are primary
interaction surfaces.
- **Network/credentials**: Broad connectors are a product feature; the launch
does not document bot-bottle-style default-deny route policy, content DLP,
or credentials held outside the sandbox.
- **Competitive read**: Adjacent, not direct. Eve sells a managed colleague;
bot-bottle lets an operator run existing coding-agent CLIs under local
containment. Eve nevertheless demonstrates the appeal of background work,
live progress, browser capability, and mobile notification.
- **Maturity**: Commercial hosted product; HN launch received 71 points and
39 comments.
### CloudRouter
- **Source**: https://github.com/manaflow-ai/manaflow/tree/main/packages/cloudrouter ;
HN launch https://news.ycombinator.com/item?id=47006393
- **Isolation/locality**: Claude Code or Codex runs locally and provisions
remote cloud VMs/GPUs for execution. Project files are uploaded to the VM;
each machine exposes auth-protected VNC, VS Code, and Jupyter surfaces.
- **Agent integration**: A skill plus CLI lets the coding agent itself start,
command, inspect, and tear down machines. Browser automation is integrated,
including snapshots and screenshots. Parallel disposable compute is the
central workflow.
- **Network/credentials**: The launch emphasizes remote resource isolation and
authenticated UI endpoints, not default-deny guest egress, payload DLP, or
proxy-held application credentials.
- **Competitive read**: The closest recent workflow competitor. It directly
addresses parallel coding agents, environmental conflict, and closing the
browser/test loop, but trades local custody for elastic cloud compute.
Cloud VMs and GPUs could be a future bot-bottle backend; they do not replace
its manifest/policy layer.
- **Maturity**: Active open-source monorepo project; HN launch received
138 points and 36 comments.
### Nucleus
- **Source**: https://github.com/coproduct-opensource/nucleus ;
HN launch https://news.ycombinator.com/item?id=46855770
- **Isolation/locality**: Firecracker microVM with an enforcing MCP tool proxy.
- **Agent integration/config**: Compositional permission envelope for
read/write/run actions. The envelope is non-escalating and can tighten or
terminate, with scoped approval tokens for gated operations.
- **Network/credentials**: Default-deny egress, DNS allowlist, iptables drift
detection, time/budget caps, and hash-chained audit logging are claimed.
Remote append-only audit storage and attestation were roadmap items at
launch.
- **Competitive read**: Direct on security architecture, especially
non-escalating policy and tamper-evident audit. It is an early execution/tool
proxy rather than a provider-neutral, one-command coding-agent product. Its
tool-level action envelope is semantically finer than bot-bottle's network
boundary; bot-bottle is stronger on turnkey agent/provider integration,
credential custody, Git mediation, and long-running operator workflow.
- **Maturity**: Early OSS experiment; HN launch received 3 points.
### yolo-cage
- **Source**: https://github.com/borenstein/yolo-cage ;
HN launch https://news.ycombinator.com/item?id=46706796
- **Isolation/locality**: Local sandbox for running multiple coding agents in
YOLO mode. The launch discussion describes a VM boundary.
- **Agent integration**: Built around the native Claude Code experience and
motivated by running many agents in parallel without permission-prompt
fatigue.
- **Network/Git/credentials**: Strict egress filtering, configurable HTTP
middleware, and mediated `git`/`gh` dispatch are the main value. The launch
discussion explicitly identifies provider credential handling as unfinished
and difficult because Claude state spans multiple host paths.
- **Competitive read**: The closest new threat-model competitor. It shares
bot-bottle's premise that filesystem isolation alone is insufficient and
that Git plus authorized HTTP channels need mediation. bot-bottle currently
leads on cross-provider support, proxy-held Claude/Codex/forge credentials,
typed per-role manifests, content DLP, and supervision. yolo-cage's simpler
pitch and narrower Claude-first setup may be easier to explain.
- **Maturity**: Early local tool; HN launch received 60 points and 76 comments.
### Sandbox Agent SDK
- **Source**: https://github.com/rivet-dev/sandbox-agent ;
HN launch https://news.ycombinator.com/item?id=46795584
- **Isolation/locality**: Does not provide the isolation primitive. It runs
inside E2B, Daytona, Modal, Cloudflare Containers, Agent Computer, BoxLite,
Docker, or another sandbox provider. Embedded mode can also run locally
without a sandbox.
- **Agent integration**: Provider-neutral Rust server/SDK exposing a common
HTTP/SSE/OpenAPI interface across Claude Code, Codex, OpenCode, Cursor, Amp,
and Pi, plus a universal event/session schema for external storage and
replay. It also exposes filesystem, managed-process, terminal, MCP, skills,
custom-tool, and computer-use APIs. TypeScript is the primary SDK surface.
- **Network/credentials**: Delegated to the chosen sandbox provider.
- **Credential posture**: Its documented convenience command extracts real
OpenAI/Anthropic credentials from local agent configuration and passes them
as environment variables into the sandbox. That is materially weaker than
bot-bottle's host-side credential custody, but it is an integration choice,
not a structural limitation: a sandbox provider could put a credential
proxy underneath the same SDK.
- **Competitive read**: A serious architectural threat despite not supplying
isolation. Sandbox Agent is trying to standardize the boundary *above* the
sandbox: one client protocol, session model, and UI/control surface across
every coding agent and runtime. If that boundary becomes the ecosystem
standard, users and application builders may choose a sandbox provider plus
Sandbox Agent rather than a vertically integrated launcher. bot-bottle's
manifests would then be valuable chiefly as a local policy/backend
implementation unless they expose an equally usable control contract.
- **Maturity**: Apache 2.0, ~1.5k stars and 426 commits at the 2026-07-27
check; HN launch received 41 points.
#### Why the Sandbox Agent architecture is strategically different
The manifest and the universal control protocol solve different layers:
- A bot-bottle manifest is a **trusted launch-time policy composition**. It
selects the agent role, isolation backend, image, skills, egress routes,
credentials, Git mediation, and supervision policy. Crucially, identity and
secret references live on the host side of the trust boundary.
- Sandbox Agent is a **runtime control and observation protocol**. A remote
client creates sessions, sends messages, handles permissions, configures
skills/MCP, manipulates files/processes/desktops, and streams normalized
events. It deliberately delegates sandbox lifecycle, Git management,
storage, network policy, and credential security to other products.
That makes it complementary in a component diagram but competitive in product
architecture. The layer that becomes the stable integration point tends to own
the ecosystem. Three plausible threat paths matter:
1. **Standard control plane, interchangeable runtimes.** Applications integrate
once with Sandbox Agent and treat E2B, Daytona, BoxLite, Docker, or a future
local microVM as replaceable compute. A provider that bundles adequate
egress and credential custody makes bot-bottle's end-to-end launcher less
necessary.
2. **Policy grows upward.** Sandbox Agent already configures permissions,
skills, MCP, custom tools, filesystem/process access, and computer use. If
it adds a declarative, host-verifiable policy document, the overlap with
agent/bottle manifests becomes substantial even if enforcement remains
delegated.
3. **UI and session ownership.** Its universal transcript schema, Inspector,
React components, event replay, and remote terminal/computer APIs can become
the natural basis for desktop, web, and mobile agent managers. bot-bottle's
security layer could remain stronger while losing the operator surface and
distribution channel.
The counter-position is not to claim that manifests and an API are mutually
exclusive. The defensible split is:
- bot-bottle owns the trusted policy and enforcement plane outside the agent;
- a provider-neutral protocol owns agent process control and normalized
events; and
- the operator UI consumes both.
This suggests an explicit compatibility decision rather than parallel,
accidental protocol design: evaluate running Sandbox Agent inside a bottle and
exposing it only through the authenticated bot-bottle control plane. If its
schema is suitable, adopting it could turn a threat into an integration while
keeping manifests as the higher-trust policy source. If it is unsuitable,
bot-bottle should still publish a stable provider-neutral session/event API so
frontends do not depend on Claude/Codex/Pi-specific process behavior.
## Comparison table
*Isolation/sandbox tools only. AGT and OAP are governance layers — see their per-project notes above.*
@@ -616,6 +817,70 @@ would be a *backend* bot-bottle could call, not a competitor to its
manifest layer. endo-familiar is in a different paradigm entirely:
capability passing rather than kernel boundaries.
**Recent entrants change two parts of this read.** yolo-cage is closer to the
actual threat model than agent-safehouse or litterbox: it combines a VM-style
boundary with mediated Git and filtered HTTP specifically for parallel coding
agents. Sandbox Agent SDK is the more important strategic entrant even though
it supplies no isolation. It can become the standard agent-control layer above
all of these runtimes, including a future bot-bottle backend. CloudRouter is
the clearest workflow challenge because its browser/desktop/GPU loop makes
parallel agents visibly more capable, not merely safer.
## Gap evaluation after the 2026-07-27 entrant scan
### Material gaps
1. **A stable provider-neutral control and event protocol.** This is the
largest newly visible gap. bot-bottle normalizes launch/provisioning across
providers, but an external UI or orchestrator still lacks one documented
contract for creating a Claude/Codex/Pi session, sending input, handling
permission/supervision events, streaming normalized output, reconnecting,
and replaying history. Sandbox Agent SDK addresses exactly this layer and
is already portable across many sandbox providers.
2. **Browser/preview closure.** CloudRouter and Eve make a browser or desktop
part of the standard agent environment and expose screenshots/live viewing
to the operator. bot-bottle can run dev servers and supports nested
containers, but it does not present a first-class browser/computer-use
primitive or an auth-protected preview surface. For coding agents expected
to verify UI work, this is a real product gap.
3. **Unified parallel-session operator UX.** Named persistent bottles and
supervision provide the substrate, but the recent products make task
switching, live progress, notifications, terminal attach, diffs, and
session history the product. Security depth will not compensate for a
visibly rougher daily loop.
4. **Normalized transcript persistence and replay.** bot-bottle preserves
provider-specific state for resume; it does not expose a provider-neutral
event record suitable for audit, replay, analytics, or a web/mobile client.
This is both a UX gap and an audit gap.
### Important, but not necessarily bot-bottle features
- **Cloud VM/GPU provisioning.** Valuable for elastic workloads and could be a
backend, but it conflicts with the local-custody default and should not
displace core policy work.
- **Automatic model routing.** Black LLAB and Eve sell task-to-model routing.
bot-bottle's provider-template boundary can host that choice without making
it part of the trusted sandbox policy.
- **A thousand SaaS connectors.** This broadens capability and blast radius.
The bot-bottle-native answer should remain explicit, scoped forge/egress
associations rather than connector count as a goal.
- **SDK-driven sandbox lifecycle as the primary configuration model.** Useful
for platform builders, but not a replacement for reviewable, host-owned
manifests. A control API and a declarative policy source are compatible;
neither should silently become the other.
### Areas where bot-bottle remains ahead
- real provider and forge credentials remain outside the agent process rather
than being extracted into its environment;
- authorized HTTP payloads are scanned, not merely destination-filtered;
- Git writes traverse a distinct gate with secret scanning and host-held
upstream credentials;
- role policy is host-owned, composable, and separate from untrusted repo
content; and
- local Firecracker/Apple Container execution preserves operator custody
without requiring a hosted sandbox platform.
## Borrowable ideas
### Already shipped or otherwise addressed
@@ -642,6 +907,19 @@ capability passing rather than kernel boundaries.
### Still worth considering
- **Sandbox Agent compatibility or an equivalent stable protocol (highest
priority):** spike running its server inside a bottle behind bot-bottle's
authenticated control plane. Compare its session/event schema, permission
model, restore semantics, and provider coverage with current provider
adapters. Adopt compatibility if it preserves the host-owned trust boundary;
otherwise specify bot-bottle's own stable API before building another UI.
- **First-class browser/preview loop** (from CloudRouter and Eve): give a
bottle an optional browser/computer-use capability plus an operator-visible,
authenticated preview/screenshot surface. Treat its network access as part
of the bottle policy, not an implicit bypass.
- **Provider-neutral transcript/event persistence** (from Sandbox Agent SDK):
retain enough normalized structure for replay and audit while preserving the
provider-native state needed for exact resume.
- **Live network activity in the supervisor TUI** (from Docker sbx): show
allowed and blocked connections and let the operator propose policy changes
from the existing supervision surface.
@@ -652,10 +930,11 @@ capability passing rather than kernel boundaries.
closer review. This needs a carefully specified trust model before it can be
more than a heuristic.
Not worth borrowing: the SDK-first programmatic API style of boxlite /
microsandbox (cuts against the declarative-manifest stance), and the
hosted-SaaS dashboard model of tilde.run (cuts against the
"infrastructure I control" goal).
Not worth borrowing: SDK-first *policy configuration* as used by boxlite /
microsandbox (cuts against the reviewable declarative-manifest stance), and
the hosted-SaaS custody model of tilde.run (cuts against the "infrastructure I
control" goal). A provider-neutral runtime-control API is a separate concern
and is worth borrowing.
## Publishing and positioning verdict
@@ -679,9 +958,15 @@ bot-bottle remains unusual in combining:
The practical wedge is “as easy as native yolo, with declarative role policy
and self-hosted custody,” including scoped access to private LAN/Tailnet
services that cloud-first runtimes cannot provide without additional network
plumbing. The main competitive risks are a local wrapper such as claudebox or
Docker sbx growing a role-manifest layer, and GUI products such as SuperHQ
adding equivalent policy and audit depth.
plumbing. The main competitive risks are now:
- a local wrapper such as yolo-cage, claudebox, or Docker sbx growing a
role-manifest and credential-custody layer;
- Sandbox Agent SDK becoming the standard control/session boundary and making
the runtime beneath it interchangeable; and
- GUI products such as SuperHQ or CloudRouter adding equivalent policy and
audit depth before bot-bottle closes the browser/preview and
parallel-session UX gaps.
## Caveats
@@ -0,0 +1,536 @@
# Sandbox Agent SDK and bot-bottle: protocol versus product
This note asks whether [Sandbox Agent SDK](https://github.com/rivet-dev/sandbox-agent)
and bot-bottle compete for the same architectural layer, whether bot-bottle
can productize the turnkey ecosystem/DX layer above it, and how far the
Docker/OCI analogy actually holds.
Research conducted 2026-07-27. Sandbox Agent SDK was at the `0.4.x` line,
Apache 2.0, and documented support for Claude Code, Codex, OpenCode, Cursor,
Amp, and Pi at the time of review.
## Summary
**The projects are complementary at the component boundary and competitive at
the product boundary.** Sandbox Agent SDK normalizes how software controls a
coding-agent process inside an arbitrary sandbox. bot-bottle decides what
sandbox to create, what trusted role and policy it receives, how credentials
and Git access cross the boundary, how traffic is constrained, and how an
operator launches and supervises the result.
The Docker analogy is useful with one correction:
- Sandbox Agent SDK is not equivalent to Linux container APIs or OCI itself.
It is closer to a **containerd shim plus a portable exec/session API for
coding agents**. It adapts incompatible agent processes to one HTTP/SSE
contract.
- A future independent agent-session specification would be the closer OCI
analogue.
- bot-bottle can credibly occupy the **Docker Engine / Compose / Desktop**
layer: packaging, policy composition, lifecycle, networking, credentials,
storage, operator UX, and a one-command experience above interchangeable
agent adapters and isolation runtimes.
That is a viable position, but “turnkey wrapper” undersells it. A thin wrapper
is replaceable. The valuable product is a **turnkey, policy-first coding-agent
runtime** whose manifest compiles trusted operator intent into multiple
enforcement planes. Sandbox Agent SDK may be one internal process-control
component of that product.
The recommended direction is:
1. Keep the bot-bottle manifest as the host-owned source of trusted policy.
2. Spike Sandbox Agent SDK as the in-bottle provider/session adapter.
3. Expose a stable, provider-neutral bot-bottle control API, compatible with
Sandbox Agent where practical.
4. Keep security decisions and authoritative audit outside the sandbox.
5. Build the ecosystem around policy packs, agent images, skills, backends,
operator UI, and trusted integrations—not around a proprietary transcript
protocol.
## What each project is today
### Sandbox Agent SDK
Sandbox Agent is a Rust server that runs alongside the coding agent. A client
connects over HTTP, streams events over SSE, and uses one API across agent
implementations. Its documented surface includes:
- creating and restoring agent sessions;
- sending messages and streaming normalized events;
- handling permissions;
- configuring MCP servers, skills, and custom tools;
- filesystem and managed-process APIs;
- interactive terminal access;
- computer-use/desktop operations;
- a universal session/transcript schema;
- an Inspector UI, React components, CLI, TypeScript SDK, and OpenAPI spec.
It can run in embedded mode or inside E2B, Daytona, Modal, Cloudflare
Containers, Agent Computer, BoxLite, Docker, and other environments. It
explicitly leaves these concerns to the caller or sandbox provider:
- sandbox creation and lifecycle;
- Git repository management;
- durable session storage;
- network policy;
- isolation strength; and
- secure credential delivery.
Its documented credential convenience path extracts real provider credentials
from local agent configuration and passes them into the sandbox environment.
That is convenient but is not an acceptable security boundary for bot-bottle.
Sources:
- [Sandbox Agent repository and architecture](https://github.com/rivet-dev/sandbox-agent)
- [Sandbox Agent documentation](https://sandboxagent.dev/docs)
- [HTTP API](https://sandboxagent.dev/docs/api-reference)
- [Universal session/transcript schema](https://sandboxagent.dev/docs/session-transcript-schema)
### bot-bottle
bot-bottle is a host-side launch, policy, and enforcement system for existing
coding-agent CLIs. Its current architecture includes:
- agent and bottle manifests with composition via `extends:`;
- a host-only trust boundary for roles, identity, and secret references;
- provider templates and plugins for Claude Code, Codex, Pi, and custom
providers;
- Firecracker on KVM Linux and Apple Container on macOS, with Docker fallback;
- image construction and provider-specific provisioning;
- default-deny inspected egress with path/method/header policy;
- payload DLP on authorized channels;
- real credentials held outside the agent and injected by the gateway;
- Git mediation, upstream credential custody, and gitleaks scanning;
- a per-host authenticated orchestrator and shared gateway;
- named bottle lifecycle, resume, supervision, and audit state; and
- a CLI/TUI intended to make full-permission agents operationally tolerable.
The provider layer currently normalizes launch-time concerns—command, image,
prompt delivery, files, skills, environment, verification, and provider-owned
egress routes. It does **not** yet expose a stable provider-neutral runtime
contract for sessions, messages, transcripts, terminals, or normalized events.
That is the gap Sandbox Agent directly illuminates.
Sources in this repository:
- [`README.md`](../../README.md)
- [`0070-per-host-orchestrator.md`](../prds/0070-per-host-orchestrator.md)
- [`0026-agent-provider-templates.md`](../prds/0026-agent-provider-templates.md)
- [`0053-user-provider-plugins.md`](../prds/0053-user-provider-plugins.md)
- [`agent_provider.py`](../../bot_bottle/agent_provider.py)
## The layer model
The cleanest architecture has four layers:
| Layer | Responsibility | Likely owner |
|---|---|---|
| Operator product | Install, select a role, launch, observe, intervene, resume, review changes | bot-bottle |
| Trusted policy and lifecycle | Compose manifest, choose backend/image, hold credentials, enforce egress/Git, persist authoritative audit | bot-bottle |
| Agent control protocol | Start provider process, create session, send input, stream normalized events, terminal/computer operations | Sandbox Agent or a compatible protocol |
| Isolation primitive | VM/container/process boundary, filesystem, CPU/memory, networking substrate | Firecracker, Apple Container, Docker, E2B, Daytona, BoxLite, etc. |
The important boundary is between trusted policy/lifecycle and agent control.
The agent-control daemon runs in the environment being treated as untrusted.
It can report what the agent says happened, but it cannot authoritatively prove
that policy was enforced. Egress decisions, credential custody, Git scanning,
bottle identity, and security audit must remain outside it.
### Proposed composition
```text
operator UI / CLI / API
|
v
bot-bottle orchestrator (trusted)
- resolves manifest
- owns bottle identity and lifecycle
- stores authoritative audit
- authenticates clients
|
+--------------------------+
| |
v v
isolation backend shared gateway (trusted)
Firecracker / Apple / Docker - egress policy + DLP
| - credential injection
| - Git mediation
v
bottle / guest (untrusted)
- Sandbox Agent server
- Claude Code / Codex / Pi subprocess
- workspace, skills, MCP configuration
```
The bot-bottle manifest would compile into both sides:
- **outside the bottle:** backend, network, egress, credentials, Git,
supervision, identity, and authoritative lifecycle;
- **inside the bottle:** selected provider, prompt, skills, MCP configuration,
startup arguments, and non-secret session metadata.
Sandbox Agent should never receive real secrets merely because its API offers
a credential extraction helper. Provider and forge requests should continue
to use bot-bottle's placeholder/proxy pattern.
## How accurate is the Docker/OCI analogy?
### The useful part
The container ecosystem separates low-level execution from a product that
ordinary developers operate. OCI defines interoperable image, runtime, and
distribution specifications. Docker Engine adds a daemon, API, CLI, object
model, images, networks, volumes, and lifecycle; Docker Desktop and related
products add installation, updates, UI, integrations, policy, and team
workflows.
The same separation can exist for coding agents:
| Container ecosystem | Agent-sandbox ecosystem |
|---|---|
| OCI/runtime contract | A future open agent session/event contract |
| `runc` / runtime adapter | Claude/Codex/Pi adapter |
| containerd shim and task/exec API | Sandbox Agent server and HTTP/SSE session API |
| containerd / CRI-style lifecycle | Sandbox-provider lifecycle APIs |
| Docker Engine / Compose | bot-bottle orchestrator + manifests + backends + gateway |
| Docker Desktop / Hub ecosystem | bot-bottle desktop/mobile UX, policy packs, agent images, skills, trusted integrations |
Sandbox Agent makes coding-agent processes portable in roughly the way a shim
makes runtimes consumable through a common lifecycle interface. bot-bottle can
make the entire safe-agent system usable without asking the operator to
assemble that plumbing.
Official container references:
- [Open Container Initiative](https://opencontainers.org/)
- [OCI Runtime Specification](https://github.com/opencontainers/runtime-spec)
- [Docker Engine architecture](https://docs.docker.com/engine/)
- [Docker alternative runtimes and containerd shims](https://docs.docker.com/engine/daemon/alternative-runtimes/)
### Where the analogy breaks
1. **Sandbox Agent is an implementation, not an independent standard.**
Its OpenAPI document is public, but the project currently owns the server,
adapters, schema, and evolution. OCI is an independently governed set of
specifications with multiple implementations.
2. **It sits above, not below, the isolation boundary.** Linux namespaces,
cgroups, VMs, and OCI runtimes create the boundary. Sandbox Agent controls a
process after some other system has created that boundary.
3. **It reaches into product territory.** Inspector, React components,
computer-use APIs, skills/MCP configuration, transcripts, and restoration
are not merely low-level primitives. Sandbox Agent can continue growing
upward into the same UI and orchestration space bot-bottle might occupy.
4. **Coding agents are semantically uneven.** Normalizing a container
lifecycle is easier than claiming full behavioral parity across Claude
Code, Codex, Cursor, Amp, OpenCode, and Pi. A universal schema can become a
lowest common denominator or accumulate provider-specific escape hatches.
5. **The security contract is not standardized.** An agent-session API says
little about whether credentials are visible, egress is controlled, Git is
mediated, or audit is trustworthy. Those are core bot-bottle concerns.
The positioning should therefore say “Docker-like product layer above an open
agent-control protocol,” not “Sandbox Agent is OCI” or “bot-bottle implements
OCI for agents.”
## Can bot-bottle be the turnkey product layer?
Yes, if it owns substantially more than launch syntax.
The turnkey promise is:
> Choose a trusted role, point it at a project, and run any supported coding
> agent with full permissions. bot-bottle builds the environment, isolates it,
> supplies only the capabilities it needs, keeps credentials outside, mediates
> external writes, and gives the operator one place to watch and intervene.
That product has several defensible jobs:
### 1. Packaging and reproducibility
- provider and toolchain images;
- pinned, verified build inputs;
- skills and MCP configuration;
- role/bottle composition;
- cached startup and portable environment definitions; and
- compatibility testing across agents and backends.
### 2. Trusted policy compilation
The manifest is valuable because one reviewable document compiles into:
- an isolation plan;
- gateway routes and DLP policy;
- credential slots;
- Git-gate repositories and identities;
- provider configuration;
- supervision behavior; and
- operator-facing preflight.
Sandbox Agent's runtime configuration does not replace this. The policy must
be resolved before an untrusted guest or agent-control daemon exists.
### 3. Security enforcement
- dedicated-kernel isolation where available;
- no direct guest route to the internet;
- credentials injected outside the agent;
- content inspection on allowed destinations;
- Git secrets scanning and upstream-key custody;
- fail-closed policy resolution; and
- authoritative host-side audit.
This is the strongest current differentiation from a generic
“Sandbox Agent + Docker/E2B” assembly.
### 4. Lifecycle and operations
- install and host preflight;
- image build/update;
- start, stop, resume, cleanup, and migration;
- concurrent named agents;
- state recovery after crashes;
- live supervision and policy remediation; and
- backend selection without changing the role definition.
### 5. Ecosystem and DX
A product layer can support:
- curated provider images;
- signed policy/bottle packs;
- reusable role templates;
- skills and MCP bundles;
- backend plugins;
- an authenticated desktop/web/mobile operator client;
- browser/preview integration;
- normalized transcripts and change review; and
- team policy distribution and compliance exports.
The analogy to Docker is strongest here: users adopt the coherent workflow and
ecosystem, not because the low-level process API is proprietary.
## Business and product positioning
“Turnkey wrapper” is understandable internally but weak externally. It implies
that the hard work lives underneath and that another wrapper can replace it.
Prefer one of:
- **The policy-first runtime for coding agents**
- **Run any coding agent with full permissions, without giving it your host or
credentials**
- **A turnkey local control plane for isolated coding agents**
- **Docker-like packaging and operations for coding agents, with the security
boundary outside the agent**
The open/product split could resemble the container ecosystem:
### Open foundation
- manifest schema and composition;
- local CLI and core orchestrator;
- provider adapters;
- Firecracker/Apple Container/Docker backends;
- gateway policy format and enforcement;
- Sandbox Agent compatibility;
- local audit and supervision; and
- conformance tests for providers/backends/policy.
### Productizable ecosystem/DX
- polished desktop and mobile clients;
- fleet/remote-host management;
- signed and curated role/image/policy registry;
- team policy distribution and administrative controls;
- durable searchable transcripts and audit exports;
- SSO, RBAC, retention, and tamper-evident audit;
- managed update/compatibility channels;
- remote browser/preview relay;
- enterprise support; and
- optional managed build/cache infrastructure.
OCI itself is not the thing Docker sells. Interoperability expands the market;
the product captures value through reliable packaging, workflow, distribution,
management, and trust. bot-bottle should follow that logic rather than trying
to make its session protocol the moat.
## Strategic threat from Sandbox Agent
Sandbox Agent is a real threat for three reasons:
1. **It can become the integration default.** A frontend or agent platform can
integrate one API and choose among many agents and sandbox vendors.
2. **It can own session data and UI.** The universal event schema, Inspector,
React components, restoration, terminal, and computer-use APIs give it a
natural path toward the operator surface.
3. **Sandbox providers can move upward.** If E2B, Daytona, BoxLite, or another
runtime combines Sandbox Agent with adequate network policy and credential
custody, it can offer much of the turnkey stack.
The threat is not that its manifest syntax is better. It currently has no
equivalent trusted policy composition. The threat is that **the ecosystem may
standardize around its API before bot-bottle has a stable external control
surface**. In that world bot-bottle is evaluated as one sandbox provider,
while the SDK and its consumers own the user relationship.
## Why bot-bottle can still win its layer
Sandbox Agent's scope exclusions align with bot-bottle's deepest work:
- it does not choose or operate the sandbox provider;
- it does not mediate Git;
- it does not own network policy;
- it does not securely deliver credentials;
- it does not durably store sessions; and
- it cannot make guest-generated telemetry authoritative.
Those are not incidental features. Together they define the trusted system
around an untrusted coding agent. bot-bottle also has a narrower and coherent
initial customer: a developer or small operator who wants existing agent CLIs
to run locally with broad permissions and bounded consequences.
The durable advantage is therefore:
> Sandbox Agent makes agents controllable. bot-bottle makes them safe and
> operable.
That sentence remains true only if bot-bottle closes its operator-DX gaps.
Security without a browser/preview loop, stable API, normalized session view,
and good parallel-task UX risks becoming an invisible backend feature.
## Integration options
### Option A — Embed Sandbox Agent inside each bottle
bot-bottle launches Sandbox Agent as the provider process supervisor and
connects it to the host orchestrator through a bottle-scoped authenticated
channel.
**Benefits**
- immediate provider-neutral session API;
- more supported agents;
- normalized streaming and transcripts;
- terminal, filesystem, process, and computer-use primitives;
- Inspector/React ecosystem; and
- less provider-specific reverse engineering in bot-bottle.
**Risks**
- `0.x` API/schema churn;
- extra binary and release-supply-chain dependency;
- lowest-common-denominator normalization;
- conflict with provider-native resume state;
- an in-guest daemon is attacker-controlled after guest compromise;
- duplicate orchestration responsibilities; and
- upstream can move into policy/lifecycle and compete more directly.
**Security rule**
Treat every event and state claim from Sandbox Agent as untrusted telemetry.
Never delegate egress authorization, credential release, bottle identity,
authoritative audit, or Git policy to it.
### Option B — Implement a Sandbox Agent-compatible endpoint
bot-bottle maps the external protocol onto its existing provider adapters and
process model without running the upstream server.
**Benefits**
- ecosystem compatibility with tighter component control;
- no in-guest daemon dependency; and
- room to preserve bot-bottle-native lifecycle semantics.
**Risks**
- large and continuing compatibility burden;
- “full feature coverage” is expensive across all providers;
- accidental protocol fork; and
- effort diverted from policy and UX differentiation.
### Option C — Define an independent bot-bottle session API
Build only the control surface bot-bottle needs.
**Benefits**
- clean fit with the trust model and persistent named bottles;
- no upstream dependency; and
- deliberate support for supervision and security events.
**Risks**
- recreates a fast-growing open-source project;
- no existing client ecosystem;
- slower browser/desktop/mobile work; and
- increases the chance that Sandbox Agent becomes the de facto standard first.
### Recommendation
Start with **Option A as a bounded compatibility spike**, not a product
commitment. Do not begin with a clean-room competing protocol.
The spike should answer:
1. Can Claude Code, Codex, and Pi retain exact native resume behavior?
2. Can Sandbox Agent run without receiving real provider credentials?
3. Can its server be reached through a bottle-scoped authenticated channel
without exposing the orchestrator or broadening guest egress?
4. Which permission events overlap or conflict with bot-bottle supervision?
5. Can normalized events be stored while clearly separating untrusted
transcript telemetry from authoritative gateway/Git audit?
6. Can manifest skills, MCP servers, prompt, and startup arguments compile
deterministically into its configuration?
7. Does its versioning policy permit a compatibility contract bot-bottle can
support?
8. What image-size, startup-time, and update burden does the binary add?
If the answers are favorable, adopt it behind a bot-bottle-owned interface and
pin/test the supported version. If not, implement the smallest compatible
subset needed by external clients before inventing a wholly separate API.
## Product roadmap implications
The competitor scan and this architecture comparison reorder the likely work:
1. **Provider-neutral control/session compatibility spike**
2. **Stable authenticated external bot-bottle API**
3. **Normalized transcript/event persistence**
4. **Parallel-session operator UI**
5. **Browser/preview/computer-use capability**
6. **Policy/image/skill distribution and signing**
7. **Remote host/fleet management**
This does not mean pausing security work. It means exposing the shipped
security work through a product surface that can compete with the SDK-plus-
sandbox ecosystem.
## Decision
Treat Sandbox Agent SDK as a potentially standard **agent process-control
layer**, not as a sandbox replacement and not as a minor complementary
library. Position bot-bottle one layer above it:
- manifests express trusted role and environment policy;
- bot-bottle compiles and enforces that policy across host, gateway, Git, and
isolation backends;
- Sandbox Agent or a compatible protocol controls the selected agent process;
and
- bot-bottle owns the turnkey operator experience.
The Docker analogy is strategically sound when stated as:
> Sandbox Agent can be the portable task/exec protocol; bot-bottle can be the
> opinionated engine, Compose-like policy layer, and Desktop-like operator
> product.
It is not sound when stated as:
> Sandbox Agent is OCI and bot-bottle is Docker.
There is no independent OCI-equivalent agent specification yet, and Sandbox
Agent already reaches into UI/session territory. Compatibility should be
pursued quickly, while the trusted manifest/enforcement plane and operator
experience remain the parts bot-bottle deliberately owns.
+15 -1
View File
@@ -48,10 +48,24 @@ class TestDbStoreIsMigrated(unittest.TestCase):
with tempfile.TemporaryDirectory() as d:
parent = Path(d) / "store"
parent.mkdir()
with patch.object(Path, "chmod", side_effect=OSError("denied")):
(parent / "test.db").touch()
with patch("os.fchmod", side_effect=OSError("denied")):
with self.assertRaisesRegex(OSError, "denied"):
_store(parent)
def test_rejects_database_symlink_without_changing_target(self):
with tempfile.TemporaryDirectory() as d:
parent = Path(d) / "store"
parent.mkdir()
target = Path(d) / "target"
target.touch(mode=0o644)
(parent / "test.db").symlink_to(target)
with self.assertRaises(OSError):
_store(parent)
self.assertEqual(0o644, stat.S_IMODE(target.stat().st_mode))
def test_returns_false_when_schema_versions_missing(self):
# DB file exists but has no schema_versions table → OperationalError → False.
with tempfile.TemporaryDirectory() as d:
+1
View File
@@ -5,6 +5,7 @@ import subprocess
import tempfile
import threading
import unittest
import urllib.error
import urllib.request
from pathlib import Path
from unittest import mock
-9
View File
@@ -70,15 +70,6 @@ def dispatch(
response = asyncio.run(request())
payload = response.json()
if response.status_code == 422:
detail = payload.get("detail", []) if isinstance(payload, dict) else []
field = ""
if isinstance(detail, list) and detail and isinstance(detail[0], dict):
location = detail[0].get("loc", ())
if isinstance(location, (list, tuple)) and len(location) > 1:
field = str(location[1])
suffix = f": {field}" if field else ""
return 400, {"error": f"invalid request body{suffix}"}
if isinstance(payload, dict) and "detail" in payload and "error" not in payload:
payload = {"error": payload["detail"]}
return response.status_code, payload
+2 -2
View File
@@ -140,7 +140,7 @@ class TestStoreGuardBranches(unittest.TestCase):
with tempfile.TemporaryDirectory() as d:
db = Path(d) / "q.db"
store = QueueStore("key", db_path=db)
with patch("pathlib.Path.chmod", side_effect=OSError("ro")):
with patch("os.fchmod", side_effect=OSError("ro")):
with self.assertRaisesRegex(OSError, "ro"):
store.migrate()
@@ -156,7 +156,7 @@ class TestStoreGuardBranches(unittest.TestCase):
with tempfile.TemporaryDirectory() as d:
db = Path(d) / "a.db"
store = AuditStore(db_path=db)
with patch("pathlib.Path.chmod", side_effect=OSError("ro")):
with patch("os.fchmod", side_effect=OSError("ro")):
with self.assertRaisesRegex(OSError, "ro"):
store.migrate()