Compare commits
3 Commits
25155a1d5a
...
5ca3209b41
| Author | SHA1 | Date | |
|---|---|---|---|
| 5ca3209b41 | |||
| 727eafe0f9 | |||
| 1ec114b6d7 |
@@ -47,6 +47,7 @@ from .supervise_types import (
|
||||
STATUS_MODIFIED,
|
||||
STATUS_REJECTED,
|
||||
TOOLS,
|
||||
TOOL_CHECK_PROPOSAL,
|
||||
TOOL_EGRESS_ALLOW,
|
||||
TOOL_EGRESS_BLOCK,
|
||||
TOOL_EGRESS_TOKEN_ALLOW,
|
||||
@@ -263,6 +264,7 @@ __all__ = [
|
||||
"TOOLS",
|
||||
"EGRESS_FORWARD_PROXY",
|
||||
"EGRESS_INTROSPECT_URL",
|
||||
"TOOL_CHECK_PROPOSAL",
|
||||
"TOOL_EGRESS_ALLOW",
|
||||
"TOOL_EGRESS_BLOCK",
|
||||
"TOOL_GITLEAKS_ALLOW",
|
||||
|
||||
@@ -2,14 +2,24 @@
|
||||
|
||||
Per-bottle MCP server exposing tools the agent calls to propose egress
|
||||
config changes when stuck. The tools are `egress-allow`,
|
||||
`egress-block`, and `list-egress-routes`.
|
||||
`egress-block`, `list-egress-routes`, and `check-proposal`.
|
||||
|
||||
Each queued tool call:
|
||||
Each queued proposal tool call:
|
||||
|
||||
1. Validates the proposed file syntactically.
|
||||
2. Writes a Proposal to the host SQLite database.
|
||||
3. Blocks polling for a matching Response row.
|
||||
4. Returns the operator's `{status, notes}` to the agent.
|
||||
3. Blocks polling for a matching Response row, up to a short grace
|
||||
window (`SUPERVISE_RESPONSE_TIMEOUT_SECONDS`, default 30s).
|
||||
4. On a decision within the window, returns the operator's
|
||||
`{status, notes}`. On timeout, returns `status: pending` **with the
|
||||
proposal id** and leaves the proposal queued — the flow is
|
||||
non-blocking past the grace window (PRD prd-new / issue #412).
|
||||
|
||||
`check-proposal` is the non-blocking companion: given a `proposal_id`
|
||||
returned by a `pending` response, it reports the current decision
|
||||
(`pending` | `approved` | `modified` | `rejected`) without re-proposing,
|
||||
so an approval made out-of-band (e.g. a web review console) can be resumed
|
||||
without holding an HTTP request open.
|
||||
|
||||
One shared server fronts every bottle (PRD 0070) and attributes each
|
||||
proposal to the calling bottle by source IP, resolved from the orchestrator
|
||||
@@ -22,7 +32,9 @@ Speaks MCP over HTTP+JSON-RPC. Methods handled:
|
||||
* `initialize` — handshake; returns server info + caps.
|
||||
* `notifications/initialized` — ack-only.
|
||||
* `tools/list` — returns the tool definitions.
|
||||
* `tools/call` — validates, queues, blocks, returns.
|
||||
* `tools/call` — validates, queues, waits out the grace
|
||||
window, returns (pending past it); or, for
|
||||
`check-proposal`, a non-blocking status poll.
|
||||
|
||||
Everything else returns JSON-RPC error -32601 (method not found).
|
||||
|
||||
@@ -232,6 +244,31 @@ TOOL_DEFINITIONS: list[dict[str, object]] = [
|
||||
),
|
||||
"inputSchema": _proposal_input_schema(),
|
||||
},
|
||||
{
|
||||
"name": _sv.TOOL_CHECK_PROPOSAL,
|
||||
"description": (
|
||||
"Poll a previously queued proposal for the operator's decision "
|
||||
"WITHOUT blocking or re-proposing. Pass the `proposal_id` you "
|
||||
"got back when an `egress-allow`/`egress-block` call returned "
|
||||
"`status: pending`. Returns the current status: `pending` (no "
|
||||
"decision yet — poll again later), `approved`, `modified`, "
|
||||
"`rejected`, or `unknown` (no such queued proposal — wrong id, "
|
||||
"or it was already resolved and read)."
|
||||
),
|
||||
"inputSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"proposal_id": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"The proposal id from a `pending` response."
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["proposal_id"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@@ -353,7 +390,7 @@ def handle_tools_call(
|
||||
deadline=deadline,
|
||||
)
|
||||
except TimeoutError:
|
||||
text = format_pending_response_text(config.response_timeout_seconds)
|
||||
text = format_pending_response_text(proposal.id, config.response_timeout_seconds)
|
||||
return {
|
||||
"content": [{"type": "text", "text": text}],
|
||||
"isError": False,
|
||||
@@ -370,6 +407,54 @@ def handle_tools_call(
|
||||
}
|
||||
|
||||
|
||||
def handle_check_proposal(
|
||||
params: dict[str, object],
|
||||
config: ServerConfig,
|
||||
) -> dict[str, object]:
|
||||
"""Non-blocking poll of a queued proposal's decision, by id.
|
||||
|
||||
Never creates a Proposal (so `check-proposal` isn't in `TOOLS`); it only
|
||||
reads the queue. Resolution order mirrors the synchronous path's terminal
|
||||
step — a decided proposal is archived here exactly as `handle_tools_call`
|
||||
archives it after `wait_for_response`, so `pending` proposals stay visible
|
||||
to the operator until they're both decided *and* polled."""
|
||||
args_raw = params.get("arguments", {})
|
||||
if not isinstance(args_raw, dict):
|
||||
raise _RpcClientError(ERR_INVALID_PARAMS, "tools/call 'arguments' must be an object")
|
||||
proposal_id = args_raw.get("proposal_id")
|
||||
if not isinstance(proposal_id, str) or not proposal_id.strip():
|
||||
raise _RpcClientError(
|
||||
ERR_INVALID_PARAMS,
|
||||
"check-proposal: 'proposal_id' is required and must be a non-empty string",
|
||||
)
|
||||
proposal_id = proposal_id.strip()
|
||||
|
||||
try:
|
||||
response = _sv.read_response(config.bottle_slug, proposal_id)
|
||||
except FileNotFoundError:
|
||||
# No decision yet — distinguish "still queued" from "unknown id".
|
||||
try:
|
||||
_sv.read_proposal(config.bottle_slug, proposal_id)
|
||||
except FileNotFoundError:
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_unknown_proposal_text(proposal_id)}],
|
||||
"isError": True,
|
||||
}
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_still_pending_text(proposal_id)}],
|
||||
"isError": False,
|
||||
}
|
||||
|
||||
try:
|
||||
_sv.archive_proposal(config.bottle_slug, proposal_id)
|
||||
except OSError as e:
|
||||
raise _RpcInternalError(f"failed to archive proposal: {e}") from e
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_response_text(response)}],
|
||||
"isError": response.status == _sv.STATUS_REJECTED,
|
||||
}
|
||||
|
||||
|
||||
def format_response_text(response: "_sv.Response") -> str:
|
||||
"""Pretty-print a Response for the tool's text content. The agent
|
||||
reads the text and decides whether to retry / give up / surface."""
|
||||
@@ -382,12 +467,35 @@ def format_response_text(response: "_sv.Response") -> str:
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def format_pending_response_text(timeout_seconds: float) -> str:
|
||||
def format_pending_response_text(proposal_id: str, timeout_seconds: float) -> str:
|
||||
"""Grace-window timeout: the proposal stays queued, and the agent is
|
||||
told the id so it can `check-proposal` instead of re-proposing."""
|
||||
return "\n".join([
|
||||
"status: pending",
|
||||
f"proposal_id: {proposal_id}",
|
||||
(
|
||||
"notes: operator response timed out after "
|
||||
f"{timeout_seconds:g}s; proposal remains queued"
|
||||
f"notes: no operator decision within {timeout_seconds:g}s; the "
|
||||
"proposal remains queued. Poll it (do not re-propose) by calling "
|
||||
f"`check-proposal` with proposal_id={proposal_id!r}."
|
||||
),
|
||||
])
|
||||
|
||||
|
||||
def format_still_pending_text(proposal_id: str) -> str:
|
||||
return "\n".join([
|
||||
"status: pending",
|
||||
f"proposal_id: {proposal_id}",
|
||||
"notes: still queued; no operator decision yet. Call `check-proposal` again later.",
|
||||
])
|
||||
|
||||
|
||||
def format_unknown_proposal_text(proposal_id: str) -> str:
|
||||
return "\n".join([
|
||||
"status: unknown",
|
||||
f"proposal_id: {proposal_id}",
|
||||
(
|
||||
"notes: no queued proposal with this id for this bottle — the id "
|
||||
"may be wrong, or the proposal was already resolved and read."
|
||||
),
|
||||
])
|
||||
|
||||
@@ -482,6 +590,11 @@ class MCPHandler(http.server.BaseHTTPRequestHandler):
|
||||
# — silently dropping base routes like api.anthropic.com on approval.
|
||||
if req.params.get("name") == _sv.TOOL_LIST_EGRESS_ROUTES:
|
||||
return self._resolved_routes_payload()
|
||||
# `check-proposal` is a non-blocking read of the calling bottle's
|
||||
# own queue — attributed by source IP like a proposal, but it
|
||||
# never queues or blocks.
|
||||
if req.params.get("name") == _sv.TOOL_CHECK_PROPOSAL:
|
||||
return handle_check_proposal(req.params, self._attributed_config(config))
|
||||
# Attribute the proposal to the source-IP-resolved bottle, so the one
|
||||
# shared server queues each bottle's proposal under its own slug.
|
||||
return handle_tools_call(req.params, self._attributed_config(config))
|
||||
|
||||
@@ -20,6 +20,10 @@ TOOL_EGRESS_ALLOW = "egress-allow"
|
||||
TOOL_GITLEAKS_ALLOW = "gitleaks-allow"
|
||||
TOOL_EGRESS_TOKEN_ALLOW = "egress-token-allow"
|
||||
TOOL_LIST_EGRESS_ROUTES = "list-egress-routes"
|
||||
# Read-only agent tool: poll a queued proposal for the operator's decision
|
||||
# without blocking or re-proposing. It never becomes a `Proposal.tool` (no
|
||||
# queue record is created for it), so it is intentionally NOT in `TOOLS`.
|
||||
TOOL_CHECK_PROPOSAL = "check-proposal"
|
||||
TOOLS: tuple[str, ...] = (
|
||||
TOOL_EGRESS_ALLOW,
|
||||
TOOL_EGRESS_BLOCK,
|
||||
@@ -156,6 +160,7 @@ __all__ = [
|
||||
"TOOLS",
|
||||
"TOOL_EGRESS_ALLOW",
|
||||
"TOOL_EGRESS_BLOCK",
|
||||
"TOOL_CHECK_PROPOSAL",
|
||||
"TOOL_EGRESS_TOKEN_ALLOW",
|
||||
"TOOL_GITLEAKS_ALLOW",
|
||||
"TOOL_LIST_EGRESS_ROUTES",
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
# PRD prd-new: Non-blocking supervise (async approval + proposal polling)
|
||||
|
||||
- **Status:** Draft
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-07-18
|
||||
- **Issue:** #412
|
||||
|
||||
## Summary
|
||||
|
||||
The per-bottle supervise MCP server (`bot_bottle/supervise_server.py`)
|
||||
answers `tools/call` **synchronously**: it queues the agent's proposal and
|
||||
blocks the tool call polling for the operator's decision. On timeout it
|
||||
returns `status: pending` and leaves the proposal queued — but it hands the
|
||||
agent **no proposal id** and offers **no way to poll a specific pending
|
||||
proposal**, so the only way to learn the outcome is to re-propose (a
|
||||
duplicate).
|
||||
|
||||
This PRD makes the MCP flow non-blocking and pollable, so an approval can
|
||||
happen out-of-band (a human taking minutes-to-hours in a review console)
|
||||
without holding an HTTP request open or wedging the agent:
|
||||
|
||||
1. Include the `proposal_id` in the `pending` response.
|
||||
2. Add a `check-proposal` MCP tool: a non-blocking status lookup by
|
||||
proposal id.
|
||||
3. Keep the short synchronous grace window for the common "operator is
|
||||
right there" fast path.
|
||||
|
||||
## Problem
|
||||
|
||||
`handle_tools_call` → `_sv.wait_for_response(...)` blocks up to
|
||||
`SUPERVISE_RESPONSE_TIMEOUT_SECONDS` (default 30s). Two problems follow:
|
||||
|
||||
- **Human latency ≠ tool-call latency.** A real review — rendered diff,
|
||||
RBAC routing to an approver, someone tapping approve on their phone — is
|
||||
minutes-to-hours. Holding the MCP request open that long is fragile
|
||||
(proxy/keepalive timeouts, the mitmproxy egress hop, and the agent
|
||||
harness's own tool-call timeout, which a long block can trip and stall
|
||||
the whole turn).
|
||||
- **No resume path.** The pending fallback already exists, but without a
|
||||
proposal id and a poll tool the agent can't reconnect to that specific
|
||||
decision — it re-proposes, duplicating the queue entry.
|
||||
|
||||
This is also the precondition for the planned web-console human-review
|
||||
flow (RBAC, audit retention, mobile) — see issue #412.
|
||||
|
||||
**Safety note:** the MCP tools only *propose* policy changes; enforcement
|
||||
stays at the egress proxy and the git-gate. Returning early on `pending`
|
||||
therefore opens no hole — the agent still cannot egress or push anything
|
||||
unapproved.
|
||||
|
||||
## Goals / success criteria
|
||||
|
||||
- A `pending` MCP response carries the `proposal_id`.
|
||||
- An agent can call `check-proposal(proposal_id)` and get the current
|
||||
state (`pending` | `approved` | `modified` | `rejected`) **without
|
||||
blocking** and **without creating a new proposal**.
|
||||
- The synchronous fast path (operator approves within the grace window) is
|
||||
unchanged: the first `tools/call` still returns the decision directly.
|
||||
- No change to enforcement, attribution (source-IP → bottle), or the
|
||||
operator-side queue/response schema.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- The git-gate `pre-receive` path (it is synchronous by nature and cannot
|
||||
poll — its async variant is reject-fast + re-push; tracked as a
|
||||
follow-up).
|
||||
- Backpressure / in-flight-proposal caps.
|
||||
- MCP server→client notifications (event-driven resume).
|
||||
- Any web-console UI (this PRD is the protocol groundwork it needs).
|
||||
|
||||
## Design
|
||||
|
||||
### `pending` response carries the id
|
||||
|
||||
`handle_tools_call`'s timeout branch formats the pending text with the
|
||||
`proposal.id` and a pointer to `check-proposal`, so the agent knows what to
|
||||
poll.
|
||||
|
||||
### `check-proposal` tool
|
||||
|
||||
A new read-only MCP tool (`TOOL_CHECK_PROPOSAL = "check-proposal"`),
|
||||
attributed to the calling bottle by source IP exactly like the proposal
|
||||
tools. Input: `{ "proposal_id": string }`. Behavior:
|
||||
|
||||
1. `read_response(slug, id)` →
|
||||
- **found**: archive the proposal (same terminal step the synchronous
|
||||
path takes) and return the decision via `format_response_text`;
|
||||
`isError` iff rejected.
|
||||
2. **not found** → `read_proposal(slug, id)` →
|
||||
- **found**: still queued → return `status: pending`.
|
||||
- **not found**: unknown id, or already resolved-and-archived (e.g. a
|
||||
second poll) → return `status: unknown`, `isError: true`.
|
||||
|
||||
Both lookups already raise `FileNotFoundError` when absent
|
||||
(`queue_store.py`), so the handler needs no new store methods. `check-`
|
||||
`proposal` is the only path (besides the synchronous response) that
|
||||
archives, so a proposal that times out to `pending` stays visible to the
|
||||
operator until it is decided and then polled.
|
||||
|
||||
### Grace window
|
||||
|
||||
Left at the existing 30s default (`SUPERVISE_RESPONSE_TIMEOUT_SECONDS`),
|
||||
which doubles as the instant-approve fast path. Tuning it down is an
|
||||
operator setting, not a code change; noted for the console rollout.
|
||||
|
||||
## Implementation chunks
|
||||
|
||||
1. **(this PR)** `TOOL_CHECK_PROPOSAL` constant; `check-proposal` tool
|
||||
definition + `handle_check_proposal`; dispatch wiring; `proposal_id` in
|
||||
the pending text; unit tests. Files: `bot_bottle/supervise_types.py`,
|
||||
`bot_bottle/supervise.py` (re-export), `bot_bottle/supervise_server.py`,
|
||||
`tests/unit/test_supervise_server.py`.
|
||||
2. **(follow-up)** git-gate `pre-receive` reject-fast + re-push.
|
||||
3. **(follow-up)** per-bottle in-flight-proposal backpressure cap.
|
||||
4. **(follow-up)** MCP notifications for event-driven resume; web-console
|
||||
review flow (RBAC, audit retention) on top.
|
||||
|
||||
## Open questions
|
||||
|
||||
- Should a resolved-but-unpolled proposal auto-archive after some TTL, or
|
||||
only on poll? (Leaning: only on poll, so a decision is never lost to a
|
||||
reaper before the agent sees it.)
|
||||
- Does the agent harness need an explicit "you have a pending proposal"
|
||||
nudge, or is returning `pending` from the original call enough? (Deferred
|
||||
to the notifications chunk.)
|
||||
@@ -0,0 +1,308 @@
|
||||
# HN discourse on agent sandbox safety — June/July 2026
|
||||
|
||||
A survey of community opinion and notable security disclosures on Hacker
|
||||
News and adjacent sources over June–July 2026. The question: what does
|
||||
the current discourse say about whether sandboxes are sufficient for
|
||||
agentic AI safety, and where does bot-bottle land against the issues
|
||||
being raised?
|
||||
|
||||
Research conducted 2026-07-18.
|
||||
|
||||
## Summary
|
||||
|
||||
The past month marks a turning point in community opinion. Earlier in
|
||||
2026, the debate was mostly "which sandbox tool is best?" By June–July,
|
||||
a cascade of critical CVEs and novel attack classes has shifted the
|
||||
framing to "sandboxes are not enough — what else do you need?" The
|
||||
attacks that drove this shift are structurally distinct: most route
|
||||
through legitimate, trusted channels (Sentry issues, MCP descriptions,
|
||||
README files) rather than exploiting the isolation boundary directly.
|
||||
|
||||
bot-bottle's architecture holds up well against the direct-escape class
|
||||
(Firecracker/Apple Container default backends, credentials never in the
|
||||
agent's env, harness entirely on the host). The remaining gap is prompt
|
||||
injection — attacker-controlled data interpreted as model instructions.
|
||||
Egress controls and prompt injection defenses are orthogonal: egress
|
||||
limits what the agent can *send out*; injection is about what it is
|
||||
*told to do*. The two don't substitute for each other. Inside a tightly-
|
||||
egressed sandbox a successful injection can't exfiltrate to unknown
|
||||
hosts, but it can still corrupt the work product, push malicious commits
|
||||
past a secret scanner, or use allowlisted channels for exfiltration.
|
||||
Those residual risks are addressed below.
|
||||
|
||||
## The sandboxing boom sets the stage
|
||||
|
||||
The preceding months generated a wave of sandbox tooling. A March 28
|
||||
Ask HN thread
|
||||
([#47444917](https://news.ycombinator.com/item?id=47444917)) catalogued
|
||||
the explosion: E2B, AIO Sandbox, AgentSphere, Yolobox, Exe.dev,
|
||||
AgentFence, DenoSandbox, Capsule (WASM), ERA, Vibekit, Daytona, Modal,
|
||||
Nono, and more — all launched within roughly 12 months. A parallel March
|
||||
9 thread ([#47185250](https://news.ycombinator.com/item?id=47185250))
|
||||
surveyed what developers were actually deploying: "containers or YOLO"
|
||||
dominated. The honest community mood was that most teams hadn't solved
|
||||
this and were shipping anyway.
|
||||
|
||||
## The June–July attack cascade
|
||||
|
||||
Six attack patterns broke in quick succession. Together they form the
|
||||
argument that the community's framing was wrong: the threat model for
|
||||
agents isn't just "code that escapes its container" — it's also prompt
|
||||
injection, where attacker-controlled data is interpreted as model
|
||||
instructions regardless of whether any isolation boundary was crossed.
|
||||
Sections 2–4 below are all the same attack class; the "trusted channel"
|
||||
label describes the delivery vector, not a different threat.
|
||||
|
||||
### 1. Sandbox escape CVEs (DuneSlide, CVE-2026-39861)
|
||||
|
||||
Cato AI Labs disclosed **DuneSlide** (CVE-2026-50548/50549, CVSS 9.8),
|
||||
a pair of flaws in Cursor 2.x. CVE-2026-50548 abuses the sandbox's
|
||||
`working_directory` parameter to point writes at system files; CVE-26-50549
|
||||
exploits a symlink-resolution fallback that fails open. Both start with
|
||||
a prompt injection and end in sandbox escape — and Cato's framing was
|
||||
blunt: "each CVE defeats a different guardrail; the problem is
|
||||
structural, not a string of one-offs."
|
||||
|
||||
Claude Code's own sandbox had a similar escape this year:
|
||||
**CVE-2026-39861** (symlink flaw). The CurXecute/MCPoison/CVE-2026-26268
|
||||
chain from Cursor added a poisoned Slack message, a swap-after-approval
|
||||
MCP config, and a Git hook as three more entry points in the same
|
||||
attack class.
|
||||
|
||||
All patched, but the pattern holds: any application-level sandbox that
|
||||
takes attacker-influenced values as path parameters is reachable from a
|
||||
prompt injection.
|
||||
|
||||
### 2. Prompt injection via MCP data (Agentjacking)
|
||||
|
||||
Tenet's "Agentjacking" technique planted a fake bug report in Sentry's
|
||||
MCP output. When an agent queries Sentry to fix open issues, the
|
||||
malicious event is rendered as structured content visually
|
||||
indistinguishable from a real Sentry event, and the agent executes the
|
||||
embedded instructions with the developer's full privileges. Hit rate
|
||||
across Claude Code and Cursor: **85%**. The route is entirely through a
|
||||
legitimately-authorized MCP channel — no isolation boundary is crossed;
|
||||
the injection arrives inbound through a channel the sandbox explicitly
|
||||
trusts.
|
||||
|
||||
The Cloud Security Alliance's summary: treat observability, bug-report,
|
||||
and integration data as **untrusted agent input**, not neutral
|
||||
development metadata.
|
||||
|
||||
### 3. README-embedded prompt injection
|
||||
|
||||
A July disclosure showed malicious instructions hidden in `README.md`
|
||||
— a file that receives no trust prompt and requires no elevated access.
|
||||
When asked point-blank whether the repo held hidden instructions, both
|
||||
Claude Sonnet 4.6 and GPT-5.5 said no. A payload written for Sonnet
|
||||
4.6 transferred unchanged to Sonnet 5, Opus 4.8, and GPT-5.5. The
|
||||
attack surface is every repo an agent is asked to work in.
|
||||
|
||||
### 4. Prompt injection via MCP tool descriptions
|
||||
|
||||
Microsoft research (June 30) showed that attacker-controlled MCP tool
|
||||
description fields can silently redirect agent behavior. The injection
|
||||
is embedded in metadata the model reads during tool selection — before
|
||||
any sandbox enforcement or egress check runs, and entirely on the
|
||||
inbound path that egress controls cannot touch.
|
||||
|
||||
### 5. MCP STDIO command injection (10 CVEs)
|
||||
|
||||
OX Security disclosed a systemic command injection class in Anthropic's
|
||||
MCP protocol, covering 10 CVEs across multiple coding agents. The
|
||||
Windsurf case (CVE-2026-30615): processing attacker-controlled HTML
|
||||
causes the agent to auto-register a malicious MCP STDIO server and
|
||||
execute arbitrary commands with no further user interaction.
|
||||
|
||||
### 6. LiteLLM gateway compromise (CVE-2026-40217, CVE-2026-42271)
|
||||
|
||||
CVE-2026-40217 exposes LiteLLM's guardrail sandbox via `exec()` with no
|
||||
source filtering. CVE-2026-42271 (exploited in the wild, added to CISA's
|
||||
KEV catalog) lets callers spawn subprocesses through MCP preview
|
||||
endpoints. The threat extends to any agent routed through a compromised
|
||||
LiteLLM proxy: the proxy can swap model responses for forged tool calls
|
||||
in transit, giving the attacker a reverse shell from the developer's
|
||||
machine.
|
||||
|
||||
## HN community opinion clusters
|
||||
|
||||
**"Move enforcement to the kernel, not the app"** — the Nono Show HN
|
||||
([#46849615](https://news.ycombinator.com/item?id=46849615)) and a
|
||||
kernel-sandbox thread
|
||||
([#47066574](https://news.ycombinator.com/item?id=47066574)) both argued
|
||||
that application-layer sandboxes are inherently bypassable by the code
|
||||
they're sandboxing. The academic framing, from *Red-Teaming the Agentic
|
||||
Red-Team* ([arXiv 2606.24496](https://arxiv.org/pdf/2606.24496)):
|
||||
"enforcement should occur at the OS level via the kernel refusing system
|
||||
calls that violate policy at runtime — not pre-execution argument
|
||||
validation in tool calls."
|
||||
|
||||
**"The harness belongs outside the sandbox"** — a May thread
|
||||
([#47990675](https://news.ycombinator.com/item?id=47990675)) converged
|
||||
on clean architectural separation: harness in one VM, tool execution in
|
||||
another. Top comment: "having the harness in one VM, and tool use applied
|
||||
to user data in another, is about as safe as you can be at present."
|
||||
Several replies described a hypervisor-like policy layer — sitting outside
|
||||
both VMs — as the right long-term model.
|
||||
|
||||
**"Sandboxes are too coarse-grained"** — a Feb thread
|
||||
([#47006445](https://news.ycombinator.com/item?id=47006445)) argued
|
||||
that VMs don't answer the real question: knowing whether an agent
|
||||
*should* be sending an email or making a transaction. "Everything's just
|
||||
in the same big box." This framing picked up traction through June–July
|
||||
as the trusted-channel attacks dominated.
|
||||
|
||||
**"MCP's trust model is the real problem"** — the month's recurring
|
||||
theme. MCP by design gives agents access to authorized external services.
|
||||
Once a trusted channel delivers a malicious payload, filesystem sandboxing
|
||||
is irrelevant. The community call: treat all MCP tool metadata and return
|
||||
values as untrusted input subject to policy validation before ingestion,
|
||||
and disable automatic MCP server loading from untrusted repositories.
|
||||
|
||||
## How bot-bottle addresses these issues
|
||||
|
||||
### What it covers well
|
||||
|
||||
**Direct sandbox escape (CVEs, container breakout)**
|
||||
|
||||
bot-bottle's default backends are Firecracker microVM (KVM Linux) and
|
||||
Apple Container (macOS). Both run the agent in a separate VM with a
|
||||
dedicated kernel — the container-escape CVE class (Dirty Pipe, runc
|
||||
escapes, DuneSlide's path-parameter abuse) requires escaping a real
|
||||
hypervisor boundary, not just a namespace. On the legacy Docker backend,
|
||||
gVisor auto-detection provides a userspace syscall barrier for hosts where
|
||||
neither KVM nor Apple Container is available.
|
||||
|
||||
The bot-bottle process itself runs entirely on the host, outside the VM.
|
||||
This is the "harness outside the sandbox" architecture the HN thread
|
||||
converged on as best practice. The bottle manifest, egress rules, and
|
||||
secrets never enter the agent VM.
|
||||
|
||||
**Credential theft on sandbox escape**
|
||||
|
||||
Even on a successful VM/container escape, the agent has nothing useful
|
||||
to steal. Credentials are injected in-flight by the gateway proxy
|
||||
(`auth.scheme` / `auth.token_ref` in the egress route config) — `printenv`
|
||||
inside the agent shows proxy URLs only. The git-gate similarly holds the
|
||||
upstream SSH credential on the host; the agent pushes through a
|
||||
gitleaks-scanned daemon that forwards clean refs upstream. An escaped
|
||||
agent gets the host filesystem, not the keys.
|
||||
|
||||
**Orphaned-agent credential risk**
|
||||
|
||||
bot-bottle is explicitly ephemeral: when the agent exits, `cli.py` tears
|
||||
down every gateway and both networks — nothing persists between runs. The
|
||||
agent never holds credentials, so there is nothing to orphan.
|
||||
|
||||
**MCP config redirection / STDIO auto-registration**
|
||||
|
||||
The trust boundary at `$HOME` means bottles live only under
|
||||
`~/.bot-bottle/bottles/` — a cloned repo cannot add egress routes or
|
||||
redirect env vars to attacker hosts (the design rationale is in
|
||||
`docs/prds/0011-per-file-md-manifest.md`). Auto-registering a malicious
|
||||
MCP STDIO server from within the agent is still sandboxed by the VM, and
|
||||
any outbound calls from that server must pass the egress allowlist and
|
||||
outbound DLP scanner.
|
||||
|
||||
**Outbound exfiltration (any injection class)**
|
||||
|
||||
Whatever triggers the agent — README injection, Agentjacking, MCP
|
||||
description poisoning — the final step in most attacks is exfiltration.
|
||||
bot-bottle's egress allowlist is default-deny with a per-bottle host
|
||||
allowlist; unknown hosts get a hard 403. Outbound DLP scanning
|
||||
(`outbound_detectors: [token_patterns, known_secrets]`) catches tokens
|
||||
and secrets in outbound bodies; the `supervise` policy (default for
|
||||
manifest routes) holds the request for operator approval rather than
|
||||
silently blocking it. Together these limit what a successful injection
|
||||
can *do* even if it succeeds at the model layer.
|
||||
|
||||
**LiteLLM / compromised-proxy attacks**
|
||||
|
||||
bot-bottle does not use LiteLLM. The model API route (e.g.
|
||||
`api.anthropic.com`) is an auto-injected provider route on the egress
|
||||
allowlist; the agent dials the gateway, not the model API directly.
|
||||
A compromised third-party proxy is not in the architecture.
|
||||
|
||||
### Where it is weaker
|
||||
|
||||
**Prompt injection**
|
||||
|
||||
Egress controls and prompt injection defenses are orthogonal. Egress
|
||||
limits what the agent can *send out* (outbound leg); prompt injection
|
||||
is about what attacker-controlled data *tells the agent to do* (inbound
|
||||
leg). The two don't substitute for each other and must be treated
|
||||
separately.
|
||||
|
||||
The inbound DLP scanner (`inbound_detectors: [naive_injection_detection]`)
|
||||
is the only runtime defense against injection arriving through allowlisted
|
||||
channels — Sentry MCP responses, MCP tool descriptions, README content.
|
||||
It is explicitly pattern-matching and will not catch a sufficiently
|
||||
crafted payload. There is no semantic / intent-level gate between what
|
||||
the model decides and what the agent executes.
|
||||
|
||||
**Blast radius within the permitted scope**
|
||||
|
||||
Inside a tightly-egressed sandbox a successful injection can't
|
||||
exfiltrate to unknown hosts, but it still has real options:
|
||||
|
||||
- *Work product corruption.* The agent can modify, delete, or backdoor
|
||||
files in the working directory. This is within its permitted scope;
|
||||
egress controls have nothing to say about it.
|
||||
|
||||
- *Malicious commits past the git-gate.* The git-gate scans outbound
|
||||
refs for secrets (gitleaks), not for semantic code intent. A prompt-
|
||||
injected agent can commit subtly malicious code — logic bombs,
|
||||
backdoored auth paths, code that exfiltrates data through the
|
||||
application's own HTTP clients at runtime — that looks clean to a
|
||||
secret scanner.
|
||||
|
||||
- *Exfiltration through allowlisted channels.* If an attacker knows or
|
||||
can predict what hosts are in the egress allowlist, those channels are
|
||||
available for exfiltration. A GitHub remote being allowlisted means
|
||||
"push to an attacker-controlled fork" is viable. A logging endpoint
|
||||
being allowlisted means structured data can leave through it. The
|
||||
outbound DLP scanner catches credential tokens and known secrets but
|
||||
not arbitrary business data.
|
||||
|
||||
- *Dependency installation within the sandbox.* An agent that runs
|
||||
`npm install` or `pip install` on attacker-specified packages executes
|
||||
code inside the sandbox with the same capabilities the agent has:
|
||||
filesystem access, tool calls, calls to allowlisted hosts. Supply chain
|
||||
injection via package names is in the same injection family, triggered
|
||||
by the same prompt-injection path.
|
||||
|
||||
### What would close the remaining gaps
|
||||
|
||||
The blast-radius risks above point at two distinct mitigations that
|
||||
don't yet exist in bot-bottle:
|
||||
|
||||
- *Outbound intent classification.* The egress addon today scans
|
||||
outbound request content for token patterns. What it lacks is
|
||||
awareness of context — it can't distinguish "agent is pushing a
|
||||
legitimate commit" from "agent was injected and is pushing a backdoor."
|
||||
The `supervise` policy is already the right shape for human-in-the-loop
|
||||
review on sensitive outbound actions; extending it with context from
|
||||
the agent's recent tool calls (what files were touched, what was the
|
||||
triggering task) would narrow the gap.
|
||||
|
||||
- *Semantic code review on git push.* gitleaks is the wrong tool for
|
||||
catching injected logic. A review step on outbound commits — even a
|
||||
simple diff summary surfaced in `cli.py supervise` before the push is
|
||||
forwarded — would close the malicious-commit path without requiring
|
||||
the agent to be fully trusted.
|
||||
|
||||
## Sources
|
||||
|
||||
- [Ask HN: The new wave of AI agent sandboxes? (Mar 2026)](https://news.ycombinator.com/item?id=47444917)
|
||||
- [OK, let's survey how everybody is sandboxing AI coding agents (Mar 2026)](https://news.ycombinator.com/item?id=47185250)
|
||||
- [The agent harness belongs outside the sandbox (May 2026)](https://news.ycombinator.com/item?id=47990675)
|
||||
- [Show HN: Nono – Kernel-enforced sandboxing for AI agents (Feb 2026)](https://news.ycombinator.com/item?id=46849615)
|
||||
- [Kernel-enforced sandbox for AI agents, MCP and LLM workloads (Feb 2026)](https://news.ycombinator.com/item?id=47066574)
|
||||
- [Sandboxes will be left in 2026 (Feb 2026)](https://news.ycombinator.com/item?id=47006445)
|
||||
- [Critical Cursor Flaws / DuneSlide – The Hacker News](https://thehackernews.com/2026/07/critical-cursor-flaws-could-let-prompt.html)
|
||||
- [Agentjacking Attack – The Hacker News](https://thehackernews.com/2026/06/agentjacking-attack-tricks-ai-coding.html)
|
||||
- [Friendly Fire: AI Agents Built to Catch Malicious Code – The Hacker News](https://thehackernews.com/2026/07/friendly-fire-ai-agents-built-to-catch.html)
|
||||
- [Microsoft Warns Poisoned MCP Tool Descriptions – The Hacker News](https://thehackernews.com/2026/06/microsoft-warns-poisoned-mcp-tool.html)
|
||||
- [MCP STDIO Command Injection Advisory – OX Security](https://www.ox.security/blog/mcp-supply-chain-advisory-rce-vulnerabilities-across-the-ai-ecosystem/)
|
||||
- [LiteLLM Vulnerability Chain – The Hacker News](https://thehackernews.com/2026/06/litellm-vulnerability-chain-lets-low.html)
|
||||
- [Red-Teaming the Agentic Red-Team (arXiv 2606.24496)](https://arxiv.org/pdf/2606.24496)
|
||||
@@ -32,7 +32,9 @@ from bot_bottle.supervise_server import (
|
||||
_RpcError,
|
||||
_RpcInternalError,
|
||||
_response_timeout_from_env,
|
||||
format_pending_response_text,
|
||||
format_response_text,
|
||||
handle_check_proposal,
|
||||
handle_initialize,
|
||||
handle_tools_call,
|
||||
handle_tools_list,
|
||||
@@ -218,6 +220,7 @@ class TestHandleToolsList(unittest.TestCase):
|
||||
_sv.TOOL_EGRESS_ALLOW,
|
||||
_sv.TOOL_EGRESS_BLOCK,
|
||||
_sv.TOOL_LIST_EGRESS_ROUTES,
|
||||
_sv.TOOL_CHECK_PROPOSAL,
|
||||
]),
|
||||
sorted(names),
|
||||
)
|
||||
@@ -484,9 +487,10 @@ class TestFormatResponseText(unittest.TestCase):
|
||||
|
||||
class TestFormatPendingResponseText(unittest.TestCase):
|
||||
def test_formats_timeout_message(self):
|
||||
text = supervise_server.format_pending_response_text(12.5)
|
||||
text = supervise_server.format_pending_response_text("prop-9", 12.5)
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn("12.5s", text)
|
||||
self.assertIn("proposal_id: prop-9", text)
|
||||
|
||||
|
||||
# --- End-to-end HTTP sanity ------------------------------------------------
|
||||
@@ -685,5 +689,129 @@ class TestResolvedRoutesPayload(unittest.TestCase):
|
||||
_handler(None)._resolved_routes_payload()
|
||||
|
||||
|
||||
class TestNonBlockingSupervise(unittest.TestCase):
|
||||
"""PRD prd-new / issue #412: pending responses carry the proposal id, and
|
||||
`check-proposal` polls a queued proposal without blocking or re-proposing."""
|
||||
|
||||
_ROUTES = "routes:\n - host: example.com\n"
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory(prefix="supervise-nonblock-test.")
|
||||
self._home_patch = use_bottle_root(Path(self._tmp.name) / ".bot-bottle")
|
||||
self.config = ServerConfig(bottle_slug="dev")
|
||||
_qs.QueueStore("dev").migrate()
|
||||
_as.AuditStore().migrate()
|
||||
|
||||
def tearDown(self):
|
||||
self._home_patch()
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _seed_proposal(self) -> "_sv.Proposal":
|
||||
p = _sv.Proposal.new(
|
||||
bottle_slug="dev",
|
||||
tool=_sv.TOOL_EGRESS_ALLOW,
|
||||
proposed_file=self._ROUTES,
|
||||
justification="need example.com",
|
||||
current_file_hash=_sv.sha256_hex(self._ROUTES),
|
||||
)
|
||||
_sv.write_proposal(p)
|
||||
return p
|
||||
|
||||
def _check(self, proposal_id: str) -> dict[str, object]:
|
||||
return handle_check_proposal({"arguments": {"proposal_id": proposal_id}}, self.config)
|
||||
|
||||
# --- pending response carries the id ---
|
||||
|
||||
def test_pending_text_includes_id_and_pointer(self):
|
||||
text = format_pending_response_text("abc-123", 30.0)
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn("proposal_id: abc-123", text)
|
||||
self.assertIn("check-proposal", text)
|
||||
|
||||
def test_tools_call_timeout_returns_pending_with_id_and_stays_queued(self):
|
||||
# No responder → the grace window expires → pending, not blocked forever.
|
||||
result = handle_tools_call(
|
||||
{
|
||||
"name": _sv.TOOL_EGRESS_ALLOW,
|
||||
"arguments": {"routes_yaml": self._ROUTES, "justification": "x"},
|
||||
},
|
||||
ServerConfig(bottle_slug="dev", response_timeout_seconds=0.05),
|
||||
)
|
||||
self.assertFalse(result["isError"]) # type: ignore[index]
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: pending", text)
|
||||
pending = _sv.list_pending_proposals("dev")
|
||||
self.assertEqual(1, len(pending)) # still queued, not archived
|
||||
self.assertIn(pending[0].id, text) # agent got the id to poll
|
||||
|
||||
# --- check-proposal poll ---
|
||||
|
||||
def test_check_returns_approved_and_archives(self):
|
||||
p = self._seed_proposal()
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=p.id, status=_sv.STATUS_APPROVED, notes="ok"))
|
||||
result = self._check(p.id)
|
||||
self.assertFalse(result["isError"])
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: approved", text)
|
||||
self.assertIn("notes: ok", text)
|
||||
with self.assertRaises(FileNotFoundError): # archived on read
|
||||
_sv.read_proposal("dev", p.id)
|
||||
|
||||
def test_check_rejected_sets_isError(self):
|
||||
p = self._seed_proposal()
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=p.id, status=_sv.STATUS_REJECTED, notes="no"))
|
||||
result = self._check(p.id)
|
||||
self.assertTrue(result["isError"])
|
||||
self.assertIn("status: rejected", result["content"][0]["text"]) # type: ignore[index]
|
||||
|
||||
def test_check_pending_when_no_decision_yet(self):
|
||||
p = self._seed_proposal()
|
||||
result = self._check(p.id)
|
||||
self.assertFalse(result["isError"])
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn(p.id, text)
|
||||
self.assertEqual(1, len(_sv.list_pending_proposals("dev"))) # not archived
|
||||
|
||||
def test_check_unknown_id_is_error(self):
|
||||
result = self._check("no-such-proposal")
|
||||
self.assertTrue(result["isError"])
|
||||
self.assertIn("status: unknown", result["content"][0]["text"]) # type: ignore[index]
|
||||
|
||||
def test_check_missing_id_raises(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": {}}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_check_empty_id_raises(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": {"proposal_id": " "}}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_check_arguments_must_be_object(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": []}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_full_nonblocking_round_trip(self):
|
||||
# 1. tools/call times out → pending with id
|
||||
result = handle_tools_call(
|
||||
{
|
||||
"name": _sv.TOOL_EGRESS_ALLOW,
|
||||
"arguments": {"routes_yaml": self._ROUTES, "justification": "x"},
|
||||
},
|
||||
ServerConfig(bottle_slug="dev", response_timeout_seconds=0.05),
|
||||
)
|
||||
pid = _sv.list_pending_proposals("dev")[0].id
|
||||
self.assertIn(pid, result["content"][0]["text"]) # type: ignore[index]
|
||||
# 2. operator decides out-of-band
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=pid, status=_sv.STATUS_APPROVED, notes="ok"))
|
||||
# 3. agent resumes by polling — no re-proposing
|
||||
poll = self._check(pid)
|
||||
self.assertFalse(poll["isError"])
|
||||
self.assertIn("status: approved", poll["content"][0]["text"]) # type: ignore[index]
|
||||
self.assertEqual([], _sv.list_pending_proposals("dev")) # resolved + archived
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
Reference in New Issue
Block a user