Compare commits

..

4 Commits

Author SHA1 Message Date
didericis-codex cf4a771e52 test(macos): isolate registration tests from container discovery
test / integration-docker (pull_request) Successful in 11s
test / unit (pull_request) Successful in 32s
lint / lint (push) Successful in 44s
test / stage-firecracker-inputs (pull_request) Successful in 4s
tracker-policy-pr / check-pr (pull_request) Successful in 8s
test / build-infra (pull_request) Successful in 4m10s
test / integration-firecracker (pull_request) Successful in 1m47s
test / coverage (pull_request) Successful in 2m32s
test / publish-infra (pull_request) Has been skipped
2026-07-21 02:58:25 +00:00
didericis d27f25e0ae fix(egress): the unattributed message must name the token-mismatch cause too
`/resolve` fail-closes on a missing/ambiguous registry row *and* on a
request whose identity token doesn't match. The message named only the
first, so a bottle that was registered correctly but sent no token read as
"not registered" and sent the reader looking for a deregistered bottle.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-21 02:57:29 +00:00
didericis 5eae8655f1 fix(egress): name the real fault when a deny-all is not an allowlist miss
An unattributed bottle, an unreachable orchestrator, and an unparseable
policy all become a deny-all Config, and a deny-all is indistinguishable
from "policy loaded, host not allowed" at the decision point — both are just
"no matching route". So every one of them reported `host X is not in the
bottle's egress.routes allowlist`, which reads as a config problem and sends
the operator hunting for a route that was never missing. Diagnosing a
bricked registration cost hours for exactly this reason.

Carry the structural reason on Config and prefer it in decide(). A genuine
allowlist miss — a policy that loaded and simply lacks the host — keeps the
original wording, so the message now tells the two cases apart.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-21 02:57:29 +00:00
didericis 60c33b5252 fix(orchestrator): reap registry rows whose bottle is no longer running
A registry row only ever left the registry two ways: an explicit
teardown_bottle (the launcher's cleanup callback) or the same-IP supersede
sweep in register(). Neither runs when the launching CLI dies hard, so the
row outlives its container.

That orphan is not inert. Source IPs are recycled by the backend's DHCP and
by_source_ip fail-closes on ambiguity, so a leftover row at a reused address
resolves no policy at all for the next bottle that lands there — and a
bottle with no policy denies every host, which surfaces to the agent as
"host X is not in the allowlist" for hosts that were never the problem.

Add reap_absent/reconcile and call it from the macOS launch path before
registering, so each launch self-heals the registry. Restores the invariant
the data plane needs: at most one active row per live address, and none for
a dead one. The second half matters as much as the first — when several rows
claim a *live* address the newest wins and the rest are swept, otherwise a
recycled address stays ambiguous, which is exactly the bricked state.

The host supplies the live set because the orchestrator runs inside the
infra container and cannot see the backend. A grace window exempts rows
younger than it, so reconciliation cannot race a bottle still coming up, and
a reconcile failure is logged rather than blocking an otherwise-fine launch.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-21 02:57:29 +00:00
30 changed files with 43 additions and 1813 deletions
-16
View File
@@ -75,22 +75,6 @@ On compatible macOS hosts, the default backend requires Apple's `container` CLI
Use `BOT_BOTTLE_BACKEND=docker ./cli.py start <agent>` on hosts where neither Apple Container nor KVM is available and Docker is the desired backend.
> **Experimental containers-in-bottle spike (#392):** a bottle may set
> `docker_access: true`. On the macOS backend this starts a guest-local,
> rootless **podman** service after the bottle is registered, exposing its
> Docker-compatible API socket — the agent still uses `docker` and `docker
> compose`. It does not mount Docker Desktop's socket or add outer VM
> capabilities. Rootless Docker was tried first and does not work here at
> all: Apple Container's capability bounding set omits `CAP_SYS_ADMIN`,
> which the kernel requires to write a multi-range `uid_map`. See
> [`docs/research/rootless-docker-in-apple-container-spike.md`](docs/research/rootless-docker-in-apple-container-spike.md).
>
> The tradeoff to understand before enabling it: podman avoids that
> requirement by falling back to a single-UID mapping, so nested containers
> provide **no isolation from the agent itself** — `root` inside a nested
> container is the agent user outside it. Nested containers are a build/test
> convenience, not a security boundary. The bottle remains the boundary.
### Firecracker on Linux
On Linux, a KVM-capable host defaults to the Firecracker backend. It needs:
@@ -142,19 +142,11 @@ def launch_consolidated(
def teardown_consolidated(
bottle_id: str,
*,
orchestrator_url: str,
gateway_name: str = GATEWAY_NAME,
timeout: float | None = None,
bottle_id: str, *, orchestrator_url: str, gateway_name: str = GATEWAY_NAME,
) -> None:
"""Deregister the bottle and remove its git-gate state from the gateway.
Both steps are idempotent so this is safe from a cleanup trap."""
from ...orchestrator.config_store import DEFAULT_TEARDOWN_TIMEOUT_SECONDS
OrchestratorClient(
orchestrator_url,
timeout=timeout if timeout is not None else DEFAULT_TEARDOWN_TIMEOUT_SECONDS,
).teardown_bottle(bottle_id)
OrchestratorClient(orchestrator_url).teardown_bottle(bottle_id)
deprovision_git_gate(DockerGatewayTransport(gateway_name), bottle_id)
+1 -5
View File
@@ -62,7 +62,6 @@ from .compose import (
write_compose_file,
)
from .consolidated_compose import consolidated_agent_compose
from ...orchestrator.config_store import resolve_teardown_timeout
from .consolidated_launch import launch_consolidated, teardown_consolidated
from ...orchestrator.gateway import DockerGateway
@@ -134,14 +133,11 @@ def launch(
token_values = egress_resolve_token_values(
plan.egress_plan.token_env_map, effective_env,
)
teardown_timeout = resolve_teardown_timeout()
ctx = launch_consolidated(
plan.egress_plan, git_gate_plan, image_ref=plan.image, tokens=token_values,
)
stack.callback(
teardown_consolidated, ctx.bottle_id,
orchestrator_url=ctx.orchestrator_url,
timeout=teardown_timeout,
teardown_consolidated, ctx.bottle_id, orchestrator_url=ctx.orchestrator_url,
)
# Step 4: install the SHARED gateway CA into the agent (replaces the
@@ -91,18 +91,12 @@ def launch_consolidated(
)
def teardown_consolidated(
bottle_id: str, *, orchestrator_url: str, timeout: float | None = None,
) -> None:
def teardown_consolidated(bottle_id: str, *, orchestrator_url: str) -> None:
"""Deregister the bottle and remove its git-gate state from the gateway
VM. Both steps are idempotent so this is safe from a cleanup trap. Does
NOT stop the infra VM — it's a persistent per-host singleton shared by
every bottle."""
from ...orchestrator.config_store import DEFAULT_TEARDOWN_TIMEOUT_SECONDS
OrchestratorClient(
orchestrator_url,
timeout=timeout if timeout is not None else DEFAULT_TEARDOWN_TIMEOUT_SECONDS,
).teardown_bottle(bottle_id)
OrchestratorClient(orchestrator_url).teardown_bottle(bottle_id)
deprovision_git_gate(infra_vm.gateway_transport(), bottle_id)
-3
View File
@@ -52,7 +52,6 @@ from ..util import AGENT_CA_BUNDLE, AGENT_CA_PATH
from . import firecracker_vm, image_builder, isolation_probe, netpool, util
from .bottle import FirecrackerBottle
from .bottle_plan import FirecrackerBottlePlan
from ...orchestrator.config_store import resolve_teardown_timeout
from .consolidated_launch import (
launch_consolidated,
teardown_consolidated,
@@ -113,7 +112,6 @@ def launch(
token_values = egress_resolve_token_values(
plan.egress_plan.token_env_map, effective_env,
)
teardown_timeout = resolve_teardown_timeout()
ctx = launch_consolidated(
plan.egress_plan, git_gate_plan,
guest_ip=slot.guest_ip,
@@ -123,7 +121,6 @@ def launch(
stack.callback(
teardown_consolidated, ctx.bottle_id,
orchestrator_url=ctx.orchestrator_url,
timeout=teardown_timeout,
)
# Step 5: install the SHARED gateway CA (replaces the per-bottle CA).
@@ -20,7 +20,6 @@ class MacosContainerBottlePlan(BottlePlan):
# bottle is registered. See launch.py's stamp for why it lives here and not
# only in the exec-time proxy env.
identity_token: str = ""
docker_access: bool = False
@property
def container_name(self) -> str:
@@ -41,7 +41,7 @@ from ...orchestrator.client import OrchestratorClient, OrchestratorClientError
from ...orchestrator.registration import registration_inputs
from ..docker.gateway_provision import deprovision_git_gate, provision_git_gate
from . import util as container_mod
from .enumerate import CONTAINER_NAME_PREFIX, EnumerationError, enumerate_active
from .enumerate import CONTAINER_NAME_PREFIX, enumerate_active
from .gateway import GATEWAY_NETWORK
from .gateway_provision import AppleGatewayTransport
from .infra import MacosInfraService, OrchestratorStartError
@@ -98,21 +98,14 @@ def live_source_ips(network: str) -> list[str]:
The reconciliation input: the orchestrator lives inside the infra
container and cannot enumerate the host's containers, so the host has to
tell it which bottles are actually up. Containers that have not been
assigned an address yet contribute nothing the reap's grace window, not
this list, is what protects an in-flight launch.
Raises `EnumerationError` when the live set cannot be determined
authoritatively: either the container listing fails or any individual
inspect fails. Callers must skip reconciliation in that case to avoid
unregistering healthy bottles."""
assigned an address yet contribute nothing (the read is non-fatal) the
reap's grace window, not this list, is what protects an in-flight
launch."""
ips: list[str] = []
for agent in enumerate_active():
name = f"{CONTAINER_NAME_PREFIX}{agent.slug}"
ip = container_mod.inspect_container_network_ip(name, network)
if ip is None:
raise EnumerationError(
f"container inspect {name!r} failed; live set is not authoritative"
)
ip = container_mod.try_container_ipv4_on_network(
f"{CONTAINER_NAME_PREFIX}{agent.slug}", network,
)
if ip:
ips.append(ip)
return ips
@@ -140,7 +133,7 @@ def register_agent(
# reconciliation failure must not block an otherwise-fine launch.
try:
client.reconcile(live_source_ips(endpoint.network))
except (OrchestratorClientError, EnumerationError) as e:
except OrchestratorClientError as e:
info(f"registry reconciliation skipped: {e}")
inputs = registration_inputs(egress_plan)
reg = client.register_bottle(
@@ -163,17 +156,11 @@ def register_agent(
)
def teardown_consolidated(
bottle_id: str, *, orchestrator_url: str, timeout: float | None = None,
) -> None:
def teardown_consolidated(bottle_id: str, *, orchestrator_url: str) -> None:
"""Deregister the bottle and remove its git-gate state from the gateway.
Both steps are idempotent so this is safe from a cleanup trap. Does NOT
stop the gateway it's a persistent per-host singleton."""
from ...orchestrator.config_store import DEFAULT_TEARDOWN_TIMEOUT_SECONDS
OrchestratorClient(
orchestrator_url,
timeout=timeout if timeout is not None else DEFAULT_TEARDOWN_TIMEOUT_SECONDS,
).teardown_bottle(bottle_id)
OrchestratorClient(orchestrator_url).teardown_bottle(bottle_id)
deprovision_git_gate(AppleGatewayTransport(), bottle_id)
@@ -9,9 +9,8 @@ from .. import ActiveAgent
from .infra import INFRA_NAME
# The name every agent container carries: `bot-bottle-<slug>`. Exported
# because callers that act on a running bottle (gateway-host rewrites,
# registry reconciliation) have to map an enumerated slug back to a
# container name.
# because reconciliation has to map an enumerated slug back to a container
# name to read its address.
CONTAINER_NAME_PREFIX = "bot-bottle-"
# The shared per-host infra container carries the same prefix as agent
# containers but is infrastructure, not a bottle — one control plane + gateway
@@ -19,10 +18,6 @@ CONTAINER_NAME_PREFIX = "bot-bottle-"
_INFRA_NAMES = frozenset({INFRA_NAME})
class EnumerationError(RuntimeError):
"""container list failed; the resulting live set is not authoritative."""
def enumerate_active() -> list[ActiveAgent]:
result = subprocess.run(
["container", "list", "--quiet"],
@@ -31,10 +26,7 @@ def enumerate_active() -> list[ActiveAgent]:
check=False,
)
if result.returncode != 0:
raise EnumerationError(
f"container list failed: "
f"{(result.stderr or '').strip() or '<no stderr>'}"
)
return []
out: list[ActiveAgent] = []
for name in sorted(line.strip() for line in result.stdout.splitlines()):
if not name.startswith(CONTAINER_NAME_PREFIX) or name in _INFRA_NAMES:
@@ -1,96 +0,0 @@
"""Stable gateway name for macOS agents, via each bottle's `/etc/hosts`.
The shared gateway's address is assigned by vmnet's DHCP and changes whenever
the infra container is recreated a source-hash bump, an image upgrade, a
crash. Every agent-facing URL (egress proxy, git-http, supervise) embeds that
address, and the proxy URL reaches the agent as **process environment** at
`container exec` time. A running process's `environ` cannot be rewritten from
outside, so a moved gateway used to strand every running bottle permanently:
not degraded, unreachable, until the bottle was relaunched and its session
thrown away.
So the agent never learns the address. It is given a stable *name*
(`GATEWAY_HOSTNAME`) in every URL, resolved through its own `/etc/hosts`.
Unlike `environ`, that is a file it can be rewritten inside a container that
is already running, so a gateway that comes back at a new address is picked up
by live bottles instead of orphaning them.
Apple Container 1.0 offers no container-name DNS on a user network (the only
nameserver an agent sees is vmnet's, which does not know container names) and
`container run` has no `--add-host`, so the entry is written by exec after the
container starts.
Writing it needs root, and the agent runs as `node`: the agent therefore
cannot repoint its own gateway name, while the host (which drives `container
exec --user root`) can. That asymmetry is deliberate keep it.
"""
from __future__ import annotations
from ...log import warn
from . import util as container_mod
from .enumerate import CONTAINER_NAME_PREFIX, enumerate_active
# The name every agent-facing gateway URL uses. Must not collide with a real
# DNS name the agent might resolve; it is bottle-local by construction.
GATEWAY_HOSTNAME = "bot-bottle-gateway"
# Marker so the rewrite is idempotent and only ever touches our own line —
# the rest of /etc/hosts (localhost, the container's own name) is preserved.
_MARKER = "# bot-bottle gateway"
def _rewrite_script(gateway_ip: str) -> str:
"""A shell one-liner that replaces our managed line in `/etc/hosts`.
Rewrites in place via a temp file + `cat` rather than `mv`, so the file
keeps its original inode, ownership, and mode a bind-mounted or
pre-created `/etc/hosts` must not be replaced by a root-owned 0644 copy
that the runtime then refuses to update.
"""
return (
"set -e; "
f"grep -v '{_MARKER}' /etc/hosts > /tmp/.bb-hosts || true; "
f"printf '%s %s %s\\n' '{gateway_ip}' '{GATEWAY_HOSTNAME}' "
f"'{_MARKER}' >> /tmp/.bb-hosts; "
"cat /tmp/.bb-hosts > /etc/hosts; "
"rm -f /tmp/.bb-hosts"
)
def set_gateway_host(container_name: str, gateway_ip: str) -> None:
"""Point `GATEWAY_HOSTNAME` at `gateway_ip` inside one running container.
Must run before the agent is exec'd: the agent's proxy URL names the
gateway, so the entry has to exist for its first connection. Idempotent
re-running with the same address is a no-op in effect.
"""
container_mod.exec_container_as_root(
container_name, ["sh", "-c", _rewrite_script(gateway_ip)],
)
def refresh_gateway_host(gateway_ip: str) -> list[str]:
"""Re-point every running bottle at the current gateway address.
Called once the shared gateway is known to be up, so a bottle stranded by
an earlier gateway restart re-attaches instead of needing a relaunch.
Returns the containers updated.
Best-effort per bottle: one container that refuses the write (already
exiting, say) must not stop the others from being repaired, and must not
fail the launch that triggered the sweep.
"""
updated: list[str] = []
for agent in enumerate_active():
name = f"{CONTAINER_NAME_PREFIX}{agent.slug}"
try:
set_gateway_host(name, gateway_ip)
updated.append(name)
# One bad bottle must not stop the sweep, so this is deliberately broad.
except Exception as e: # noqa: BLE001 # pylint: disable=broad-exception-caught
warn(f"could not re-point {name} at the gateway: {e}")
return updated
__all__ = ["GATEWAY_HOSTNAME", "set_gateway_host", "refresh_gateway_host"]
+16 -64
View File
@@ -59,14 +59,7 @@ from ..docker.egress import EGRESS_PORT
from ..util import AGENT_CA_BUNDLE, AGENT_CA_PATH
from . import util as container_mod
from .bottle import MacosContainerBottle
from .gateway_hosts import (
GATEWAY_HOSTNAME,
refresh_gateway_host,
set_gateway_host,
)
from . import rootless_podman
from .bottle_plan import MacosContainerBottlePlan
from ...orchestrator.config_store import resolve_teardown_timeout
from .consolidated_launch import (
GatewayEndpoint,
ensure_gateway,
@@ -107,11 +100,6 @@ def launch(
# Step 1: the per-host singletons. Must precede the agent run — its
# proxy env needs the gateway's address at `container run` time.
endpoint = ensure_gateway()
# The gateway's address may have changed since these bottles launched
# (any infra recreate re-runs DHCP). They name the gateway rather than
# address it, so re-pointing /etc/hosts re-attaches them in place
# instead of leaving them stranded until relaunch.
refresh_gateway_host(endpoint.gateway_ip)
# Step 2: mint this bottle's deploy keys, then point it at the SHARED
# gateway's CA + git-http/supervise ports.
@@ -129,9 +117,6 @@ def launch(
# attribution key; `--cap-drop CAP_NET_RAW` at run is what makes it
# unforgeable. Poll: `container run --detach` can return before vmnet's
# DHCP has assigned the address.
# Resolve the gateway name before anything execs: every agent-facing
# URL uses it, so the entry must exist for the first connection.
set_gateway_host(plan.container_name, endpoint.gateway_ip)
source_ip = container_mod.wait_container_ipv4_on_network(
plan.container_name, endpoint.network,
)
@@ -144,7 +129,6 @@ def launch(
token_values = egress_resolve_token_values(
plan.egress_plan.token_env_map, effective_env,
)
teardown_timeout = resolve_teardown_timeout()
ctx = register_agent(
plan.egress_plan,
plan.git_gate_plan,
@@ -156,7 +140,6 @@ def launch(
stack.callback(
teardown_consolidated, ctx.bottle_id,
orchestrator_url=ctx.orchestrator_url,
timeout=teardown_timeout,
)
info(
f"agent {plan.container_name} registered "
@@ -172,10 +155,6 @@ def launch(
# token above, so — unlike the run-time env — the plan CAN carry it.
plan = dataclasses.replace(plan, identity_token=ctx.identity_token)
exec_env = {
**_identity_proxy_env(endpoint, ctx.identity_token),
**rootless_podman.guest_env(plan.docker_access),
}
bottle = MacosContainerBottle(
plan.container_name,
teardown,
@@ -189,16 +168,10 @@ def launch(
),
terminal_color=plan.spec.color,
agent_workdir=plan.workspace_plan.workdir,
exec_env=exec_env,
exec_env=_identity_proxy_env(endpoint, ctx.identity_token),
)
bottle.prompt_path = provision(plan, bottle)
if plan.docker_access:
rootless_podman.prepare_guest_devices(
plan.container_name, container_mod.exec_container_as_root,
)
rootless_podman.start(bottle)
yield bottle
finally:
teardown()
@@ -210,22 +183,15 @@ def _build_images(plan: MacosContainerBottlePlan) -> MacosContainerBottlePlan:
committed = read_committed_image(plan.slug)
if committed and container_mod.image_exists(committed):
info(f"using committed image {committed!r}")
plan = dataclasses.replace(
return dataclasses.replace(
plan,
agent_provision=dataclasses.replace(
plan.agent_provision, image=committed,
),
)
else:
container_mod.build_image(
plan.image, _REPO_DIR, dockerfile=plan.dockerfile_path,
)
if plan.docker_access:
image = rootless_podman.build_image(plan.image, container_mod.build_image)
plan = dataclasses.replace(
plan,
agent_provision=dataclasses.replace(plan.agent_provision, image=image),
)
container_mod.build_image(
plan.image, _REPO_DIR, dockerfile=plan.dockerfile_path,
)
return plan
@@ -265,20 +231,13 @@ def _stamp_agent_urls(
) -> MacosContainerBottlePlan:
"""Point the agent's git-gate insteadOf rewrites + supervise MCP at the
shared gateway's ports. Both bypass the egress proxy (NO_PROXY covers the
gateway name).
Addressed by `GATEWAY_HOSTNAME`, never by IP: these URLs are baked into
the agent's gitconfig and MCP config at provision time, so an address here
would strand the bottle the moment the gateway moved. The name is resolved
per connection through `/etc/hosts`, which stays rewritable while the
bottle runs."""
del endpoint # addressed by name; the address reaches the bottle via /etc/hosts
gateway address)."""
git_gate_url = (
f"http://{GATEWAY_HOSTNAME}:{_GIT_HTTP_PORT}"
f"http://{endpoint.gateway_ip}:{_GIT_HTTP_PORT}"
if plan.git_gate_plan.upstreams else ""
)
supervise_url = (
f"http://{GATEWAY_HOSTNAME}:{SUPERVISE_PORT}/"
f"http://{endpoint.gateway_ip}:{SUPERVISE_PORT}/"
if plan.supervise_plan is not None else ""
)
return dataclasses.replace(
@@ -288,25 +247,19 @@ def _stamp_agent_urls(
)
def _proxy_url(identity_token: str = "") -> str:
def _proxy_url(gateway_ip: str, identity_token: str = "") -> str:
"""The agent's egress proxy URL. The identity token rides as proxy
credentials the gateway reads Proxy-Authorization, resolves the
(source_ip, token) pair against the control plane, and strips it before
upstream. Without a valid pair `/resolve` denies the request (#366).
Names the gateway rather than addressing it: this URL reaches the agent as
process environment, which cannot be rewritten once the agent is running,
so an address baked here is unfixable if the gateway moves."""
upstream. Without a valid pair `/resolve` denies the request (#366)."""
cred = f"bottle:{identity_token}@" if identity_token else ""
return f"http://{cred}{GATEWAY_HOSTNAME}:{EGRESS_PORT}"
return f"http://{cred}{gateway_ip}:{EGRESS_PORT}"
def _no_proxy() -> str:
def _no_proxy(gateway_ip: str) -> str:
# git-http + supervise live on the gateway and must NOT go through the
# egress proxy — the agent reaches them directly by name. Deliberately
# address-free: NO_PROXY is baked into the run-time env and is therefore
# just as unfixable as the proxy URL if the gateway moves.
return f"localhost,127.0.0.1,{GATEWAY_HOSTNAME}"
# egress proxy — the agent reaches them directly by its address.
return f"localhost,127.0.0.1,{gateway_ip}"
def _identity_proxy_env(
@@ -323,8 +276,7 @@ def _identity_proxy_env(
`_agent_env_entries`."""
if not identity_token:
return {}
del endpoint # the gateway is named, not addressed
url = _proxy_url(identity_token)
url = _proxy_url(endpoint.gateway_ip, identity_token)
return {
"HTTPS_PROXY": url, "HTTP_PROXY": url,
"https_proxy": url, "http_proxy": url,
@@ -386,7 +338,7 @@ def _agent_env_entries(
# silently dropping attribution for. Without it a process that egresses
# before the exec-time env still fails closed — the agent network is
# host-only, so there is no route off it except the gateway.
no_proxy = _no_proxy()
no_proxy = _no_proxy(endpoint.gateway_ip)
env = [
f"NO_PROXY={no_proxy}",
f"no_proxy={no_proxy}",
@@ -44,5 +44,4 @@ def resolve_plan(
egress_plan=egress_plan,
supervise_plan=supervise_plan,
agent_provision=agent_provision_plan,
docker_access=manifest.bottle.docker_access,
)
@@ -1,89 +0,0 @@
#!/bin/sh
set -eu
uid="$(id -u)"
if [ "$uid" -eq 0 ]; then
echo "refusing to run rootless podman as root" >&2
exit 1
fi
for command in podman docker fuse-overlayfs slirp4netns; do
command -v "$command" >/dev/null 2>&1 || {
echo "missing rootless podman prerequisite: $command" >&2
exit 1
}
done
# The inverse of the rootless-Docker check, and the whole point of the podman
# variant: a subordinate range would push podman onto newuidmap, which cannot
# write a multi-range uid_map without CAP_SYS_ADMIN in this guest. An empty
# range keeps it on the single-UID self-mapping an unprivileged process may
# write itself.
if grep -q "^$(id -un):" /etc/subuid 2>/dev/null; then
echo "unexpected subordinate UID range for $(id -un): podman would" >&2
echo "require CAP_SYS_ADMIN via newuidmap in this guest" >&2
exit 1
fi
for device in /dev/fuse /dev/net/tun; do
[ -r "$device" ] && [ -w "$device" ] || {
echo "device $device is not readable/writable by $(id -un)" >&2
exit 1
}
done
export XDG_RUNTIME_DIR="${XDG_RUNTIME_DIR:-/tmp/bot-bottle-podman-run}"
config="$HOME/.config/containers"
mkdir -p "$XDG_RUNTIME_DIR" "$config"
chmod 700 "$XDG_RUNTIME_DIR"
# ignore_chown_errors is required, not incidental: with a single-UID mapping
# there is no second UID for image layers to be chowned to, so layers that
# record other owners would otherwise fail to extract.
cat > "$config/storage.conf" <<'CONF'
[storage]
driver="overlay"
[storage.options.overlay]
mount_program="/usr/bin/fuse-overlayfs"
ignore_chown_errors="true"
CONF
# No cgroup delegation reaches this guest, so asking podman to manage cgroups
# fails; events_logger=file avoids the journald socket that is equally absent.
cat > "$config/containers.conf" <<'CONF'
[containers]
cgroups="disabled"
[engine]
cgroup_manager="cgroupfs"
events_logger="file"
CONF
# Registry pulls egress through the bottle's proxy like everything else. The
# token-bearing proxy URL is already in the agent's environment; persisting it
# inside this disposable VM does not broaden its authority.
python3 - <<'PY'
import json
import os
from pathlib import Path
proxy = os.environ.get("HTTPS_PROXY") or os.environ.get("https_proxy", "")
no_proxy = os.environ.get("NO_PROXY") or os.environ.get("no_proxy", "")
config = {"proxies": {"default": {
"httpProxy": proxy,
"httpsProxy": proxy,
"noProxy": no_proxy,
}}}
path = Path.home() / ".docker" / "config.json"
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(config), encoding="utf-8")
path.chmod(0o600)
PY
if docker info >/dev/null 2>&1; then
exit 0
fi
log=/tmp/bot-bottle-rootless-podman.log
nohup podman system service --time=0 \
"unix://$XDG_RUNTIME_DIR/podman.sock" \
>"$log" 2>&1 </dev/null &
@@ -1,132 +0,0 @@
"""Experimental rootless podman bootstrap for Apple-container bottles.
The service and every nested container remain inside the existing per-bottle
VM. This module refuses to compensate for missing prerequisites with outer
capabilities, a privileged container, or a host Docker socket.
Podman is used rather than rootless Docker for one specific reason: Apple
Container's capability bounding set omits `CAP_SYS_ADMIN`, which the kernel
requires to write a multi-range `uid_map` via `newuidmap`. Rootless Docker
has no path that avoids that write. Podman does with no subordinate UID
range configured it falls back to a single-UID self-mapping, which an
unprivileged process may write itself. See
`docs/research/rootless-docker-in-apple-container-spike.md`.
That fallback is why `build_image` *removes* the agent user's `/etc/subuid`
and `/etc/subgid` entries instead of adding them: their presence is precisely
what would send podman down the `newuidmap` path that cannot work here.
The agent still talks to `docker` and `docker compose`; those speak to
podman's Docker-compatible API socket, so nothing in the agent's habits
changes.
"""
from __future__ import annotations
import shlex
import shutil
import tempfile
import time
from pathlib import Path
from typing import Callable
from ...log import die, info
_INIT = "/usr/local/libexec/bot-bottle/rootless-podman-init"
_RUNTIME_DIR = "/tmp/bot-bottle-podman-run"
_SOCKET = f"{_RUNTIME_DIR}/podman.sock"
_LOG = "/tmp/bot-bottle-rootless-podman.log"
READY_RETRIES = 30
# Apple Container creates both device nodes 0600 root:root, so the agent user
# cannot open them: /dev/fuse blocks the fuse-overlayfs storage driver and
# /dev/net/tun blocks slirp4netns, which rootless podman uses for the default
# bridge network that stock compose files expect. Relaxing the modes needs no
# capability the bottle does not already hold — unlike CAP_SYS_ADMIN, which is
# what killed the rootless-Docker approach.
_GUEST_DEVICES = ("/dev/fuse", "/dev/net/tun")
def build_image(
base_image: str,
build: Callable[..., None],
) -> str:
"""Layer spike-only tooling on an already-built provider image."""
image = f"{base_image}-rootless-podman"
init_script = Path(__file__).with_name("rootless-podman-init.sh")
with tempfile.TemporaryDirectory(prefix="bot-bottle-rootless-podman.") as tmp:
context = Path(tmp)
shutil.copy2(init_script, context / "rootless-podman-init.sh")
(context / "Dockerfile").write_text(
"FROM docker:28-cli AS docker_cli\n"
f"FROM {base_image}\n"
"USER root\n"
"COPY --from=docker_cli /usr/local/bin/docker /usr/local/bin/docker\n"
"COPY --from=docker_cli /usr/local/libexec/docker/cli-plugins/"
"docker-compose /usr/local/libexec/docker/cli-plugins/docker-compose\n"
"RUN apt-get update \\\n"
" && apt-get install -y --no-install-recommends podman "
"fuse-overlayfs slirp4netns uidmap \\\n"
" && rm -rf /var/lib/apt/lists/* \\\n"
# Deliberate: an empty subordinate range keeps podman on the
# single-UID mapping that needs no CAP_SYS_ADMIN. Adding ranges
# here would reintroduce the newuidmap failure this spike exists
# to route around.
" && sed -i '/^node:/d' /etc/subuid /etc/subgid\n"
"COPY rootless-podman-init.sh "
"/usr/local/libexec/bot-bottle/rootless-podman-init\n"
"RUN chmod 0755 /usr/local/libexec/bot-bottle/rootless-podman-init\n"
"USER node\n",
encoding="utf-8",
)
build(image, str(context), dockerfile=str(context / "Dockerfile"))
return image
def guest_env(enabled: bool) -> dict[str, str]:
"""Environment consumed by the Docker CLI inside an enabled bottle."""
if not enabled:
return {}
return {
"DOCKER_HOST": f"unix://{_SOCKET}",
"XDG_RUNTIME_DIR": _RUNTIME_DIR,
}
def prepare_guest_devices(container_name: str, exec_as_root: Callable[..., None]) -> None:
"""Make /dev/fuse and /dev/net/tun openable by the agent user.
Runs as root inside the bottle because the agent must not be able to
re-mode device nodes itself. No outer capability is involved.
"""
exec_as_root(
container_name,
["sh", "-c", f"chmod 0666 {' '.join(_GUEST_DEVICES)}"],
)
def start(bottle: object) -> None:
"""Start and verify the unprivileged service through the bottle exec API."""
info("starting experimental rootless podman service")
result = bottle.exec(shlex.quote(_INIT)) # type: ignore[attr-defined]
if result.returncode != 0:
detail = (result.stderr or result.stdout or "").strip()
die(f"rootless podman bootstrap failed: {detail or '<no output>'}")
for _ in range(READY_RETRIES):
result = bottle.exec("docker info >/dev/null 2>&1") # type: ignore[attr-defined]
if result.returncode == 0:
info("rootless podman service is ready")
return
time.sleep(0.2)
logs = bottle.exec( # type: ignore[attr-defined]
f"tail -n 80 {_LOG} 2>/dev/null || true"
)
die(
"rootless podman did not become ready without additional outer "
f"privileges:\n{(logs.stdout or logs.stderr or '<no log>').strip()}"
)
__all__ = ["build_image", "guest_env", "prepare_guest_devices", "start"]
@@ -360,21 +360,6 @@ def exec_container(name: str, argv: list[str]) -> None:
)
def exec_container_as_root(name: str, argv: list[str]) -> None:
"""`exec_container`, but as uid 0 inside the container.
For host-driven maintenance the agent itself must not be able to perform
rewriting `/etc/hosts` to point the gateway name at an address. The agent
runs as `node`, so it cannot repoint its own gateway; the host can.
"""
result = _run_container_op([_CONTAINER, "exec", "--user", "root", name, *argv])
if result.returncode != 0:
die(
f"container exec (root) in {name} failed: "
f"{(result.stderr or '').strip() or '<no stderr>'}"
)
def _run_container_op(cmd: list[str]) -> subprocess.CompletedProcess[str]:
result = subprocess.run(
cmd,
@@ -572,41 +557,6 @@ def try_container_ipv4_on_network(name: str, network: str) -> str:
return ""
def inspect_container_network_ip(name: str, network: str) -> str | None:
"""IP of `name` on `network`, distinguishing inspect failure from "not yet".
Returns:
- the IP string when the container has one on `network`
- "" when inspect succeeds but no address is assigned yet (in-flight DHCP)
- None when the inspect command itself fails (authoritative list impossible)
"""
result = subprocess.run(
[_CONTAINER, "inspect", name],
capture_output=True, text=True, check=False,
)
if result.returncode != 0:
return None
try:
data = json.loads(result.stdout or "[]")
except json.JSONDecodeError:
return None
if isinstance(data, list):
data = data[0] if data else {}
if not isinstance(data, dict):
return None
status = data.get("status")
networks = status.get("networks") if isinstance(status, dict) else None
if not isinstance(networks, list):
return ""
for entry in networks:
if not isinstance(entry, dict) or entry.get("network") != network:
continue
raw = entry.get("ipv4Address")
if isinstance(raw, str) and raw:
return raw.split("/", 1)[0]
return ""
def wait_container_ipv4_on_network(
name: str, network: str, *, timeout: float = 15.0, poll: float = 0.25,
) -> str:
-11
View File
@@ -44,9 +44,6 @@ class ManifestBottle:
# daemon that exposes egress MCP tools to the agent. Set
# `supervise: false` to skip the gateway.
supervise: bool = True
# Experimental guest-local container engine (issue #392). Backends must
# implement this without granting access to a host/shared daemon.
docker_access: bool = False
@classmethod
def from_dict(cls, name: str, raw: object) -> "ManifestBottle":
@@ -126,15 +123,7 @@ class ManifestBottle:
f"(was {type(supervise_raw).__name__})"
)
docker_access_raw = d.get("docker_access", False)
if not isinstance(docker_access_raw, bool):
raise ManifestError(
f"bottle '{name}' docker_access must be a boolean "
f"(was {type(docker_access_raw).__name__})"
)
return cls(
env=env, agent_provider=agent_provider, git=git,
git_user=git_user, egress=egress, supervise=supervise_raw,
docker_access=docker_access_raw,
)
-8
View File
@@ -54,7 +54,6 @@ def _merge_two_bottles_runtime(base: "ManifestBottle", override: "ManifestBottle
git_user=merged_git_user,
egress=merged_egress,
supervise=override.supervise,
docker_access=override.docker_access,
)
@@ -207,7 +206,6 @@ def _fold_two_bottles(
git_user=merged_git_user,
egress=merged_egress,
supervise=later.supervise,
docker_access=later.docker_access,
), merged_repos_raw
@@ -268,11 +266,6 @@ def _merge_bottles(
merged_supervise = (
child.supervise if "supervise" in child_raw else parent.supervise
)
merged_docker_access = (
child.docker_access
if "docker_access" in child_raw
else parent.docker_access
)
validate_egress_routes(name, merged_egress.routes)
return ManifestBottle(
@@ -282,7 +275,6 @@ def _merge_bottles(
git_user=merged_git_user,
egress=merged_egress,
supervise=merged_supervise,
docker_access=merged_docker_access,
)
+1 -4
View File
@@ -16,10 +16,7 @@ _FILENAME_RX = re.compile(r"^[a-z][a-z0-9-]*$")
# sets dies with a "did you mean" pointer: typos should not silently
# ghost into an empty config.
BOTTLE_KEYS = frozenset(
{
"env", "extends", "agent_provider", "git-gate", "egress", "supervise",
"docker_access",
}
{"env", "extends", "agent_provider", "git-gate", "egress", "supervise"}
)
AGENT_KEYS_REQUIRED: frozenset[str] = frozenset()
AGENT_KEYS_OPTIONAL = frozenset({"bottle", "skills", "git-gate"})
-107
View File
@@ -1,107 +0,0 @@
"""Per-host orchestrator configuration store (settings in bot-bottle.db).
Co-tenants the shared `bot-bottle.db` via the `DbStore` framework. Settings
are readable by the host launch path directly (no HTTP round-trip to the
orchestrator), so they take effect even before the orchestrator is reachable.
"""
from __future__ import annotations
import os
import sqlite3
from pathlib import Path
from ..db_store import DbStore
from ..migrations import TableMigrations
from ..paths import host_db_path
TEARDOWN_TIMEOUT_ENV = "BOT_BOTTLE_ORCHESTRATOR_TEARDOWN_TIMEOUT_SECONDS"
DEFAULT_TEARDOWN_TIMEOUT_SECONDS = 30.0
_MIGRATIONS = TableMigrations(
"orchestrator_config",
[
"""
CREATE TABLE IF NOT EXISTS orchestrator_config (
id INTEGER PRIMARY KEY CHECK (id = 1),
teardown_timeout_seconds REAL
)
""",
],
)
class OrchestratorConfigStore(DbStore):
"""Orchestrator settings in the shared host DB."""
def __init__(self, db_path: Path | None = None) -> None:
super().__init__(db_path or host_db_path(), _MIGRATIONS)
def _connect(self) -> sqlite3.Connection:
conn = super()._connect()
conn.execute("PRAGMA busy_timeout=5000")
return conn
def get_teardown_timeout_seconds(self) -> float | None:
"""Return the configured teardown timeout, or None if not set."""
try:
with self._connection() as conn:
row = conn.execute(
"SELECT teardown_timeout_seconds FROM orchestrator_config WHERE id = 1"
).fetchone()
except sqlite3.OperationalError:
return None
return row["teardown_timeout_seconds"] if row else None
def set_teardown_timeout_seconds(self, value: float) -> None:
"""Persist the teardown timeout."""
with self._connection() as conn:
conn.execute(
"INSERT OR REPLACE INTO orchestrator_config"
" (id, teardown_timeout_seconds) VALUES (1, ?)",
(value,),
)
self._chmod()
def delete_teardown_timeout_seconds(self) -> bool:
"""Clear the stored teardown timeout. Returns True if a value existed."""
with self._connection() as conn:
cur = conn.execute(
"UPDATE orchestrator_config SET teardown_timeout_seconds = NULL"
" WHERE id = 1 AND teardown_timeout_seconds IS NOT NULL"
)
return cur.rowcount > 0
def resolve_teardown_timeout(db_path: Path | None = None) -> float:
"""Return the teardown timeout to use, in priority order:
1. ``BOT_BOTTLE_ORCHESTRATOR_TEARDOWN_TIMEOUT_SECONDS`` env var
2. ``teardown_timeout_seconds`` in the orchestrator config DB
3. ``DEFAULT_TEARDOWN_TIMEOUT_SECONDS`` (30 s)
"""
raw = os.environ.get(TEARDOWN_TIMEOUT_ENV, "").strip()
if raw:
try:
value = float(raw)
if value > 0:
return value
except ValueError:
pass
store = OrchestratorConfigStore(db_path)
if not store.is_migrated():
store.migrate()
db_value = store.get_teardown_timeout_seconds()
if db_value is not None and db_value > 0:
return db_value
return DEFAULT_TEARDOWN_TIMEOUT_SECONDS
__all__ = [
"OrchestratorConfigStore",
"resolve_teardown_timeout",
"TEARDOWN_TIMEOUT_ENV",
"DEFAULT_TEARDOWN_TIMEOUT_SECONDS",
]
@@ -1,86 +0,0 @@
# Egress proxy OOMs on large downloads
Found on 2026-07-21 while running the rootless-podman spike
(`docs/research/rootless-docker-in-apple-container-spike.md`). Recorded
rather than fixed — the fix is a security-relevant decision, not a
mechanical patch.
## Summary
A single large HTTPS download through the gateway kills the egress
proxy. `mitmdump` buffers whole response bodies so the DLP detectors can
scan them, grows past the gateway container's memory limit, and is
OOM-killed by the cgroup. Nothing restarts it.
Two properties make this worse than a failed download:
- **The gateway is a per-host singleton.** Every bottle shares it, so
one bottle's download takes egress away from all of them.
- **There is no restart on death.** The gateway supervisor is
`while : ; do wait ; done`; a killed daemon stays dead until the infra
container is recreated.
So ordinary agent activity — pulling a container image, downloading a
model or dataset, fetching a large tarball — is a denial of service
against every other bottle on the host. No malice required, though it is
trivially reachable on purpose.
## Evidence
Triggered by `docker compose up` pulling `quay.io/fedora/python-312`
(two layers, ~82MB and ~83MB) inside a bottle. The pull itself
succeeded; the *next* request failed:
```
initializing source docker://quay.io/fedora/python-312:latest:
pinging container registry quay.io: Get "https://quay.io/v2/":
proxyconnect tcp: dial tcp 192.168.128.39:9099: connect: connection refused
```
From the gateway's `dmesg`:
```
python3 invoked oom-killer: gfp_mask=0x100cca(GFP_HIGHUSER_MOVABLE), order=0
oom-kill:constraint=CONSTRAINT_MEMCG,
oom_memcg=/container/bot-bottle-mac-infra,
task_memcg=/container/bot-bottle-mac-infra,task=mitmdump,pid=118
Memory cgroup out of memory: Killed process 118 (mitmdump)
total-vm:1391936kB, anon-rss:997768kB
```
~1GB RSS against a 1024MB container. Note the amplification: ~165MB of
layers produced ~1GB of resident memory, so the buffering is several
copies deep (encoded body, decoded body, and the text conversion the
regex detectors scan).
Afterwards the gateway container was still running and healthy-looking —
orchestrator, supervise, and git-http all alive — with no `mitmdump`
process at all, and it stayed that way until the container was
recreated. A liveness check on the container would not have caught this.
## Reproduction
1. Launch any bottle with an egress route to a host serving a large file.
2. Download >~150MB over HTTPS through the proxy.
3. `dmesg | grep -i oom` inside `bot-bottle-mac-infra`, and note that no
`mitmdump` process remains.
Beware a false negative when checking: truncating the process listing
(`cut -c1-45`) cuts before the binary name, because `mitmdump` runs as
`/usr/local/bin/python3.12 /usr/local/bin/mitmdump …`.
## Fix options, not yet chosen
1. **Restart dead daemons.** Smallest change and strictly an
improvement: an OOM then degrades one download instead of removing
egress for every bottle. Does not stop the OOM.
2. **Cap the scanned body size.** Above a threshold, stop buffering —
either skip the scan or stream it. This is the root-cause fix and a
security decision: a size threshold is exactly the hole an exfiltrator
would aim for, so "skip above N" trades a DoS for a covert channel.
Streaming with a bounded window keeps coverage, at more complexity.
3. **Raise the gateway's memory limit.** Moves the threshold; does not
remove it.
Worth noting that (1) and (2) are complementary — the restart gap is
worth closing regardless of how the memory behaviour is resolved.
@@ -1,353 +0,0 @@
# Rootless Docker inside Apple Container bottles
Spike branch: `spike/rootless-docker-macos` (`a4d8461`)
## Summary
**Negative result.** Rootless Docker cannot run inside an Apple
Container bottle without granting the bottle `CAP_SYS_ADMIN`. This is a
kernel constraint on writing multi-range `uid_map`, not a packaging gap
we can close with a better init script, a different base image, or more
careful `/etc/subuid` handling.
The spike was built on the premise — stated in
`bot_bottle/backend/macos_container/rootless_docker.py` — that it would
*"deliberately refuse to compensate for missing prerequisites with outer
capabilities, a privileged container, or a host Docker socket."* That
premise is exactly what the experiment falsified. The two ways forward
are to abandon the premise (add `CAP_SYS_ADMIN` to the bottle, and with
it most of the isolation the bottle exists to provide) or to abandon
rootless Docker.
Recommendation: abandon rootless Docker. Podman does not have this
problem — see [Podman is not blocked by
this](#podman-is-not-blocked-by-this) below.
## Local environment
Tested on 2026-07-21:
```console
$ sw_vers
ProductName: macOS
ProductVersion: 26.5.1
BuildVersion: 25F80
$ container --version
container CLI version 1.0.0 (build: release, commit: ee848e3)
$ uname -a # inside the bottle
Linux ... 6.18.15 #1 SMP Tue Mar 17 01:36:53 UTC 2026 aarch64 GNU/Linux
```
## The failure
`tests/integration/test_macos_rootless_docker_spike.py` builds the
image, launches the bottle, and dies in `rootless_docker.start`:
```
+ exec rootlesskit --net=slirp4netns --mtu=65520 ... dockerd-rootless.sh
[rootlesskit:parent] error: failed to setup UID/GID map:
newuidmap 1100 [0 1000 1 1 100000 65536] failed:
newuidmap: write to uid_map failed: Operation not permitted
```
## Why it fails
Every prerequisite you would normally suspect is present and correct in
the guest:
| Check | Result |
| --- | --- |
| `/usr/bin/newuidmap` | `-rwsr-xr-x root root` — setuid bit intact, survived the OCI export |
| `/` mount options | `rw,relatime`**not** `nosuid` |
| `NoNewPrivs` | `0` |
| `Seccomp` | `0`, no filters |
| `/etc/subuid`, `/etc/subgid` | `node:100000:65536` in both |
| user namespace | `user:[4026531837]`, identical to pid 1 — the *initial* userns |
| `unshare -U -r true` | succeeds |
| `/proc/sys/user/max_user_namespaces` | `4505` |
The one thing that is missing is in the capability bounding set that
Apple Container gives the container:
```
CapBnd: 00000000a80425fb
= chown, dac_override, fowner, fsetid, kill, setgid, setuid, setpcap,
net_bind_service, net_raw, sys_chroot, mknod, audit_write, setfcap
```
No `CAP_SYS_ADMIN`. That is the whole story, and the chain is:
1. The kernel's `map_write()` gates writing a `uid_map` on
`file_ns_capable(file, ns, CAP_SYS_ADMIN)` — capability over the
**new** user namespace, evaluated against the credentials that opened
`/proc/<pid>/uid_map`.
2. `newuidmap` is setuid-root, so it runs with euid 0 — but its
capability sets are clamped by the bounding set, which has no
`CAP_SYS_ADMIN`.
3. `cap_capable()` has a shortcut that grants *all* capabilities when
the caller's userns is the new namespace's parent **and**
`ns->owner == cred->euid`. It does not apply: the namespace was
created by `node` (uid 1000) while `newuidmap` runs as euid 0.
4. So the check falls through to the effective-set test in the initial
userns, which fails. `EPERM`.
Note that the single-line unprivileged path (`unshare -U -r`) works
precisely because it does not go through `newuidmap` and does not need
`CAP_SYS_ADMIN`. Only the multi-range subuid mapping that rootless
Docker requires does.
This is the same constraint that makes upstream's `dind-rootless` image
require `--privileged`. It is not specific to Apple Container, except
that Apple Container gives us no bounding set that includes
`CAP_SYS_ADMIN` by default.
## It does work with the capability — which is the point
Adding the capability clears the failure immediately, and exposes one
further, much smaller blocker: `/dev/net/tun` exists (the kernel has
tun; `/proc/misc` lists `200 tun`) but Apple Container creates it
`crw------- root root`, so uid 1000 cannot open it and `slirp4netns`
fails with `open: Permission denied`. A `chmod 0666 /dev/net/tun` as
root inside the bottle fixes that, and needs no capability beyond what
the bottle already has.
With both applied by hand, the daemon comes up completely:
```console
$ container run --rm -u root --cap-add CAP_SYS_ADMIN \
bot-bottle-claude:latest-rootless-docker sh -c '...'
Server Version: 20.10.24+dfsg1
Storage Driver: fuse-overlayfs
Cgroup Driver: none
Cgroup Version: 2
API listen on /tmp/rt/docker.sock
```
So `rootless-docker-init.sh` and `rootless_docker.py` are *correct*.
The spike did not fail on a bug. It failed on its own premise.
Two secondary findings from that run, relevant if anyone revisits this:
- Debian's `docker.io` package pins Docker **20.10** (EOL), not the 28.x
implied by the `docker:28-cli` compose plugin the image copies in.
- `Cgroup Driver: none` — no resource limits on nested containers.
## Why we should not just add the capability
`CAP_SYS_ADMIN` is close to a superset of "root" in practical terms —
mount, `pivot_root`, namespace manipulation, and a long tail of
subsystem-specific powers. Granting it to the agent bottle would
undercut the containment argument the rest of the backend is built
around, including the deliberately narrow choices immediately adjacent
to it in `launch.py` (`--cap-drop CAP_NET_RAW`, no `NET_ADMIN`, a
host-only agent network). Trading all of that for nested `docker
compose` is a bad exchange.
## Podman is not blocked by this
Sanity-checked on the same host, same kernel, same runtime, so the
comparison is apples to apples:
| Scenario | Result |
| --- | --- |
| Podman rootless, `/etc/subuid` populated | **Fails identically**`newuidmap: write to uid_map failed: Operation not permitted` |
| Podman rootless, no subuid ranges, `--network=host` | **Works**, no added capabilities |
| Podman rootless, no subuid ranges, default netns, `/dev/net/tun` at `0600` | Fails — `slirp4netns: open("/dev/net/tun"): Permission denied` |
| Podman rootless, no subuid ranges, default netns, `/dev/net/tun` at `0666` | **Works**, no added capabilities |
The difference is that podman degrades gracefully when no subuid range
is available: it falls back to a single-UID self-mapping, which an
unprivileged process may write itself, so `newuidmap` is never invoked
and `CAP_SYS_ADMIN` is never needed. Docker's rootless mode has no
equivalent fallback.
The cost of that fallback is real and should be weighed before building
on it: with a single-UID mapping, every UID inside a nested container
collapses onto the bottle's own uid 1000. There is no UID separation
between the agent and anything it runs — `root` in a nested container is
the agent user outside it. It also requires `ignore_chown_errors` on the
storage driver. Whether that is acceptable depends on whether the bottle
boundary (which is unchanged) or the nested-container boundary (which is
effectively nil) is the one we are relying on.
## What the podman spike then needed
The podman implementation that replaced the Docker one on this branch
turned up two more device-node blockers of the same shape as
`/dev/net/tun` — Apple Container creates the node, but 0600 root:root:
- **`/dev/fuse`** — blocks the `fuse-overlayfs` storage driver
(`fuse: failed to open /dev/fuse: Permission denied`). Without it the
only working driver is `vfs`, which copies whole layers per container.
- **`/dev/net/tun`** — blocks `slirp4netns`, which rootless podman uses
for the default bridge network.
Both are fixed by `chmod 0666` as root inside the bottle, which needs no
capability the bottle does not already hold. This is categorically
different from the `CAP_SYS_ADMIN` requirement: it is a permission on a
node that already exists, not an outer privilege grant.
One design note worth recording: the agent-facing surface stays `docker`
and `docker compose`, pointed at podman's Docker-compatible API socket
via `DOCKER_HOST`. Setting `netns="host"` in `containers.conf` does *not*
propagate through that compat API — stock `docker run` and compose files
request bridge networking explicitly — so slirp4netns (and therefore the
`/dev/net/tun` chmod) is required for ordinary compose files to work at
all. Host networking remains available per-workload via
`--network=host`.
Verified working in a bottle with zero added capabilities: fuse-overlayfs
storage, the compat API socket, `docker run` on both bridge and host
networking, and published ports.
### Nested pulls collide with our own egress DLP
The first live run got podman up and `docker compose` running, then
failed on the image pull:
```
web Pulling
initializing source docker://python:3.12-alpine: reading manifest ...
StatusCode: 403, egress DLP: Generic Bearer JWT found in body
```
This is bot-bottle's own egress scanner, not a podman problem. The
Docker registry auth flow carries a bearer JWT *by protocol*, and the
`token_patterns` detector's `Generic Bearer JWT` rule
(`Bearer\s+[A-Za-z0-9._\-]{50,}`) matches it on every pull. Any bottle
that pulls images will hit this.
The fix is per-route detector scoping, which the egress config already
supports — drop `token_patterns` on the registry hosts and keep
`known_secrets`:
```json
{"host": "registry-1.docker.io",
"dlp": {"outbound_detectors": ["known_secrets"]}}
```
That is the right trade rather than a grudging one: `known_secrets`
matches the bottle's *actual* credential values, so real exfil through a
registry host is still caught. `token_patterns` on a registry route only
ever produces protocol noise.
Worth generalising later: any manifest enabling `docker_access` needs
this on its registry routes, so it probably belongs in a shared
registry-route snippet rather than being copy-pasted per bottle.
### And then registry auth collides with the Authorization strip
With DLP scoped, the pull failed differently: `unauthorized:
authentication required`. This one is architectural.
`egress_addon.py` strips agent-set `Authorization` unconditionally
before forwarding — deliberately, so an agent cannot smuggle a
credential out in a header the DLP detectors don't recognise. A route
may carry gateway-injected auth instead, but only from a *static* token
in an env var (`auth_scheme` + `token_env`).
Docker registry auth doesn't fit that shape. The client fetches a
short-lived, per-repository-scope bearer token from `auth.docker.io` and
presents it to `registry-1.docker.io`. There is no static token to
inject, and the token the client legitimately obtained is stripped.
Measured inside a bottle, by hand:
| Step | Result |
| --- | --- |
| Fetch token from `auth.docker.io` | 200, 5409-byte token body |
| Manifest request **with** that valid token | 401 |
| Manifest request with **no** Authorization | 401 — identical |
A valid token behaves exactly like sending none, which is direct
evidence the header never arrives. Any nested-container workflow that
pulls from a registry is blocked on this, so it is not a detail that can
be deferred: pulling base images is most of what nested containers are
for.
### Registries that skip the token dance work today
Not every registry needs the stripped header. Measured directly:
| Registry | Manifest request with no `Authorization` |
| --- | --- |
| `quay.io` | 200 |
| `mcr.microsoft.com` | 200 |
| `registry.k8s.io` | 307 (redirect, no auth) |
| `ghcr.io` | 401 |
| `registry-1.docker.io` | 401 |
So "just add the registry to the bottle config" genuinely works — for
quay, MCR, registry.k8s.io, or any unauthenticated internal registry.
Docker Hub and GHCR are the ones that need the strip resolved. The
acceptance test uses quay for exactly this reason.
Resolving it for Docker Hub means picking one of:
1. **Per-route opt-in to preserve client Authorization.** Smallest
change. Note the compounding effect on exactly these routes: the DLP
scoping above already removed `token_patterns` there, so a
preserved-auth registry route is one where the agent may send bearer
tokens that neither the strip nor the pattern detector inspects.
`known_secrets` still applies, so the bottle's real credentials are
still caught.
2. **A registry-aware gateway** that performs the token dance itself and
injects the result. Preserves the invariant fully; materially more
work, and it makes the gateway speak a specific registry protocol.
3. **Pre-seed images at provision time** (host-side `container image
save` into podman storage), so bottles never pull at runtime.
Preserves the invariant, and limits nested containers to
pre-approved images — which fits the custody positioning, at the cost
of no ad-hoc `docker pull`.
4. **Stop.** Nested containers are not supported on this backend.
### Podman 4.3.1 silently swallows container exit codes
Debian bookworm — which the current agent base image is built on —
ships podman 4.3.1. Through its Docker-compatible API, `docker run`
returns 0 no matter what the container did:
| Command | podman 4.3.1 | podman 5.4.2 |
| --- | --- | --- |
| `docker run … sh -c 'exit 7'` (compat API) | **0** | 7 |
| `docker run … sh -c 'exit 0'` (compat API) | 0 | 0 |
| `podman run … sh -c 'exit 7'` (native) | 7 | 7 |
This is worse than a broken feature: every failing command an agent runs
via `docker run` reports success. A test suite, a build step, or a CI
script inside a bottle would pass while failing. It also silently
defeated the acceptance test's egress-containment assertion, which is
why that assertion now checks an in-band marker rather than an exit
code.
Podman 5.4.2 (Debian trixie) fixes it, but needs two packages that
bookworm's podman does not: `passt` (podman 5's default network tool)
and `nftables` (netavark shells out to `nft`; without it every run fails
with `unable to upgrade to tcp, received 500`). With both installed,
exit codes propagate correctly and the compat API behaves.
The open question this leaves is where podman 5 comes from, since the
agent base is bookworm-based:
1. **Move the agent images to Debian trixie.** Trixie is current stable.
Correct, and the blast radius is every bottle, not just this feature.
2. **Drop the compat socket and use podman natively** (`podman-docker`
provides a `docker` shim; compose comes from `podman-compose`).
Native podman propagates exit codes correctly even on 4.3.1. Contained
to this feature, at the cost of `docker compose` becoming
`docker-compose`/`podman-compose`.
3. **Ship bookworm's podman 4.3.1 with the compat socket** — not viable.
Silent false success is a correctness bug agents cannot see.
## Recommendation
1. Do not revive rootless Docker on this backend. This document is the
record of why.
2. Nested containers, if wanted, come from podman under the
single-mapping constraint — with the explicit understanding that the
nested-container boundary carries no security weight. `root` in a
nested container is the agent user outside it.
3. Nested containers are therefore a build/test convenience. The bottle
remains the security boundary, exactly as it was.
@@ -1,154 +0,0 @@
"""Live-Mac acceptance spike for guest-local rootless podman (issue #392).
Run explicitly on an Apple Silicon/macOS 26 host:
BOT_BOTTLE_ROOTLESS_PODMAN_SPIKE=1 \
python3 -m unittest tests.integration.test_macos_rootless_podman_spike -v
The opt-in is deliberate: ordinary Linux CI cannot execute Apple Container.
Podman rather than Docker because Apple Container's capability bounding set
omits CAP_SYS_ADMIN; see
docs/research/rootless-docker-in-apple-container-spike.md. The agent-facing
surface is still `docker` and `docker compose`, which talk to podman's
Docker-compatible API socket.
"""
from __future__ import annotations
import os
import platform
import shutil
import tempfile
import unittest
from pathlib import Path
from bot_bottle.backend import BottleSpec, get_bottle_backend
from bot_bottle.manifest import ManifestIndex
@unittest.skipUnless(
platform.system() == "Darwin"
and os.environ.get("BOT_BOTTLE_ROOTLESS_PODMAN_SPIKE") == "1",
"requires an explicit live-Mac rootless-podman spike run",
)
class TestMacosRootlessPodmanSpike(unittest.TestCase):
def test_compose_stays_inside_registered_bottle(self) -> None:
workspace = Path(tempfile.mkdtemp(prefix="rootless-podman-spike."))
stage = Path(tempfile.mkdtemp(prefix="rootless-podman-stage."))
try:
(workspace / "index.html").write_text("bottle-compose-ok\n")
(workspace / "compose.yaml").write_text(
"services:\n"
" web:\n"
" image: quay.io/prometheus/busybox\n"
" working_dir: /workspace\n"
" command: httpd -f -p 8000 -h /workspace\n"
" volumes: ['.:/workspace']\n"
" ports: ['18080:8000']\n",
encoding="utf-8",
)
manifest = ManifestIndex.from_json_obj({
"bottles": {"dev": {
"docker_access": True,
# A deliberately tiny image. Pulling a ~165MB one
# OOM-kills the shared egress proxy, which buffers whole
# response bodies to scan them — a real defect, but a
# separate one from what this test covers. See the
# research note.
#
# quay.io deliberately, not Docker Hub: the egress proxy
# strips agent-set Authorization (so an agent cannot
# smuggle a credential out in a header), and Docker Hub
# requires a client-fetched, per-scope bearer token that
# the strip therefore removes. quay serves manifests with
# no Authorization at all, so a plain route is enough.
#
# token_patterns is still scoped off: registry traffic
# carries bearer JWTs by protocol and trips the generic
# rule. known_secrets stays on — it matches the bottle's
# own credentials, which is the detector that catches
# real exfil.
"egress": {"routes": [
{"host": "quay.io", "dlp": {
"outbound_detectors": ["known_secrets"],
}},
{"host": "cdn01.quay.io", "dlp": {
"outbound_detectors": ["known_secrets"],
}},
]},
}},
"agents": {"spike": {
"bottle": "dev", "skills": [], "prompt": "",
}},
})
spec = BottleSpec(
manifest=manifest,
agent_name="spike",
copy_cwd=True,
user_cwd=str(workspace),
)
backend = get_bottle_backend("macos-container")
plan = backend.prepare(spec, stage_dir=stage)
with backend.launch(plan) as bottle:
workdir = plan.workspace_plan.workdir
checks = (
"docker info >/dev/null && docker compose version && "
f"cd {workdir} && docker compose up -d --wait && "
"curl --fail --silent http://127.0.0.1:18080/ | "
"grep -q bottle-compose-ok"
)
result = bottle.exec(checks)
self.assertEqual(
0, result.returncode,
f"stdout={result.stdout!r}\nstderr={result.stderr!r}",
)
# podman's compat API reports rootlessness through its own
# native endpoint; the Docker-shaped SecurityOptions field does
# not carry it.
inspect = bottle.exec(
"podman info --format '{{.Host.Security.Rootless}}'"
)
self.assertIn("true", inspect.stdout.lower())
self.assertEqual(
0,
bottle.exec(
"test \"$(id -u)\" -ne 0"
).returncode,
"the podman service must not be running as bottle root",
)
self.assertNotEqual(
0,
bottle.exec("test -S /var/run/docker.sock").returncode,
"spike must never expose a host/rootful Docker socket",
)
# Asserted on an in-band marker, not on `docker run`'s exit
# code: podman 4.3.1's Docker-compat API swallows the
# container's status and returns 0 for everything, so an
# exit-code assertion here passes whether egress was blocked
# or wide open. That silent false pass is worse than no check
# at all, and it is exactly this check — the one proving a
# nested container cannot escape the egress path.
#
# busybox ships wget, so a failure here means egress was
# refused rather than the binary being absent.
direct = bottle.exec(
"docker run --rm --env HTTP_PROXY= --env HTTPS_PROXY= "
"--env http_proxy= --env https_proxy= "
"quay.io/prometheus/busybox sh -c "
"'wget -T 4 -qO- https://evil.example.com/ "
"&& echo ESCAPED || echo CONTAINED'"
)
self.assertIn(
"CONTAINED", direct.stdout,
"an inner container obtained direct, unproxied egress: "
f"stdout={direct.stdout!r} stderr={direct.stderr!r}",
)
self.assertNotIn("ESCAPED", direct.stdout)
finally:
shutil.rmtree(workspace, ignore_errors=True)
shutil.rmtree(stage, ignore_errors=True)
if __name__ == "__main__":
unittest.main()
+2 -37
View File
@@ -147,7 +147,7 @@ class TestLiveSourceIps(unittest.TestCase):
def test_maps_slugs_to_container_addresses(self) -> None:
from bot_bottle.backend.macos_container.consolidated_launch import live_source_ips
with patch(f"{_MOD}.enumerate_active", return_value=self._agents("a", "b")), \
patch(f"{_MOD}.container_mod.inspect_container_network_ip",
patch(f"{_MOD}.container_mod.try_container_ipv4_on_network",
side_effect=["10.0.0.1", "10.0.0.2"]) as ip:
got = live_source_ips("net0")
self.assertEqual(["10.0.0.1", "10.0.0.2"], got)
@@ -158,30 +158,10 @@ class TestLiveSourceIps(unittest.TestCase):
nothing the reap's grace window, not this list, protects it."""
from bot_bottle.backend.macos_container.consolidated_launch import live_source_ips
with patch(f"{_MOD}.enumerate_active", return_value=self._agents("a", "b")), \
patch(f"{_MOD}.container_mod.inspect_container_network_ip",
patch(f"{_MOD}.container_mod.try_container_ipv4_on_network",
side_effect=["", "10.0.0.2"]):
self.assertEqual(["10.0.0.2"], live_source_ips("net0"))
def test_container_list_failure_raises(self) -> None:
"""If container list fails, the live set is not authoritative and
reconciliation must be skipped."""
from bot_bottle.backend.macos_container.consolidated_launch import live_source_ips
from bot_bottle.backend.macos_container.enumerate import EnumerationError
with patch(f"{_MOD}.enumerate_active",
side_effect=EnumerationError("container list failed")):
with self.assertRaises(EnumerationError):
live_source_ips("net0")
def test_per_container_inspect_failure_raises(self) -> None:
"""If any individual inspect fails, the live set is not authoritative."""
from bot_bottle.backend.macos_container.consolidated_launch import live_source_ips
from bot_bottle.backend.macos_container.enumerate import EnumerationError
with patch(f"{_MOD}.enumerate_active", return_value=self._agents("a", "b")), \
patch(f"{_MOD}.container_mod.inspect_container_network_ip",
side_effect=["10.0.0.1", None]):
with self.assertRaises(EnumerationError):
live_source_ips("net0")
class TestRegisterAgentReconciles(unittest.TestCase):
"""Registration self-heals the registry first: an orphan row at a recycled
@@ -221,18 +201,3 @@ class TestRegisterAgentReconciles(unittest.TestCase):
client.reconcile.side_effect = OrchestratorClientError("unreachable")
self._register(client)
client.register_bottle.assert_called_once()
def test_enumeration_error_does_not_block_the_launch(self) -> None:
"""A partial container listing must not abort the launch — skip
reconciliation and proceed, just as with an unreachable orchestrator."""
from bot_bottle.backend.macos_container.enumerate import EnumerationError
client = _client()
with patch(f"{_MOD}.OrchestratorClient", return_value=client), \
patch(f"{_MOD}.provision_git_gate"), \
patch(f"{_MOD}.live_source_ips",
side_effect=EnumerationError("container list failed")):
register_agent(
_egress_plan(), _git_plan(),
source_ip="10.0.0.7", endpoint=_endpoint(),
)
client.register_bottle.assert_called_once()
+2 -4
View File
@@ -66,10 +66,8 @@ class TestMacosContainerEnumerate(unittest.TestCase):
agents = self._enumerate("bot-bottle-mac-infra\nbot-bottle-dev-abc\n")
self.assertEqual(["dev-abc"], [a.slug for a in agents])
def test_raises_when_the_cli_fails(self):
from bot_bottle.backend.macos_container.enumerate import EnumerationError
with self.assertRaises(EnumerationError):
self._enumerate("", returncode=1)
def test_empty_when_the_cli_fails(self):
self.assertEqual([], self._enumerate("", returncode=1))
if __name__ == "__main__":
@@ -17,12 +17,10 @@ from unittest.mock import patch
from bot_bottle.backend.macos_container.bottle import MacosContainerBottle
from bot_bottle.backend.macos_container.bottle_plan import MacosContainerBottlePlan
from bot_bottle.backend.macos_container.consolidated_launch import GatewayEndpoint
from bot_bottle.backend.macos_container.gateway_hosts import GATEWAY_HOSTNAME
from bot_bottle.backend.macos_container.launch import (
_agent_run_argv,
_identity_proxy_env,
)
from bot_bottle.backend.macos_container.rootless_podman import guest_env
from bot_bottle.manifest import ManifestIndex
_BOTTLE = "bot_bottle.backend.macos_container.bottle"
@@ -77,7 +75,6 @@ def _plan(
),
agent_git_gate_url=agent_git_gate_url,
agent_supervise_url=agent_supervise_url,
docker_access=False,
))
@@ -131,13 +128,7 @@ class TestAgentRunArgv(unittest.TestCase):
"""git-http + supervise live on the gateway and must be reached
directly, not through its own egress proxy."""
entry = next(a for a in self.argv if a.startswith("NO_PROXY="))
self.assertIn(GATEWAY_HOSTNAME, entry)
def test_no_proxy_names_the_gateway_and_never_addresses_it(self) -> None:
"""NO_PROXY is baked into the run-time env, so an address here is as
unfixable as the proxy URL if the gateway moves."""
entry = next(a for a in self.argv if a.startswith("NO_PROXY="))
self.assertNotIn("192.168.128.3", entry)
self.assertIn("192.168.128.3", entry)
def test_forwarded_secrets_stay_off_argv(self) -> None:
"""Bare name → inherited from the run process env, so the value never
@@ -155,7 +146,7 @@ class TestIdentityTokenDelivery(unittest.TestCase):
def test_exec_env_carries_the_token_as_proxy_credentials(self) -> None:
env = _identity_proxy_env(_endpoint(), "s3cret")
self.assertEqual(
f"http://bottle:s3cret@{GATEWAY_HOSTNAME}:9099", env["HTTP_PROXY"],
"http://bottle:s3cret@192.168.128.3:9099", env["HTTP_PROXY"],
)
self.assertEqual(env["HTTP_PROXY"], env["https_proxy"])
@@ -180,18 +171,6 @@ class TestIdentityTokenDelivery(unittest.TestCase):
self.assertNotIn("--env", argv)
class TestRootlessPodmanEnvironment(unittest.TestCase):
def test_disabled_bottle_gets_no_docker_environment(self) -> None:
self.assertEqual({}, guest_env(False))
def test_enabled_bottle_uses_only_guest_local_socket(self) -> None:
env = guest_env(True)
self.assertEqual(
"unix:///tmp/bot-bottle-podman-run/podman.sock", env["DOCKER_HOST"],
)
self.assertNotIn("/var/run/docker.sock", " ".join(env.values()))
class TestPlanIdentityToken(unittest.TestCase):
"""git-gate's gitconfig extraHeader and the supervise MCP --header read
`getattr(plan, "identity_token", "")` at provision time and both bypass the
@@ -235,7 +214,7 @@ class TestPlanIdentityToken(unittest.TestCase):
self.assertIn("HTTP_PROXY", argv)
self.assertNotIn("s3cret", " ".join(argv))
self.assertEqual(
f"http://bottle:s3cret@{GATEWAY_HOSTNAME}:9099", kwargs["env"]["HTTP_PROXY"],
"http://bottle:s3cret@192.168.128.3:9099", kwargs["env"]["HTTP_PROXY"],
)
-68
View File
@@ -272,30 +272,6 @@ resolver #2
),
)
def test_exec_container_as_root_selects_root_user(self):
completed = util.subprocess.CompletedProcess(
args=[], returncode=0, stdout="", stderr="",
)
with patch.object(util, "_run_container_op", return_value=completed) as run:
util.exec_container_as_root("bot-bottle-demo", ["true"])
run.assert_called_once_with([
"container", "exec", "--user", "root", "bot-bottle-demo", "true",
])
def test_exec_container_as_root_reports_failure(self):
failed = util.subprocess.CompletedProcess(
args=[], returncode=1, stdout="", stderr="permission denied\n",
)
with patch.object(util, "_run_container_op", return_value=failed), \
patch.object(util, "die", side_effect=SystemExit("die")) as die:
with self.assertRaises(SystemExit):
util.exec_container_as_root("bot-bottle-demo", ["true"])
die.assert_called_once_with(
"container exec (root) in bot-bottle-demo failed: permission denied",
)
def _completed(stdout: str, returncode: int = 0):
return util.subprocess.CompletedProcess(args=[], returncode=returncode, stdout=stdout, stderr="")
@@ -334,50 +310,6 @@ class TestInspectDigests(unittest.TestCase):
self.assertEqual({}, util.container_env("x"))
class TestInspectContainerNetworkIp(unittest.TestCase):
"""inspect_container_network_ip must distinguish inspect failure (None)
from 'no DHCP address yet' (""), which is the invariant live_source_ips
relies on to skip reconciliation on partial snapshots."""
_NETWORK = "bot-bottle-mac-gateway"
def _inspect(self, stdout: str, returncode: int = 0) -> str | None:
cp = util.subprocess.CompletedProcess(
args=[], returncode=returncode, stdout=stdout, stderr="",
)
with patch.object(util.subprocess, "run", return_value=cp):
return util.inspect_container_network_ip("bot-bottle-abc", self._NETWORK)
def _entry(self, ip: str = "192.168.128.5") -> str:
return (
f'[{{"status":{{"networks":['
f'{{"network":"{self._NETWORK}","ipv4Address":"{ip}"}}'
f']}}}}]'
)
def test_returns_ip_when_inspect_succeeds(self) -> None:
self.assertEqual("192.168.128.5", self._inspect(self._entry()))
def test_strips_cidr_prefix(self) -> None:
self.assertEqual("192.168.128.5", self._inspect(self._entry("192.168.128.5/24")))
def test_returns_empty_string_when_no_address_assigned_yet(self) -> None:
no_ip = f'[{{"status":{{"networks":[{{"network":"{self._NETWORK}","ipv4Address":""}}]}}}}]'
self.assertEqual("", self._inspect(no_ip))
def test_returns_empty_string_when_network_list_absent(self) -> None:
self.assertEqual("", self._inspect('[{"status":{}}]'))
def test_returns_none_on_nonzero_exit(self) -> None:
self.assertIsNone(self._inspect("", returncode=1))
def test_returns_none_on_malformed_json(self) -> None:
self.assertIsNone(self._inspect("not-json"))
def test_returns_none_on_unexpected_json_shape(self) -> None:
self.assertIsNone(self._inspect("null"))
class TestWaitContainerIpv4(unittest.TestCase):
def test_returns_address_once_dhcp_assigns_it(self):
with patch.object(util, "try_container_ipv4_on_network", side_effect=["", "", "192.168.128.4"]), \
-140
View File
@@ -1,140 +0,0 @@
"""Unit: stable gateway name via each bottle's /etc/hosts (issue #443).
The gateway's address moves whenever the infra container is recreated. Agents
name it instead of addressing it, and the name resolves through `/etc/hosts`
a file, so it stays rewritable while the bottle runs, unlike the `environ` the
proxy URL is delivered in.
"""
from __future__ import annotations
import unittest
from types import SimpleNamespace
from unittest.mock import patch
from bot_bottle.backend.macos_container.gateway_hosts import (
GATEWAY_HOSTNAME,
refresh_gateway_host,
set_gateway_host,
)
_MOD = "bot_bottle.backend.macos_container.gateway_hosts"
class TestSetGatewayHost(unittest.TestCase):
def _script(self, exec_root: object) -> str:
argv = exec_root.call_args.args[1] # type: ignore[attr-defined]
self.assertEqual(["sh", "-c"], argv[:2])
return argv[2]
def test_writes_the_address_against_the_stable_name(self) -> None:
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
set_gateway_host("bot-bottle-demo", "192.168.128.19")
self.assertEqual("bot-bottle-demo", ex.call_args.args[0])
script = self._script(ex)
self.assertIn("192.168.128.19", script)
self.assertIn(GATEWAY_HOSTNAME, script)
def test_runs_as_root_so_the_agent_cannot_repoint_itself(self) -> None:
"""The agent runs as `node`. If it could rewrite /etc/hosts it could
aim its own gateway name elsewhere, so the write must go through the
root-only helper."""
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
set_gateway_host("bot-bottle-demo", "10.0.0.1")
ex.assert_called_once()
def test_is_idempotent_by_removing_its_own_line_first(self) -> None:
"""Re-pointing must replace the managed entry, not append a second one
two entries for the same name would resolve by luck of ordering."""
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
set_gateway_host("bot-bottle-demo", "10.0.0.1")
self.assertIn("grep -v", self._script(ex))
def test_preserves_the_rest_of_the_hosts_file(self) -> None:
"""localhost and the container's own name must survive the rewrite."""
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
set_gateway_host("bot-bottle-demo", "10.0.0.1")
script = self._script(ex)
# Filter-and-append, never a truncating write of just our line.
self.assertIn("/etc/hosts >", script)
self.assertIn(">> /tmp/.bb-hosts", script)
def test_keeps_the_original_inode(self) -> None:
"""`cat >` rather than `mv`: a pre-created /etc/hosts must keep its
ownership and mode, not be replaced by a root-owned copy."""
script = None
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
set_gateway_host("bot-bottle-demo", "10.0.0.1")
script = self._script(ex)
self.assertIn("cat /tmp/.bb-hosts > /etc/hosts", script)
self.assertNotIn("mv ", script)
class TestRefreshGatewayHost(unittest.TestCase):
"""The re-attach sweep: bottles stranded by an earlier gateway restart get
re-pointed in place instead of needing a relaunch."""
def _agents(self, *slugs: str) -> list[SimpleNamespace]:
return [SimpleNamespace(slug=s) for s in slugs]
def test_repoints_every_running_bottle(self) -> None:
with patch(f"{_MOD}.enumerate_active", return_value=self._agents("a", "b")), \
patch(f"{_MOD}.set_gateway_host") as setter:
updated = refresh_gateway_host("192.168.128.19")
self.assertEqual(["bot-bottle-a", "bot-bottle-b"], updated)
self.assertEqual(
[("bot-bottle-a", "192.168.128.19"), ("bot-bottle-b", "192.168.128.19")],
[c.args for c in setter.call_args_list],
)
def test_one_failing_bottle_does_not_stop_the_sweep(self) -> None:
"""A container that is already exiting must not block the repair of
its neighbours, nor fail the launch that triggered the sweep."""
def _flaky(name: str, _ip: str) -> None:
if name == "bot-bottle-a":
raise RuntimeError("container is exiting")
with patch(f"{_MOD}.enumerate_active", return_value=self._agents("a", "b")), \
patch(f"{_MOD}.set_gateway_host", side_effect=_flaky), \
patch(f"{_MOD}.warn") as warn:
updated = refresh_gateway_host("10.0.0.1")
self.assertEqual(["bot-bottle-b"], updated)
warn.assert_called_once()
def test_no_running_bottles_is_a_clean_no_op(self) -> None:
with patch(f"{_MOD}.enumerate_active", return_value=[]), \
patch(f"{_MOD}.set_gateway_host") as setter:
self.assertEqual([], refresh_gateway_host("10.0.0.1"))
setter.assert_not_called()
class TestLaunchWiring(unittest.TestCase):
"""Ordering matters: the name must resolve before anything execs, and the
stranded-bottle sweep must run once the gateway is known to be up."""
def test_launch_sets_the_host_entry_before_reading_the_source_ip(self) -> None:
"""The agent's every URL names the gateway, so the entry has to exist
before the first connection which means before the agent execs."""
import inspect
from bot_bottle.backend.macos_container import launch
src = inspect.getsource(launch)
set_at = src.index("set_gateway_host(plan.container_name")
exec_at = src.index("wait_container_ipv4_on_network")
self.assertLess(set_at, exec_at)
def test_launch_refreshes_stranded_bottles_after_ensure_gateway(self) -> None:
import inspect
from bot_bottle.backend.macos_container import launch
src = inspect.getsource(launch)
ensure_at = src.index("endpoint = ensure_gateway()")
refresh_at = src.index("refresh_gateway_host(endpoint.gateway_ip)")
self.assertLess(ensure_at, refresh_at)
if __name__ == "__main__":
unittest.main()
-147
View File
@@ -1,147 +0,0 @@
"""Unit coverage for the fail-closed macOS rootless-podman spike."""
from __future__ import annotations
import unittest
from dataclasses import dataclass
from pathlib import Path
from types import SimpleNamespace
from typing import cast
from unittest.mock import patch
from bot_bottle.backend.macos_container import rootless_podman
from bot_bottle.backend.macos_container import launch as launch_mod
from bot_bottle.backend.macos_container.bottle_plan import MacosContainerBottlePlan
class _Bottle:
def __init__(self, results: list[SimpleNamespace]) -> None:
self.results = results
self.commands: list[str] = []
def exec(self, command: str) -> SimpleNamespace:
self.commands.append(command)
return self.results.pop(0)
def _result(returncode: int, *, stdout: str = "", stderr: str = "") -> SimpleNamespace:
return SimpleNamespace(returncode=returncode, stdout=stdout, stderr=stderr)
@dataclass(frozen=True)
class _AgentProvision:
image: str
@dataclass(frozen=True)
class _Plan:
slug: str
image: str
dockerfile_path: str
docker_access: bool
agent_provision: _AgentProvision
class TestRootlessPodmanStart(unittest.TestCase):
def test_bootstraps_then_waits_for_guest_local_service(self) -> None:
bottle = _Bottle([_result(0), _result(1), _result(0)])
with patch.object(rootless_podman.time, "sleep"):
rootless_podman.start(bottle)
self.assertIn("rootless-podman-init", bottle.commands[0])
self.assertEqual(2, bottle.commands.count("docker info >/dev/null 2>&1"))
def test_bootstrap_failure_is_fatal_without_privilege_fallback(self) -> None:
bottle = _Bottle([_result(1, stderr="slirp4netns missing")])
with patch.object(rootless_podman, "die", side_effect=RuntimeError) as die:
with self.assertRaises(RuntimeError):
rootless_podman.start(bottle)
self.assertIn("slirp4netns missing", die.call_args.args[0])
self.assertEqual(1, len(bottle.commands))
def test_timeout_reports_guest_log(self) -> None:
bottle = _Bottle(
[_result(0)]
+ [_result(1) for _ in range(rootless_podman.READY_RETRIES)]
+ [_result(0, stdout="operation not permitted")]
)
with patch.object(rootless_podman.time, "sleep"), \
patch.object(rootless_podman, "die", side_effect=RuntimeError) as die:
with self.assertRaises(RuntimeError):
rootless_podman.start(bottle)
self.assertIn("operation not permitted", die.call_args.args[0])
class TestRootlessPodmanDevices(unittest.TestCase):
def test_relaxes_only_the_two_blocked_device_nodes_as_root(self) -> None:
calls: list[tuple[str, list[str]]] = []
rootless_podman.prepare_guest_devices(
"bottle-1", lambda name, argv: calls.append((name, argv)),
)
self.assertEqual(1, len(calls))
name, argv = calls[0]
self.assertEqual("bottle-1", name)
self.assertIn("chmod 0666 /dev/fuse /dev/net/tun", argv[-1])
class TestRootlessPodmanImage(unittest.TestCase):
def test_layers_tooling_without_changing_base_image(self) -> None:
calls: list[tuple[str, str, str]] = []
def build(image: str, context: str, *, dockerfile: str) -> None:
calls.append((image, context, dockerfile))
text = Path(dockerfile).read_text(encoding="utf-8")
self.assertIn("FROM agent:base", text)
self.assertIn("podman fuse-overlayfs slirp4netns uidmap", text)
self.assertIn("USER node", text)
self.assertTrue((Path(context) / "rootless-podman-init.sh").is_file())
image = rootless_podman.build_image("agent:base", build)
self.assertEqual("agent:base-rootless-podman", image)
self.assertEqual("agent:base-rootless-podman", calls[0][0])
def test_strips_subordinate_ranges_so_podman_avoids_newuidmap(self) -> None:
"""The single-UID fallback is the entire reason podman works here.
A subordinate range would send podman down the newuidmap path, which
cannot write a multi-range uid_map without CAP_SYS_ADMIN in an Apple
Container guest the failure that killed the rootless-Docker spike.
"""
seen: list[str] = []
def build(image: str, context: str, *, dockerfile: str) -> None:
seen.append(Path(dockerfile).read_text(encoding="utf-8"))
rootless_podman.build_image("agent:base", build)
text = seen[0]
self.assertIn("sed -i '/^node:/d' /etc/subuid /etc/subgid", text)
self.assertNotIn("subuid", text.replace(
"sed -i '/^node:/d' /etc/subuid /etc/subgid", "",
))
def test_launch_builds_base_then_rootless_variant(self) -> None:
plan = cast(MacosContainerBottlePlan, cast(object, _Plan(
slug="dev-abc",
image="agent:base",
dockerfile_path="/repo/Dockerfile",
docker_access=True,
agent_provision=_AgentProvision(image="agent:base"),
)))
with patch.object(launch_mod, "read_committed_image", return_value=None), \
patch.object(launch_mod.container_mod, "build_image") as build, \
patch.object(
launch_mod.rootless_podman,
"build_image",
return_value="agent:base-rootless-podman",
) as build_rootless:
result = launch_mod._build_images(plan) # pylint: disable=protected-access
build.assert_called_once_with(
"agent:base", launch_mod._REPO_DIR, # pylint: disable=protected-access
dockerfile="/repo/Dockerfile",
)
build_rootless.assert_called_once_with("agent:base", build)
self.assertEqual("agent:base-rootless-podman", result.agent_provision.image)
if __name__ == "__main__":
unittest.main()
-6
View File
@@ -56,12 +56,6 @@ class TestMergeBottlesRuntime(unittest.TestCase):
result = merge_bottles_runtime([base, override])
self.assertFalse(result.supervise)
def test_docker_access_later_wins(self):
result = merge_bottles_runtime([
_bottle(docker_access=False), _bottle(docker_access=True),
])
self.assertTrue(result.docker_access)
def test_three_bottles_merged_left_to_right(self):
b1 = _bottle(env={"A": "1", "B": "1", "C": "1"})
b2 = _bottle(env={"B": "2", "C": "2"})
+1 -8
View File
@@ -44,20 +44,13 @@ class TestBottleValidation(unittest.TestCase):
with self.assertRaises(ManifestError):
ManifestBottle.from_dict("b", {"supervise": "yes"})
def test_docker_access_not_bool(self) -> None:
with self.assertRaises(ManifestError):
ManifestBottle.from_dict("b", {"docker_access": "yes"})
def test_removed_runtime_field(self) -> None:
with self.assertRaises(ManifestError):
ManifestBottle.from_dict("b", {"runtime": "runsc"})
def test_valid_minimal(self) -> None:
b = ManifestBottle.from_dict(
"b", {"supervise": False, "docker_access": True, "env": {"X": "1"}},
)
b = ManifestBottle.from_dict("b", {"supervise": False, "env": {"X": "1"}})
self.assertFalse(b.supervise)
self.assertTrue(b.docker_access)
self.assertEqual({"X": "1"}, dict(b.env))
@@ -1,147 +0,0 @@
"""Unit: OrchestratorConfigStore and resolve_teardown_timeout.
Also verifies the lifecycle ordering invariant: resolve_teardown_timeout()
must be called before launch_consolidated() / register_agent() so that a
resolver failure cannot leave an orphaned registration with no teardown
callback.
"""
from __future__ import annotations
import inspect
import os
import tempfile
import unittest
from pathlib import Path
from types import ModuleType
from bot_bottle.orchestrator.config_store import (
DEFAULT_TEARDOWN_TIMEOUT_SECONDS,
TEARDOWN_TIMEOUT_ENV,
OrchestratorConfigStore,
resolve_teardown_timeout,
)
class TestOrchestratorConfigStore(unittest.TestCase):
def setUp(self) -> None:
self._tmp = tempfile.TemporaryDirectory()
self.db = Path(self._tmp.name) / "test.db"
self.store = OrchestratorConfigStore(self.db)
self.store.migrate()
def tearDown(self) -> None:
self._tmp.cleanup()
def test_get_returns_none_when_not_set(self) -> None:
self.assertIsNone(self.store.get_teardown_timeout_seconds())
def test_set_and_get_roundtrip(self) -> None:
self.store.set_teardown_timeout_seconds(42.5)
self.assertEqual(42.5, self.store.get_teardown_timeout_seconds())
def test_set_overwrites_existing_value(self) -> None:
self.store.set_teardown_timeout_seconds(10.0)
self.store.set_teardown_timeout_seconds(20.0)
self.assertEqual(20.0, self.store.get_teardown_timeout_seconds())
def test_delete_clears_value_and_returns_true(self) -> None:
self.store.set_teardown_timeout_seconds(30.0)
deleted = self.store.delete_teardown_timeout_seconds()
self.assertTrue(deleted)
self.assertIsNone(self.store.get_teardown_timeout_seconds())
def test_delete_absent_returns_false(self) -> None:
self.assertFalse(self.store.delete_teardown_timeout_seconds())
def test_is_migrated_true_after_migrate(self) -> None:
self.assertTrue(self.store.is_migrated())
def test_is_migrated_false_before_migrate(self) -> None:
store = OrchestratorConfigStore(Path(self._tmp.name) / "new.db")
self.assertFalse(store.is_migrated())
class TestResolveTeardownTimeout(unittest.TestCase):
def setUp(self) -> None:
self._tmp = tempfile.TemporaryDirectory()
self.db = Path(self._tmp.name) / "cfg.db"
def tearDown(self) -> None:
self._tmp.cleanup()
os.environ.pop(TEARDOWN_TIMEOUT_ENV, None)
def test_returns_default_when_nothing_configured(self) -> None:
self.assertEqual(
DEFAULT_TEARDOWN_TIMEOUT_SECONDS,
resolve_teardown_timeout(self.db),
)
def test_env_var_overrides_default(self) -> None:
os.environ[TEARDOWN_TIMEOUT_ENV] = "99"
self.assertEqual(99.0, resolve_teardown_timeout(self.db))
def test_env_var_overrides_db_value(self) -> None:
store = OrchestratorConfigStore(self.db)
store.migrate()
store.set_teardown_timeout_seconds(55.0)
os.environ[TEARDOWN_TIMEOUT_ENV] = "77"
self.assertEqual(77.0, resolve_teardown_timeout(self.db))
def test_db_value_overrides_default(self) -> None:
store = OrchestratorConfigStore(self.db)
store.migrate()
store.set_teardown_timeout_seconds(42.0)
self.assertEqual(42.0, resolve_teardown_timeout(self.db))
def test_invalid_env_var_falls_through_to_default(self) -> None:
os.environ[TEARDOWN_TIMEOUT_ENV] = "not-a-number"
self.assertEqual(DEFAULT_TEARDOWN_TIMEOUT_SECONDS, resolve_teardown_timeout(self.db))
def test_non_positive_env_var_falls_through_to_default(self) -> None:
os.environ[TEARDOWN_TIMEOUT_ENV] = "0"
self.assertEqual(DEFAULT_TEARDOWN_TIMEOUT_SECONDS, resolve_teardown_timeout(self.db))
def test_non_positive_db_value_falls_through_to_default(self) -> None:
store = OrchestratorConfigStore(self.db)
store.migrate()
store.set_teardown_timeout_seconds(0.0)
self.assertEqual(DEFAULT_TEARDOWN_TIMEOUT_SECONDS, resolve_teardown_timeout(self.db))
def test_migrates_db_on_first_call(self) -> None:
result = resolve_teardown_timeout(self.db)
self.assertEqual(DEFAULT_TEARDOWN_TIMEOUT_SECONDS, result)
self.assertTrue(OrchestratorConfigStore(self.db).is_migrated())
class TestTeardownTimeoutResolvedBeforeRegistration(unittest.TestCase):
"""Ordering invariant: if resolve_teardown_timeout() raises, the bottle
must not yet be registered no orphaned state can result."""
def _src(self, module: ModuleType) -> str:
return inspect.getsource(module)
def test_docker_resolves_timeout_before_launch_consolidated(self) -> None:
from bot_bottle.backend.docker import launch
src = self._src(launch)
resolve_at = src.index("teardown_timeout = resolve_teardown_timeout()")
launch_at = src.index("ctx = launch_consolidated(")
self.assertLess(resolve_at, launch_at)
def test_firecracker_resolves_timeout_before_launch_consolidated(self) -> None:
from bot_bottle.backend.firecracker import launch
src = self._src(launch)
resolve_at = src.index("teardown_timeout = resolve_teardown_timeout()")
launch_at = src.index("ctx = launch_consolidated(")
self.assertLess(resolve_at, launch_at)
def test_macos_resolves_timeout_before_register_agent(self) -> None:
from bot_bottle.backend.macos_container import launch
src = self._src(launch)
resolve_at = src.index("teardown_timeout = resolve_teardown_timeout()")
register_at = src.index("ctx = register_agent(")
self.assertLess(resolve_at, register_at)
if __name__ == "__main__":
unittest.main()