Compare commits
5 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 55def3732e | |||
| 6a22b9f654 | |||
| a4d8461081 | |||
| 6c7b9c2f31 | |||
| 72e35a1343 |
@@ -75,6 +75,22 @@ On compatible macOS hosts, the default backend requires Apple's `container` CLI
|
||||
|
||||
Use `BOT_BOTTLE_BACKEND=docker ./cli.py start <agent>` on hosts where neither Apple Container nor KVM is available and Docker is the desired backend.
|
||||
|
||||
> **Experimental containers-in-bottle spike (#392):** a bottle may set
|
||||
> `docker_access: true`. On the macOS backend this starts a guest-local,
|
||||
> rootless **podman** service after the bottle is registered, exposing its
|
||||
> Docker-compatible API socket — the agent still uses `docker` and `docker
|
||||
> compose`. It does not mount Docker Desktop's socket or add outer VM
|
||||
> capabilities. Rootless Docker was tried first and does not work here at
|
||||
> all: Apple Container's capability bounding set omits `CAP_SYS_ADMIN`,
|
||||
> which the kernel requires to write a multi-range `uid_map`. See
|
||||
> [`docs/research/rootless-docker-in-apple-container-spike.md`](docs/research/rootless-docker-in-apple-container-spike.md).
|
||||
>
|
||||
> The tradeoff to understand before enabling it: podman avoids that
|
||||
> requirement by falling back to a single-UID mapping, so nested containers
|
||||
> provide **no isolation from the agent itself** — `root` inside a nested
|
||||
> container is the agent user outside it. Nested containers are a build/test
|
||||
> convenience, not a security boundary. The bottle remains the boundary.
|
||||
|
||||
### Firecracker on Linux
|
||||
|
||||
On Linux, a KVM-capable host defaults to the Firecracker backend. It needs:
|
||||
|
||||
@@ -20,6 +20,7 @@ class MacosContainerBottlePlan(BottlePlan):
|
||||
# bottle is registered. See launch.py's stamp for why it lives here and not
|
||||
# only in the exec-time proxy env.
|
||||
identity_token: str = ""
|
||||
docker_access: bool = False
|
||||
|
||||
@property
|
||||
def container_name(self) -> str:
|
||||
|
||||
@@ -64,6 +64,7 @@ from .gateway_hosts import (
|
||||
refresh_gateway_host,
|
||||
set_gateway_host,
|
||||
)
|
||||
from . import rootless_podman
|
||||
from .bottle_plan import MacosContainerBottlePlan
|
||||
from ...orchestrator.config_store import resolve_teardown_timeout
|
||||
from .consolidated_launch import (
|
||||
@@ -171,6 +172,10 @@ def launch(
|
||||
# token above, so — unlike the run-time env — the plan CAN carry it.
|
||||
plan = dataclasses.replace(plan, identity_token=ctx.identity_token)
|
||||
|
||||
exec_env = {
|
||||
**_identity_proxy_env(endpoint, ctx.identity_token),
|
||||
**rootless_podman.guest_env(plan.docker_access),
|
||||
}
|
||||
bottle = MacosContainerBottle(
|
||||
plan.container_name,
|
||||
teardown,
|
||||
@@ -184,10 +189,16 @@ def launch(
|
||||
),
|
||||
terminal_color=plan.spec.color,
|
||||
agent_workdir=plan.workspace_plan.workdir,
|
||||
exec_env=_identity_proxy_env(endpoint, ctx.identity_token),
|
||||
exec_env=exec_env,
|
||||
)
|
||||
bottle.prompt_path = provision(plan, bottle)
|
||||
|
||||
if plan.docker_access:
|
||||
rootless_podman.prepare_guest_devices(
|
||||
plan.container_name, container_mod.exec_container_as_root,
|
||||
)
|
||||
rootless_podman.start(bottle)
|
||||
|
||||
yield bottle
|
||||
finally:
|
||||
teardown()
|
||||
@@ -199,15 +210,22 @@ def _build_images(plan: MacosContainerBottlePlan) -> MacosContainerBottlePlan:
|
||||
committed = read_committed_image(plan.slug)
|
||||
if committed and container_mod.image_exists(committed):
|
||||
info(f"using committed image {committed!r}")
|
||||
return dataclasses.replace(
|
||||
plan = dataclasses.replace(
|
||||
plan,
|
||||
agent_provision=dataclasses.replace(
|
||||
plan.agent_provision, image=committed,
|
||||
),
|
||||
)
|
||||
container_mod.build_image(
|
||||
plan.image, _REPO_DIR, dockerfile=plan.dockerfile_path,
|
||||
)
|
||||
else:
|
||||
container_mod.build_image(
|
||||
plan.image, _REPO_DIR, dockerfile=plan.dockerfile_path,
|
||||
)
|
||||
if plan.docker_access:
|
||||
image = rootless_podman.build_image(plan.image, container_mod.build_image)
|
||||
plan = dataclasses.replace(
|
||||
plan,
|
||||
agent_provision=dataclasses.replace(plan.agent_provision, image=image),
|
||||
)
|
||||
return plan
|
||||
|
||||
|
||||
|
||||
@@ -44,4 +44,5 @@ def resolve_plan(
|
||||
egress_plan=egress_plan,
|
||||
supervise_plan=supervise_plan,
|
||||
agent_provision=agent_provision_plan,
|
||||
docker_access=manifest.bottle.docker_access,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
|
||||
uid="$(id -u)"
|
||||
if [ "$uid" -eq 0 ]; then
|
||||
echo "refusing to run rootless podman as root" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
for command in podman docker fuse-overlayfs slirp4netns; do
|
||||
command -v "$command" >/dev/null 2>&1 || {
|
||||
echo "missing rootless podman prerequisite: $command" >&2
|
||||
exit 1
|
||||
}
|
||||
done
|
||||
|
||||
# The inverse of the rootless-Docker check, and the whole point of the podman
|
||||
# variant: a subordinate range would push podman onto newuidmap, which cannot
|
||||
# write a multi-range uid_map without CAP_SYS_ADMIN in this guest. An empty
|
||||
# range keeps it on the single-UID self-mapping an unprivileged process may
|
||||
# write itself.
|
||||
if grep -q "^$(id -un):" /etc/subuid 2>/dev/null; then
|
||||
echo "unexpected subordinate UID range for $(id -un): podman would" >&2
|
||||
echo "require CAP_SYS_ADMIN via newuidmap in this guest" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
for device in /dev/fuse /dev/net/tun; do
|
||||
[ -r "$device" ] && [ -w "$device" ] || {
|
||||
echo "device $device is not readable/writable by $(id -un)" >&2
|
||||
exit 1
|
||||
}
|
||||
done
|
||||
|
||||
export XDG_RUNTIME_DIR="${XDG_RUNTIME_DIR:-/tmp/bot-bottle-podman-run}"
|
||||
config="$HOME/.config/containers"
|
||||
mkdir -p "$XDG_RUNTIME_DIR" "$config"
|
||||
chmod 700 "$XDG_RUNTIME_DIR"
|
||||
|
||||
# ignore_chown_errors is required, not incidental: with a single-UID mapping
|
||||
# there is no second UID for image layers to be chowned to, so layers that
|
||||
# record other owners would otherwise fail to extract.
|
||||
cat > "$config/storage.conf" <<'CONF'
|
||||
[storage]
|
||||
driver="overlay"
|
||||
[storage.options.overlay]
|
||||
mount_program="/usr/bin/fuse-overlayfs"
|
||||
ignore_chown_errors="true"
|
||||
CONF
|
||||
|
||||
# No cgroup delegation reaches this guest, so asking podman to manage cgroups
|
||||
# fails; events_logger=file avoids the journald socket that is equally absent.
|
||||
cat > "$config/containers.conf" <<'CONF'
|
||||
[containers]
|
||||
cgroups="disabled"
|
||||
[engine]
|
||||
cgroup_manager="cgroupfs"
|
||||
events_logger="file"
|
||||
CONF
|
||||
|
||||
# Registry pulls egress through the bottle's proxy like everything else. The
|
||||
# token-bearing proxy URL is already in the agent's environment; persisting it
|
||||
# inside this disposable VM does not broaden its authority.
|
||||
python3 - <<'PY'
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
proxy = os.environ.get("HTTPS_PROXY") or os.environ.get("https_proxy", "")
|
||||
no_proxy = os.environ.get("NO_PROXY") or os.environ.get("no_proxy", "")
|
||||
config = {"proxies": {"default": {
|
||||
"httpProxy": proxy,
|
||||
"httpsProxy": proxy,
|
||||
"noProxy": no_proxy,
|
||||
}}}
|
||||
path = Path.home() / ".docker" / "config.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(config), encoding="utf-8")
|
||||
path.chmod(0o600)
|
||||
PY
|
||||
|
||||
if docker info >/dev/null 2>&1; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
log=/tmp/bot-bottle-rootless-podman.log
|
||||
nohup podman system service --time=0 \
|
||||
"unix://$XDG_RUNTIME_DIR/podman.sock" \
|
||||
>"$log" 2>&1 </dev/null &
|
||||
@@ -0,0 +1,132 @@
|
||||
"""Experimental rootless podman bootstrap for Apple-container bottles.
|
||||
|
||||
The service and every nested container remain inside the existing per-bottle
|
||||
VM. This module refuses to compensate for missing prerequisites with outer
|
||||
capabilities, a privileged container, or a host Docker socket.
|
||||
|
||||
Podman is used rather than rootless Docker for one specific reason: Apple
|
||||
Container's capability bounding set omits `CAP_SYS_ADMIN`, which the kernel
|
||||
requires to write a multi-range `uid_map` via `newuidmap`. Rootless Docker
|
||||
has no path that avoids that write. Podman does — with no subordinate UID
|
||||
range configured it falls back to a single-UID self-mapping, which an
|
||||
unprivileged process may write itself. See
|
||||
`docs/research/rootless-docker-in-apple-container-spike.md`.
|
||||
|
||||
That fallback is why `build_image` *removes* the agent user's `/etc/subuid`
|
||||
and `/etc/subgid` entries instead of adding them: their presence is precisely
|
||||
what would send podman down the `newuidmap` path that cannot work here.
|
||||
|
||||
The agent still talks to `docker` and `docker compose`; those speak to
|
||||
podman's Docker-compatible API socket, so nothing in the agent's habits
|
||||
changes.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import shlex
|
||||
import shutil
|
||||
import tempfile
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
from ...log import die, info
|
||||
|
||||
_INIT = "/usr/local/libexec/bot-bottle/rootless-podman-init"
|
||||
_RUNTIME_DIR = "/tmp/bot-bottle-podman-run"
|
||||
_SOCKET = f"{_RUNTIME_DIR}/podman.sock"
|
||||
_LOG = "/tmp/bot-bottle-rootless-podman.log"
|
||||
READY_RETRIES = 30
|
||||
|
||||
# Apple Container creates both device nodes 0600 root:root, so the agent user
|
||||
# cannot open them: /dev/fuse blocks the fuse-overlayfs storage driver and
|
||||
# /dev/net/tun blocks slirp4netns, which rootless podman uses for the default
|
||||
# bridge network that stock compose files expect. Relaxing the modes needs no
|
||||
# capability the bottle does not already hold — unlike CAP_SYS_ADMIN, which is
|
||||
# what killed the rootless-Docker approach.
|
||||
_GUEST_DEVICES = ("/dev/fuse", "/dev/net/tun")
|
||||
|
||||
|
||||
def build_image(
|
||||
base_image: str,
|
||||
build: Callable[..., None],
|
||||
) -> str:
|
||||
"""Layer spike-only tooling on an already-built provider image."""
|
||||
image = f"{base_image}-rootless-podman"
|
||||
init_script = Path(__file__).with_name("rootless-podman-init.sh")
|
||||
with tempfile.TemporaryDirectory(prefix="bot-bottle-rootless-podman.") as tmp:
|
||||
context = Path(tmp)
|
||||
shutil.copy2(init_script, context / "rootless-podman-init.sh")
|
||||
(context / "Dockerfile").write_text(
|
||||
"FROM docker:28-cli AS docker_cli\n"
|
||||
f"FROM {base_image}\n"
|
||||
"USER root\n"
|
||||
"COPY --from=docker_cli /usr/local/bin/docker /usr/local/bin/docker\n"
|
||||
"COPY --from=docker_cli /usr/local/libexec/docker/cli-plugins/"
|
||||
"docker-compose /usr/local/libexec/docker/cli-plugins/docker-compose\n"
|
||||
"RUN apt-get update \\\n"
|
||||
" && apt-get install -y --no-install-recommends podman "
|
||||
"fuse-overlayfs slirp4netns uidmap \\\n"
|
||||
" && rm -rf /var/lib/apt/lists/* \\\n"
|
||||
# Deliberate: an empty subordinate range keeps podman on the
|
||||
# single-UID mapping that needs no CAP_SYS_ADMIN. Adding ranges
|
||||
# here would reintroduce the newuidmap failure this spike exists
|
||||
# to route around.
|
||||
" && sed -i '/^node:/d' /etc/subuid /etc/subgid\n"
|
||||
"COPY rootless-podman-init.sh "
|
||||
"/usr/local/libexec/bot-bottle/rootless-podman-init\n"
|
||||
"RUN chmod 0755 /usr/local/libexec/bot-bottle/rootless-podman-init\n"
|
||||
"USER node\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
build(image, str(context), dockerfile=str(context / "Dockerfile"))
|
||||
return image
|
||||
|
||||
|
||||
def guest_env(enabled: bool) -> dict[str, str]:
|
||||
"""Environment consumed by the Docker CLI inside an enabled bottle."""
|
||||
if not enabled:
|
||||
return {}
|
||||
return {
|
||||
"DOCKER_HOST": f"unix://{_SOCKET}",
|
||||
"XDG_RUNTIME_DIR": _RUNTIME_DIR,
|
||||
}
|
||||
|
||||
|
||||
def prepare_guest_devices(container_name: str, exec_as_root: Callable[..., None]) -> None:
|
||||
"""Make /dev/fuse and /dev/net/tun openable by the agent user.
|
||||
|
||||
Runs as root inside the bottle because the agent must not be able to
|
||||
re-mode device nodes itself. No outer capability is involved.
|
||||
"""
|
||||
exec_as_root(
|
||||
container_name,
|
||||
["sh", "-c", f"chmod 0666 {' '.join(_GUEST_DEVICES)}"],
|
||||
)
|
||||
|
||||
|
||||
def start(bottle: object) -> None:
|
||||
"""Start and verify the unprivileged service through the bottle exec API."""
|
||||
info("starting experimental rootless podman service")
|
||||
result = bottle.exec(shlex.quote(_INIT)) # type: ignore[attr-defined]
|
||||
if result.returncode != 0:
|
||||
detail = (result.stderr or result.stdout or "").strip()
|
||||
die(f"rootless podman bootstrap failed: {detail or '<no output>'}")
|
||||
|
||||
for _ in range(READY_RETRIES):
|
||||
result = bottle.exec("docker info >/dev/null 2>&1") # type: ignore[attr-defined]
|
||||
if result.returncode == 0:
|
||||
info("rootless podman service is ready")
|
||||
return
|
||||
time.sleep(0.2)
|
||||
|
||||
logs = bottle.exec( # type: ignore[attr-defined]
|
||||
f"tail -n 80 {_LOG} 2>/dev/null || true"
|
||||
)
|
||||
die(
|
||||
"rootless podman did not become ready without additional outer "
|
||||
f"privileges:\n{(logs.stdout or logs.stderr or '<no log>').strip()}"
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["build_image", "guest_env", "prepare_guest_devices", "start"]
|
||||
@@ -44,6 +44,9 @@ class ManifestBottle:
|
||||
# daemon that exposes egress MCP tools to the agent. Set
|
||||
# `supervise: false` to skip the gateway.
|
||||
supervise: bool = True
|
||||
# Experimental guest-local container engine (issue #392). Backends must
|
||||
# implement this without granting access to a host/shared daemon.
|
||||
docker_access: bool = False
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, name: str, raw: object) -> "ManifestBottle":
|
||||
@@ -123,7 +126,15 @@ class ManifestBottle:
|
||||
f"(was {type(supervise_raw).__name__})"
|
||||
)
|
||||
|
||||
docker_access_raw = d.get("docker_access", False)
|
||||
if not isinstance(docker_access_raw, bool):
|
||||
raise ManifestError(
|
||||
f"bottle '{name}' docker_access must be a boolean "
|
||||
f"(was {type(docker_access_raw).__name__})"
|
||||
)
|
||||
|
||||
return cls(
|
||||
env=env, agent_provider=agent_provider, git=git,
|
||||
git_user=git_user, egress=egress, supervise=supervise_raw,
|
||||
docker_access=docker_access_raw,
|
||||
)
|
||||
|
||||
@@ -54,6 +54,7 @@ def _merge_two_bottles_runtime(base: "ManifestBottle", override: "ManifestBottle
|
||||
git_user=merged_git_user,
|
||||
egress=merged_egress,
|
||||
supervise=override.supervise,
|
||||
docker_access=override.docker_access,
|
||||
)
|
||||
|
||||
|
||||
@@ -206,6 +207,7 @@ def _fold_two_bottles(
|
||||
git_user=merged_git_user,
|
||||
egress=merged_egress,
|
||||
supervise=later.supervise,
|
||||
docker_access=later.docker_access,
|
||||
), merged_repos_raw
|
||||
|
||||
|
||||
@@ -266,6 +268,11 @@ def _merge_bottles(
|
||||
merged_supervise = (
|
||||
child.supervise if "supervise" in child_raw else parent.supervise
|
||||
)
|
||||
merged_docker_access = (
|
||||
child.docker_access
|
||||
if "docker_access" in child_raw
|
||||
else parent.docker_access
|
||||
)
|
||||
validate_egress_routes(name, merged_egress.routes)
|
||||
|
||||
return ManifestBottle(
|
||||
@@ -275,6 +282,7 @@ def _merge_bottles(
|
||||
git_user=merged_git_user,
|
||||
egress=merged_egress,
|
||||
supervise=merged_supervise,
|
||||
docker_access=merged_docker_access,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -16,7 +16,10 @@ _FILENAME_RX = re.compile(r"^[a-z][a-z0-9-]*$")
|
||||
# sets dies with a "did you mean" pointer: typos should not silently
|
||||
# ghost into an empty config.
|
||||
BOTTLE_KEYS = frozenset(
|
||||
{"env", "extends", "agent_provider", "git-gate", "egress", "supervise"}
|
||||
{
|
||||
"env", "extends", "agent_provider", "git-gate", "egress", "supervise",
|
||||
"docker_access",
|
||||
}
|
||||
)
|
||||
AGENT_KEYS_REQUIRED: frozenset[str] = frozenset()
|
||||
AGENT_KEYS_OPTIONAL = frozenset({"bottle", "skills", "git-gate"})
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
# Egress proxy OOMs on large downloads
|
||||
|
||||
Found on 2026-07-21 while running the rootless-podman spike
|
||||
(`docs/research/rootless-docker-in-apple-container-spike.md`). Recorded
|
||||
rather than fixed — the fix is a security-relevant decision, not a
|
||||
mechanical patch.
|
||||
|
||||
## Summary
|
||||
|
||||
A single large HTTPS download through the gateway kills the egress
|
||||
proxy. `mitmdump` buffers whole response bodies so the DLP detectors can
|
||||
scan them, grows past the gateway container's memory limit, and is
|
||||
OOM-killed by the cgroup. Nothing restarts it.
|
||||
|
||||
Two properties make this worse than a failed download:
|
||||
|
||||
- **The gateway is a per-host singleton.** Every bottle shares it, so
|
||||
one bottle's download takes egress away from all of them.
|
||||
- **There is no restart on death.** The gateway supervisor is
|
||||
`while : ; do wait ; done`; a killed daemon stays dead until the infra
|
||||
container is recreated.
|
||||
|
||||
So ordinary agent activity — pulling a container image, downloading a
|
||||
model or dataset, fetching a large tarball — is a denial of service
|
||||
against every other bottle on the host. No malice required, though it is
|
||||
trivially reachable on purpose.
|
||||
|
||||
## Evidence
|
||||
|
||||
Triggered by `docker compose up` pulling `quay.io/fedora/python-312`
|
||||
(two layers, ~82MB and ~83MB) inside a bottle. The pull itself
|
||||
succeeded; the *next* request failed:
|
||||
|
||||
```
|
||||
initializing source docker://quay.io/fedora/python-312:latest:
|
||||
pinging container registry quay.io: Get "https://quay.io/v2/":
|
||||
proxyconnect tcp: dial tcp 192.168.128.39:9099: connect: connection refused
|
||||
```
|
||||
|
||||
From the gateway's `dmesg`:
|
||||
|
||||
```
|
||||
python3 invoked oom-killer: gfp_mask=0x100cca(GFP_HIGHUSER_MOVABLE), order=0
|
||||
oom-kill:constraint=CONSTRAINT_MEMCG,
|
||||
oom_memcg=/container/bot-bottle-mac-infra,
|
||||
task_memcg=/container/bot-bottle-mac-infra,task=mitmdump,pid=118
|
||||
Memory cgroup out of memory: Killed process 118 (mitmdump)
|
||||
total-vm:1391936kB, anon-rss:997768kB
|
||||
```
|
||||
|
||||
~1GB RSS against a 1024MB container. Note the amplification: ~165MB of
|
||||
layers produced ~1GB of resident memory, so the buffering is several
|
||||
copies deep (encoded body, decoded body, and the text conversion the
|
||||
regex detectors scan).
|
||||
|
||||
Afterwards the gateway container was still running and healthy-looking —
|
||||
orchestrator, supervise, and git-http all alive — with no `mitmdump`
|
||||
process at all, and it stayed that way until the container was
|
||||
recreated. A liveness check on the container would not have caught this.
|
||||
|
||||
## Reproduction
|
||||
|
||||
1. Launch any bottle with an egress route to a host serving a large file.
|
||||
2. Download >~150MB over HTTPS through the proxy.
|
||||
3. `dmesg | grep -i oom` inside `bot-bottle-mac-infra`, and note that no
|
||||
`mitmdump` process remains.
|
||||
|
||||
Beware a false negative when checking: truncating the process listing
|
||||
(`cut -c1-45`) cuts before the binary name, because `mitmdump` runs as
|
||||
`/usr/local/bin/python3.12 /usr/local/bin/mitmdump …`.
|
||||
|
||||
## Fix options, not yet chosen
|
||||
|
||||
1. **Restart dead daemons.** Smallest change and strictly an
|
||||
improvement: an OOM then degrades one download instead of removing
|
||||
egress for every bottle. Does not stop the OOM.
|
||||
2. **Cap the scanned body size.** Above a threshold, stop buffering —
|
||||
either skip the scan or stream it. This is the root-cause fix and a
|
||||
security decision: a size threshold is exactly the hole an exfiltrator
|
||||
would aim for, so "skip above N" trades a DoS for a covert channel.
|
||||
Streaming with a bounded window keeps coverage, at more complexity.
|
||||
3. **Raise the gateway's memory limit.** Moves the threshold; does not
|
||||
remove it.
|
||||
|
||||
Worth noting that (1) and (2) are complementary — the restart gap is
|
||||
worth closing regardless of how the memory behaviour is resolved.
|
||||
@@ -0,0 +1,353 @@
|
||||
# Rootless Docker inside Apple Container bottles
|
||||
|
||||
Spike branch: `spike/rootless-docker-macos` (`a4d8461`)
|
||||
|
||||
## Summary
|
||||
|
||||
**Negative result.** Rootless Docker cannot run inside an Apple
|
||||
Container bottle without granting the bottle `CAP_SYS_ADMIN`. This is a
|
||||
kernel constraint on writing multi-range `uid_map`, not a packaging gap
|
||||
we can close with a better init script, a different base image, or more
|
||||
careful `/etc/subuid` handling.
|
||||
|
||||
The spike was built on the premise — stated in
|
||||
`bot_bottle/backend/macos_container/rootless_docker.py` — that it would
|
||||
*"deliberately refuse to compensate for missing prerequisites with outer
|
||||
capabilities, a privileged container, or a host Docker socket."* That
|
||||
premise is exactly what the experiment falsified. The two ways forward
|
||||
are to abandon the premise (add `CAP_SYS_ADMIN` to the bottle, and with
|
||||
it most of the isolation the bottle exists to provide) or to abandon
|
||||
rootless Docker.
|
||||
|
||||
Recommendation: abandon rootless Docker. Podman does not have this
|
||||
problem — see [Podman is not blocked by
|
||||
this](#podman-is-not-blocked-by-this) below.
|
||||
|
||||
## Local environment
|
||||
|
||||
Tested on 2026-07-21:
|
||||
|
||||
```console
|
||||
$ sw_vers
|
||||
ProductName: macOS
|
||||
ProductVersion: 26.5.1
|
||||
BuildVersion: 25F80
|
||||
|
||||
$ container --version
|
||||
container CLI version 1.0.0 (build: release, commit: ee848e3)
|
||||
|
||||
$ uname -a # inside the bottle
|
||||
Linux ... 6.18.15 #1 SMP Tue Mar 17 01:36:53 UTC 2026 aarch64 GNU/Linux
|
||||
```
|
||||
|
||||
## The failure
|
||||
|
||||
`tests/integration/test_macos_rootless_docker_spike.py` builds the
|
||||
image, launches the bottle, and dies in `rootless_docker.start`:
|
||||
|
||||
```
|
||||
+ exec rootlesskit --net=slirp4netns --mtu=65520 ... dockerd-rootless.sh
|
||||
[rootlesskit:parent] error: failed to setup UID/GID map:
|
||||
newuidmap 1100 [0 1000 1 1 100000 65536] failed:
|
||||
newuidmap: write to uid_map failed: Operation not permitted
|
||||
```
|
||||
|
||||
## Why it fails
|
||||
|
||||
Every prerequisite you would normally suspect is present and correct in
|
||||
the guest:
|
||||
|
||||
| Check | Result |
|
||||
| --- | --- |
|
||||
| `/usr/bin/newuidmap` | `-rwsr-xr-x root root` — setuid bit intact, survived the OCI export |
|
||||
| `/` mount options | `rw,relatime` — **not** `nosuid` |
|
||||
| `NoNewPrivs` | `0` |
|
||||
| `Seccomp` | `0`, no filters |
|
||||
| `/etc/subuid`, `/etc/subgid` | `node:100000:65536` in both |
|
||||
| user namespace | `user:[4026531837]`, identical to pid 1 — the *initial* userns |
|
||||
| `unshare -U -r true` | succeeds |
|
||||
| `/proc/sys/user/max_user_namespaces` | `4505` |
|
||||
|
||||
The one thing that is missing is in the capability bounding set that
|
||||
Apple Container gives the container:
|
||||
|
||||
```
|
||||
CapBnd: 00000000a80425fb
|
||||
= chown, dac_override, fowner, fsetid, kill, setgid, setuid, setpcap,
|
||||
net_bind_service, net_raw, sys_chroot, mknod, audit_write, setfcap
|
||||
```
|
||||
|
||||
No `CAP_SYS_ADMIN`. That is the whole story, and the chain is:
|
||||
|
||||
1. The kernel's `map_write()` gates writing a `uid_map` on
|
||||
`file_ns_capable(file, ns, CAP_SYS_ADMIN)` — capability over the
|
||||
**new** user namespace, evaluated against the credentials that opened
|
||||
`/proc/<pid>/uid_map`.
|
||||
2. `newuidmap` is setuid-root, so it runs with euid 0 — but its
|
||||
capability sets are clamped by the bounding set, which has no
|
||||
`CAP_SYS_ADMIN`.
|
||||
3. `cap_capable()` has a shortcut that grants *all* capabilities when
|
||||
the caller's userns is the new namespace's parent **and**
|
||||
`ns->owner == cred->euid`. It does not apply: the namespace was
|
||||
created by `node` (uid 1000) while `newuidmap` runs as euid 0.
|
||||
4. So the check falls through to the effective-set test in the initial
|
||||
userns, which fails. `EPERM`.
|
||||
|
||||
Note that the single-line unprivileged path (`unshare -U -r`) works
|
||||
precisely because it does not go through `newuidmap` and does not need
|
||||
`CAP_SYS_ADMIN`. Only the multi-range subuid mapping that rootless
|
||||
Docker requires does.
|
||||
|
||||
This is the same constraint that makes upstream's `dind-rootless` image
|
||||
require `--privileged`. It is not specific to Apple Container, except
|
||||
that Apple Container gives us no bounding set that includes
|
||||
`CAP_SYS_ADMIN` by default.
|
||||
|
||||
## It does work with the capability — which is the point
|
||||
|
||||
Adding the capability clears the failure immediately, and exposes one
|
||||
further, much smaller blocker: `/dev/net/tun` exists (the kernel has
|
||||
tun; `/proc/misc` lists `200 tun`) but Apple Container creates it
|
||||
`crw------- root root`, so uid 1000 cannot open it and `slirp4netns`
|
||||
fails with `open: Permission denied`. A `chmod 0666 /dev/net/tun` as
|
||||
root inside the bottle fixes that, and needs no capability beyond what
|
||||
the bottle already has.
|
||||
|
||||
With both applied by hand, the daemon comes up completely:
|
||||
|
||||
```console
|
||||
$ container run --rm -u root --cap-add CAP_SYS_ADMIN \
|
||||
bot-bottle-claude:latest-rootless-docker sh -c '...'
|
||||
Server Version: 20.10.24+dfsg1
|
||||
Storage Driver: fuse-overlayfs
|
||||
Cgroup Driver: none
|
||||
Cgroup Version: 2
|
||||
API listen on /tmp/rt/docker.sock
|
||||
```
|
||||
|
||||
So `rootless-docker-init.sh` and `rootless_docker.py` are *correct*.
|
||||
The spike did not fail on a bug. It failed on its own premise.
|
||||
|
||||
Two secondary findings from that run, relevant if anyone revisits this:
|
||||
|
||||
- Debian's `docker.io` package pins Docker **20.10** (EOL), not the 28.x
|
||||
implied by the `docker:28-cli` compose plugin the image copies in.
|
||||
- `Cgroup Driver: none` — no resource limits on nested containers.
|
||||
|
||||
## Why we should not just add the capability
|
||||
|
||||
`CAP_SYS_ADMIN` is close to a superset of "root" in practical terms —
|
||||
mount, `pivot_root`, namespace manipulation, and a long tail of
|
||||
subsystem-specific powers. Granting it to the agent bottle would
|
||||
undercut the containment argument the rest of the backend is built
|
||||
around, including the deliberately narrow choices immediately adjacent
|
||||
to it in `launch.py` (`--cap-drop CAP_NET_RAW`, no `NET_ADMIN`, a
|
||||
host-only agent network). Trading all of that for nested `docker
|
||||
compose` is a bad exchange.
|
||||
|
||||
## Podman is not blocked by this
|
||||
|
||||
Sanity-checked on the same host, same kernel, same runtime, so the
|
||||
comparison is apples to apples:
|
||||
|
||||
| Scenario | Result |
|
||||
| --- | --- |
|
||||
| Podman rootless, `/etc/subuid` populated | **Fails identically** — `newuidmap: write to uid_map failed: Operation not permitted` |
|
||||
| Podman rootless, no subuid ranges, `--network=host` | **Works**, no added capabilities |
|
||||
| Podman rootless, no subuid ranges, default netns, `/dev/net/tun` at `0600` | Fails — `slirp4netns: open("/dev/net/tun"): Permission denied` |
|
||||
| Podman rootless, no subuid ranges, default netns, `/dev/net/tun` at `0666` | **Works**, no added capabilities |
|
||||
|
||||
The difference is that podman degrades gracefully when no subuid range
|
||||
is available: it falls back to a single-UID self-mapping, which an
|
||||
unprivileged process may write itself, so `newuidmap` is never invoked
|
||||
and `CAP_SYS_ADMIN` is never needed. Docker's rootless mode has no
|
||||
equivalent fallback.
|
||||
|
||||
The cost of that fallback is real and should be weighed before building
|
||||
on it: with a single-UID mapping, every UID inside a nested container
|
||||
collapses onto the bottle's own uid 1000. There is no UID separation
|
||||
between the agent and anything it runs — `root` in a nested container is
|
||||
the agent user outside it. It also requires `ignore_chown_errors` on the
|
||||
storage driver. Whether that is acceptable depends on whether the bottle
|
||||
boundary (which is unchanged) or the nested-container boundary (which is
|
||||
effectively nil) is the one we are relying on.
|
||||
|
||||
## What the podman spike then needed
|
||||
|
||||
The podman implementation that replaced the Docker one on this branch
|
||||
turned up two more device-node blockers of the same shape as
|
||||
`/dev/net/tun` — Apple Container creates the node, but 0600 root:root:
|
||||
|
||||
- **`/dev/fuse`** — blocks the `fuse-overlayfs` storage driver
|
||||
(`fuse: failed to open /dev/fuse: Permission denied`). Without it the
|
||||
only working driver is `vfs`, which copies whole layers per container.
|
||||
- **`/dev/net/tun`** — blocks `slirp4netns`, which rootless podman uses
|
||||
for the default bridge network.
|
||||
|
||||
Both are fixed by `chmod 0666` as root inside the bottle, which needs no
|
||||
capability the bottle does not already hold. This is categorically
|
||||
different from the `CAP_SYS_ADMIN` requirement: it is a permission on a
|
||||
node that already exists, not an outer privilege grant.
|
||||
|
||||
One design note worth recording: the agent-facing surface stays `docker`
|
||||
and `docker compose`, pointed at podman's Docker-compatible API socket
|
||||
via `DOCKER_HOST`. Setting `netns="host"` in `containers.conf` does *not*
|
||||
propagate through that compat API — stock `docker run` and compose files
|
||||
request bridge networking explicitly — so slirp4netns (and therefore the
|
||||
`/dev/net/tun` chmod) is required for ordinary compose files to work at
|
||||
all. Host networking remains available per-workload via
|
||||
`--network=host`.
|
||||
|
||||
Verified working in a bottle with zero added capabilities: fuse-overlayfs
|
||||
storage, the compat API socket, `docker run` on both bridge and host
|
||||
networking, and published ports.
|
||||
|
||||
### Nested pulls collide with our own egress DLP
|
||||
|
||||
The first live run got podman up and `docker compose` running, then
|
||||
failed on the image pull:
|
||||
|
||||
```
|
||||
web Pulling
|
||||
initializing source docker://python:3.12-alpine: reading manifest ...
|
||||
StatusCode: 403, egress DLP: Generic Bearer JWT found in body
|
||||
```
|
||||
|
||||
This is bot-bottle's own egress scanner, not a podman problem. The
|
||||
Docker registry auth flow carries a bearer JWT *by protocol*, and the
|
||||
`token_patterns` detector's `Generic Bearer JWT` rule
|
||||
(`Bearer\s+[A-Za-z0-9._\-]{50,}`) matches it on every pull. Any bottle
|
||||
that pulls images will hit this.
|
||||
|
||||
The fix is per-route detector scoping, which the egress config already
|
||||
supports — drop `token_patterns` on the registry hosts and keep
|
||||
`known_secrets`:
|
||||
|
||||
```json
|
||||
{"host": "registry-1.docker.io",
|
||||
"dlp": {"outbound_detectors": ["known_secrets"]}}
|
||||
```
|
||||
|
||||
That is the right trade rather than a grudging one: `known_secrets`
|
||||
matches the bottle's *actual* credential values, so real exfil through a
|
||||
registry host is still caught. `token_patterns` on a registry route only
|
||||
ever produces protocol noise.
|
||||
|
||||
Worth generalising later: any manifest enabling `docker_access` needs
|
||||
this on its registry routes, so it probably belongs in a shared
|
||||
registry-route snippet rather than being copy-pasted per bottle.
|
||||
|
||||
### And then registry auth collides with the Authorization strip
|
||||
|
||||
With DLP scoped, the pull failed differently: `unauthorized:
|
||||
authentication required`. This one is architectural.
|
||||
|
||||
`egress_addon.py` strips agent-set `Authorization` unconditionally
|
||||
before forwarding — deliberately, so an agent cannot smuggle a
|
||||
credential out in a header the DLP detectors don't recognise. A route
|
||||
may carry gateway-injected auth instead, but only from a *static* token
|
||||
in an env var (`auth_scheme` + `token_env`).
|
||||
|
||||
Docker registry auth doesn't fit that shape. The client fetches a
|
||||
short-lived, per-repository-scope bearer token from `auth.docker.io` and
|
||||
presents it to `registry-1.docker.io`. There is no static token to
|
||||
inject, and the token the client legitimately obtained is stripped.
|
||||
|
||||
Measured inside a bottle, by hand:
|
||||
|
||||
| Step | Result |
|
||||
| --- | --- |
|
||||
| Fetch token from `auth.docker.io` | 200, 5409-byte token body |
|
||||
| Manifest request **with** that valid token | 401 |
|
||||
| Manifest request with **no** Authorization | 401 — identical |
|
||||
|
||||
A valid token behaves exactly like sending none, which is direct
|
||||
evidence the header never arrives. Any nested-container workflow that
|
||||
pulls from a registry is blocked on this, so it is not a detail that can
|
||||
be deferred: pulling base images is most of what nested containers are
|
||||
for.
|
||||
|
||||
### Registries that skip the token dance work today
|
||||
|
||||
Not every registry needs the stripped header. Measured directly:
|
||||
|
||||
| Registry | Manifest request with no `Authorization` |
|
||||
| --- | --- |
|
||||
| `quay.io` | 200 |
|
||||
| `mcr.microsoft.com` | 200 |
|
||||
| `registry.k8s.io` | 307 (redirect, no auth) |
|
||||
| `ghcr.io` | 401 |
|
||||
| `registry-1.docker.io` | 401 |
|
||||
|
||||
So "just add the registry to the bottle config" genuinely works — for
|
||||
quay, MCR, registry.k8s.io, or any unauthenticated internal registry.
|
||||
Docker Hub and GHCR are the ones that need the strip resolved. The
|
||||
acceptance test uses quay for exactly this reason.
|
||||
|
||||
Resolving it for Docker Hub means picking one of:
|
||||
|
||||
1. **Per-route opt-in to preserve client Authorization.** Smallest
|
||||
change. Note the compounding effect on exactly these routes: the DLP
|
||||
scoping above already removed `token_patterns` there, so a
|
||||
preserved-auth registry route is one where the agent may send bearer
|
||||
tokens that neither the strip nor the pattern detector inspects.
|
||||
`known_secrets` still applies, so the bottle's real credentials are
|
||||
still caught.
|
||||
2. **A registry-aware gateway** that performs the token dance itself and
|
||||
injects the result. Preserves the invariant fully; materially more
|
||||
work, and it makes the gateway speak a specific registry protocol.
|
||||
3. **Pre-seed images at provision time** (host-side `container image
|
||||
save` into podman storage), so bottles never pull at runtime.
|
||||
Preserves the invariant, and limits nested containers to
|
||||
pre-approved images — which fits the custody positioning, at the cost
|
||||
of no ad-hoc `docker pull`.
|
||||
4. **Stop.** Nested containers are not supported on this backend.
|
||||
|
||||
### Podman 4.3.1 silently swallows container exit codes
|
||||
|
||||
Debian bookworm — which the current agent base image is built on —
|
||||
ships podman 4.3.1. Through its Docker-compatible API, `docker run`
|
||||
returns 0 no matter what the container did:
|
||||
|
||||
| Command | podman 4.3.1 | podman 5.4.2 |
|
||||
| --- | --- | --- |
|
||||
| `docker run … sh -c 'exit 7'` (compat API) | **0** | 7 |
|
||||
| `docker run … sh -c 'exit 0'` (compat API) | 0 | 0 |
|
||||
| `podman run … sh -c 'exit 7'` (native) | 7 | 7 |
|
||||
|
||||
This is worse than a broken feature: every failing command an agent runs
|
||||
via `docker run` reports success. A test suite, a build step, or a CI
|
||||
script inside a bottle would pass while failing. It also silently
|
||||
defeated the acceptance test's egress-containment assertion, which is
|
||||
why that assertion now checks an in-band marker rather than an exit
|
||||
code.
|
||||
|
||||
Podman 5.4.2 (Debian trixie) fixes it, but needs two packages that
|
||||
bookworm's podman does not: `passt` (podman 5's default network tool)
|
||||
and `nftables` (netavark shells out to `nft`; without it every run fails
|
||||
with `unable to upgrade to tcp, received 500`). With both installed,
|
||||
exit codes propagate correctly and the compat API behaves.
|
||||
|
||||
The open question this leaves is where podman 5 comes from, since the
|
||||
agent base is bookworm-based:
|
||||
|
||||
1. **Move the agent images to Debian trixie.** Trixie is current stable.
|
||||
Correct, and the blast radius is every bottle, not just this feature.
|
||||
2. **Drop the compat socket and use podman natively** (`podman-docker`
|
||||
provides a `docker` shim; compose comes from `podman-compose`).
|
||||
Native podman propagates exit codes correctly even on 4.3.1. Contained
|
||||
to this feature, at the cost of `docker compose` becoming
|
||||
`docker-compose`/`podman-compose`.
|
||||
3. **Ship bookworm's podman 4.3.1 with the compat socket** — not viable.
|
||||
Silent false success is a correctness bug agents cannot see.
|
||||
|
||||
## Recommendation
|
||||
|
||||
1. Do not revive rootless Docker on this backend. This document is the
|
||||
record of why.
|
||||
2. Nested containers, if wanted, come from podman under the
|
||||
single-mapping constraint — with the explicit understanding that the
|
||||
nested-container boundary carries no security weight. `root` in a
|
||||
nested container is the agent user outside it.
|
||||
3. Nested containers are therefore a build/test convenience. The bottle
|
||||
remains the security boundary, exactly as it was.
|
||||
@@ -0,0 +1,154 @@
|
||||
"""Live-Mac acceptance spike for guest-local rootless podman (issue #392).
|
||||
|
||||
Run explicitly on an Apple Silicon/macOS 26 host:
|
||||
|
||||
BOT_BOTTLE_ROOTLESS_PODMAN_SPIKE=1 \
|
||||
python3 -m unittest tests.integration.test_macos_rootless_podman_spike -v
|
||||
|
||||
The opt-in is deliberate: ordinary Linux CI cannot execute Apple Container.
|
||||
|
||||
Podman rather than Docker because Apple Container's capability bounding set
|
||||
omits CAP_SYS_ADMIN; see
|
||||
docs/research/rootless-docker-in-apple-container-spike.md. The agent-facing
|
||||
surface is still `docker` and `docker compose`, which talk to podman's
|
||||
Docker-compatible API socket.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import platform
|
||||
import shutil
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from bot_bottle.backend import BottleSpec, get_bottle_backend
|
||||
from bot_bottle.manifest import ManifestIndex
|
||||
|
||||
|
||||
@unittest.skipUnless(
|
||||
platform.system() == "Darwin"
|
||||
and os.environ.get("BOT_BOTTLE_ROOTLESS_PODMAN_SPIKE") == "1",
|
||||
"requires an explicit live-Mac rootless-podman spike run",
|
||||
)
|
||||
class TestMacosRootlessPodmanSpike(unittest.TestCase):
|
||||
def test_compose_stays_inside_registered_bottle(self) -> None:
|
||||
workspace = Path(tempfile.mkdtemp(prefix="rootless-podman-spike."))
|
||||
stage = Path(tempfile.mkdtemp(prefix="rootless-podman-stage."))
|
||||
try:
|
||||
(workspace / "index.html").write_text("bottle-compose-ok\n")
|
||||
(workspace / "compose.yaml").write_text(
|
||||
"services:\n"
|
||||
" web:\n"
|
||||
" image: quay.io/prometheus/busybox\n"
|
||||
" working_dir: /workspace\n"
|
||||
" command: httpd -f -p 8000 -h /workspace\n"
|
||||
" volumes: ['.:/workspace']\n"
|
||||
" ports: ['18080:8000']\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
manifest = ManifestIndex.from_json_obj({
|
||||
"bottles": {"dev": {
|
||||
"docker_access": True,
|
||||
# A deliberately tiny image. Pulling a ~165MB one
|
||||
# OOM-kills the shared egress proxy, which buffers whole
|
||||
# response bodies to scan them — a real defect, but a
|
||||
# separate one from what this test covers. See the
|
||||
# research note.
|
||||
#
|
||||
# quay.io deliberately, not Docker Hub: the egress proxy
|
||||
# strips agent-set Authorization (so an agent cannot
|
||||
# smuggle a credential out in a header), and Docker Hub
|
||||
# requires a client-fetched, per-scope bearer token that
|
||||
# the strip therefore removes. quay serves manifests with
|
||||
# no Authorization at all, so a plain route is enough.
|
||||
#
|
||||
# token_patterns is still scoped off: registry traffic
|
||||
# carries bearer JWTs by protocol and trips the generic
|
||||
# rule. known_secrets stays on — it matches the bottle's
|
||||
# own credentials, which is the detector that catches
|
||||
# real exfil.
|
||||
"egress": {"routes": [
|
||||
{"host": "quay.io", "dlp": {
|
||||
"outbound_detectors": ["known_secrets"],
|
||||
}},
|
||||
{"host": "cdn01.quay.io", "dlp": {
|
||||
"outbound_detectors": ["known_secrets"],
|
||||
}},
|
||||
]},
|
||||
}},
|
||||
"agents": {"spike": {
|
||||
"bottle": "dev", "skills": [], "prompt": "",
|
||||
}},
|
||||
})
|
||||
spec = BottleSpec(
|
||||
manifest=manifest,
|
||||
agent_name="spike",
|
||||
copy_cwd=True,
|
||||
user_cwd=str(workspace),
|
||||
)
|
||||
backend = get_bottle_backend("macos-container")
|
||||
plan = backend.prepare(spec, stage_dir=stage)
|
||||
with backend.launch(plan) as bottle:
|
||||
workdir = plan.workspace_plan.workdir
|
||||
checks = (
|
||||
"docker info >/dev/null && docker compose version && "
|
||||
f"cd {workdir} && docker compose up -d --wait && "
|
||||
"curl --fail --silent http://127.0.0.1:18080/ | "
|
||||
"grep -q bottle-compose-ok"
|
||||
)
|
||||
result = bottle.exec(checks)
|
||||
self.assertEqual(
|
||||
0, result.returncode,
|
||||
f"stdout={result.stdout!r}\nstderr={result.stderr!r}",
|
||||
)
|
||||
# podman's compat API reports rootlessness through its own
|
||||
# native endpoint; the Docker-shaped SecurityOptions field does
|
||||
# not carry it.
|
||||
inspect = bottle.exec(
|
||||
"podman info --format '{{.Host.Security.Rootless}}'"
|
||||
)
|
||||
self.assertIn("true", inspect.stdout.lower())
|
||||
self.assertEqual(
|
||||
0,
|
||||
bottle.exec(
|
||||
"test \"$(id -u)\" -ne 0"
|
||||
).returncode,
|
||||
"the podman service must not be running as bottle root",
|
||||
)
|
||||
self.assertNotEqual(
|
||||
0,
|
||||
bottle.exec("test -S /var/run/docker.sock").returncode,
|
||||
"spike must never expose a host/rootful Docker socket",
|
||||
)
|
||||
# Asserted on an in-band marker, not on `docker run`'s exit
|
||||
# code: podman 4.3.1's Docker-compat API swallows the
|
||||
# container's status and returns 0 for everything, so an
|
||||
# exit-code assertion here passes whether egress was blocked
|
||||
# or wide open. That silent false pass is worse than no check
|
||||
# at all, and it is exactly this check — the one proving a
|
||||
# nested container cannot escape the egress path.
|
||||
#
|
||||
# busybox ships wget, so a failure here means egress was
|
||||
# refused rather than the binary being absent.
|
||||
direct = bottle.exec(
|
||||
"docker run --rm --env HTTP_PROXY= --env HTTPS_PROXY= "
|
||||
"--env http_proxy= --env https_proxy= "
|
||||
"quay.io/prometheus/busybox sh -c "
|
||||
"'wget -T 4 -qO- https://evil.example.com/ "
|
||||
"&& echo ESCAPED || echo CONTAINED'"
|
||||
)
|
||||
self.assertIn(
|
||||
"CONTAINED", direct.stdout,
|
||||
"an inner container obtained direct, unproxied egress: "
|
||||
f"stdout={direct.stdout!r} stderr={direct.stderr!r}",
|
||||
)
|
||||
self.assertNotIn("ESCAPED", direct.stdout)
|
||||
finally:
|
||||
shutil.rmtree(workspace, ignore_errors=True)
|
||||
shutil.rmtree(stage, ignore_errors=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -22,6 +22,7 @@ from bot_bottle.backend.macos_container.launch import (
|
||||
_agent_run_argv,
|
||||
_identity_proxy_env,
|
||||
)
|
||||
from bot_bottle.backend.macos_container.rootless_podman import guest_env
|
||||
from bot_bottle.manifest import ManifestIndex
|
||||
|
||||
_BOTTLE = "bot_bottle.backend.macos_container.bottle"
|
||||
@@ -76,6 +77,7 @@ def _plan(
|
||||
),
|
||||
agent_git_gate_url=agent_git_gate_url,
|
||||
agent_supervise_url=agent_supervise_url,
|
||||
docker_access=False,
|
||||
))
|
||||
|
||||
|
||||
@@ -178,6 +180,18 @@ class TestIdentityTokenDelivery(unittest.TestCase):
|
||||
self.assertNotIn("--env", argv)
|
||||
|
||||
|
||||
class TestRootlessPodmanEnvironment(unittest.TestCase):
|
||||
def test_disabled_bottle_gets_no_docker_environment(self) -> None:
|
||||
self.assertEqual({}, guest_env(False))
|
||||
|
||||
def test_enabled_bottle_uses_only_guest_local_socket(self) -> None:
|
||||
env = guest_env(True)
|
||||
self.assertEqual(
|
||||
"unix:///tmp/bot-bottle-podman-run/podman.sock", env["DOCKER_HOST"],
|
||||
)
|
||||
self.assertNotIn("/var/run/docker.sock", " ".join(env.values()))
|
||||
|
||||
|
||||
class TestPlanIdentityToken(unittest.TestCase):
|
||||
"""git-gate's gitconfig extraHeader and the supervise MCP --header read
|
||||
`getattr(plan, "identity_token", "")` at provision time and both bypass the
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
"""Unit coverage for the fail-closed macOS rootless-podman spike."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import unittest
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
from typing import cast
|
||||
from unittest.mock import patch
|
||||
|
||||
from bot_bottle.backend.macos_container import rootless_podman
|
||||
from bot_bottle.backend.macos_container import launch as launch_mod
|
||||
from bot_bottle.backend.macos_container.bottle_plan import MacosContainerBottlePlan
|
||||
|
||||
|
||||
class _Bottle:
|
||||
def __init__(self, results: list[SimpleNamespace]) -> None:
|
||||
self.results = results
|
||||
self.commands: list[str] = []
|
||||
|
||||
def exec(self, command: str) -> SimpleNamespace:
|
||||
self.commands.append(command)
|
||||
return self.results.pop(0)
|
||||
|
||||
|
||||
def _result(returncode: int, *, stdout: str = "", stderr: str = "") -> SimpleNamespace:
|
||||
return SimpleNamespace(returncode=returncode, stdout=stdout, stderr=stderr)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _AgentProvision:
|
||||
image: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _Plan:
|
||||
slug: str
|
||||
image: str
|
||||
dockerfile_path: str
|
||||
docker_access: bool
|
||||
agent_provision: _AgentProvision
|
||||
|
||||
|
||||
class TestRootlessPodmanStart(unittest.TestCase):
|
||||
def test_bootstraps_then_waits_for_guest_local_service(self) -> None:
|
||||
bottle = _Bottle([_result(0), _result(1), _result(0)])
|
||||
with patch.object(rootless_podman.time, "sleep"):
|
||||
rootless_podman.start(bottle)
|
||||
self.assertIn("rootless-podman-init", bottle.commands[0])
|
||||
self.assertEqual(2, bottle.commands.count("docker info >/dev/null 2>&1"))
|
||||
|
||||
def test_bootstrap_failure_is_fatal_without_privilege_fallback(self) -> None:
|
||||
bottle = _Bottle([_result(1, stderr="slirp4netns missing")])
|
||||
with patch.object(rootless_podman, "die", side_effect=RuntimeError) as die:
|
||||
with self.assertRaises(RuntimeError):
|
||||
rootless_podman.start(bottle)
|
||||
self.assertIn("slirp4netns missing", die.call_args.args[0])
|
||||
self.assertEqual(1, len(bottle.commands))
|
||||
|
||||
def test_timeout_reports_guest_log(self) -> None:
|
||||
bottle = _Bottle(
|
||||
[_result(0)]
|
||||
+ [_result(1) for _ in range(rootless_podman.READY_RETRIES)]
|
||||
+ [_result(0, stdout="operation not permitted")]
|
||||
)
|
||||
with patch.object(rootless_podman.time, "sleep"), \
|
||||
patch.object(rootless_podman, "die", side_effect=RuntimeError) as die:
|
||||
with self.assertRaises(RuntimeError):
|
||||
rootless_podman.start(bottle)
|
||||
self.assertIn("operation not permitted", die.call_args.args[0])
|
||||
|
||||
|
||||
class TestRootlessPodmanDevices(unittest.TestCase):
|
||||
def test_relaxes_only_the_two_blocked_device_nodes_as_root(self) -> None:
|
||||
calls: list[tuple[str, list[str]]] = []
|
||||
rootless_podman.prepare_guest_devices(
|
||||
"bottle-1", lambda name, argv: calls.append((name, argv)),
|
||||
)
|
||||
self.assertEqual(1, len(calls))
|
||||
name, argv = calls[0]
|
||||
self.assertEqual("bottle-1", name)
|
||||
self.assertIn("chmod 0666 /dev/fuse /dev/net/tun", argv[-1])
|
||||
|
||||
|
||||
class TestRootlessPodmanImage(unittest.TestCase):
|
||||
def test_layers_tooling_without_changing_base_image(self) -> None:
|
||||
calls: list[tuple[str, str, str]] = []
|
||||
|
||||
def build(image: str, context: str, *, dockerfile: str) -> None:
|
||||
calls.append((image, context, dockerfile))
|
||||
text = Path(dockerfile).read_text(encoding="utf-8")
|
||||
self.assertIn("FROM agent:base", text)
|
||||
self.assertIn("podman fuse-overlayfs slirp4netns uidmap", text)
|
||||
self.assertIn("USER node", text)
|
||||
self.assertTrue((Path(context) / "rootless-podman-init.sh").is_file())
|
||||
|
||||
image = rootless_podman.build_image("agent:base", build)
|
||||
self.assertEqual("agent:base-rootless-podman", image)
|
||||
self.assertEqual("agent:base-rootless-podman", calls[0][0])
|
||||
|
||||
def test_strips_subordinate_ranges_so_podman_avoids_newuidmap(self) -> None:
|
||||
"""The single-UID fallback is the entire reason podman works here.
|
||||
|
||||
A subordinate range would send podman down the newuidmap path, which
|
||||
cannot write a multi-range uid_map without CAP_SYS_ADMIN in an Apple
|
||||
Container guest — the failure that killed the rootless-Docker spike.
|
||||
"""
|
||||
seen: list[str] = []
|
||||
|
||||
def build(image: str, context: str, *, dockerfile: str) -> None:
|
||||
seen.append(Path(dockerfile).read_text(encoding="utf-8"))
|
||||
|
||||
rootless_podman.build_image("agent:base", build)
|
||||
text = seen[0]
|
||||
self.assertIn("sed -i '/^node:/d' /etc/subuid /etc/subgid", text)
|
||||
self.assertNotIn("subuid", text.replace(
|
||||
"sed -i '/^node:/d' /etc/subuid /etc/subgid", "",
|
||||
))
|
||||
|
||||
def test_launch_builds_base_then_rootless_variant(self) -> None:
|
||||
plan = cast(MacosContainerBottlePlan, cast(object, _Plan(
|
||||
slug="dev-abc",
|
||||
image="agent:base",
|
||||
dockerfile_path="/repo/Dockerfile",
|
||||
docker_access=True,
|
||||
agent_provision=_AgentProvision(image="agent:base"),
|
||||
)))
|
||||
with patch.object(launch_mod, "read_committed_image", return_value=None), \
|
||||
patch.object(launch_mod.container_mod, "build_image") as build, \
|
||||
patch.object(
|
||||
launch_mod.rootless_podman,
|
||||
"build_image",
|
||||
return_value="agent:base-rootless-podman",
|
||||
) as build_rootless:
|
||||
result = launch_mod._build_images(plan) # pylint: disable=protected-access
|
||||
|
||||
build.assert_called_once_with(
|
||||
"agent:base", launch_mod._REPO_DIR, # pylint: disable=protected-access
|
||||
dockerfile="/repo/Dockerfile",
|
||||
)
|
||||
build_rootless.assert_called_once_with("agent:base", build)
|
||||
self.assertEqual("agent:base-rootless-podman", result.agent_provision.image)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -56,6 +56,12 @@ class TestMergeBottlesRuntime(unittest.TestCase):
|
||||
result = merge_bottles_runtime([base, override])
|
||||
self.assertFalse(result.supervise)
|
||||
|
||||
def test_docker_access_later_wins(self):
|
||||
result = merge_bottles_runtime([
|
||||
_bottle(docker_access=False), _bottle(docker_access=True),
|
||||
])
|
||||
self.assertTrue(result.docker_access)
|
||||
|
||||
def test_three_bottles_merged_left_to_right(self):
|
||||
b1 = _bottle(env={"A": "1", "B": "1", "C": "1"})
|
||||
b2 = _bottle(env={"B": "2", "C": "2"})
|
||||
|
||||
@@ -44,13 +44,20 @@ class TestBottleValidation(unittest.TestCase):
|
||||
with self.assertRaises(ManifestError):
|
||||
ManifestBottle.from_dict("b", {"supervise": "yes"})
|
||||
|
||||
def test_docker_access_not_bool(self) -> None:
|
||||
with self.assertRaises(ManifestError):
|
||||
ManifestBottle.from_dict("b", {"docker_access": "yes"})
|
||||
|
||||
def test_removed_runtime_field(self) -> None:
|
||||
with self.assertRaises(ManifestError):
|
||||
ManifestBottle.from_dict("b", {"runtime": "runsc"})
|
||||
|
||||
def test_valid_minimal(self) -> None:
|
||||
b = ManifestBottle.from_dict("b", {"supervise": False, "env": {"X": "1"}})
|
||||
b = ManifestBottle.from_dict(
|
||||
"b", {"supervise": False, "docker_access": True, "env": {"X": "1"}},
|
||||
)
|
||||
self.assertFalse(b.supervise)
|
||||
self.assertTrue(b.docker_access)
|
||||
self.assertEqual({"X": "1"}, dict(b.env))
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user