Compare commits

..

25 Commits

Author SHA1 Message Date
didericis-codex fd295d4c14 fix(gateway): contain output pump shutdown races
test / image-input-builds (pull_request) Failing after 12m49s
test / unit (pull_request) Has started running
test / coverage (pull_request) Blocked by required conditions
test / integration-docker (pull_request) Blocked by required conditions
tracker-policy-pr / check-pr (pull_request) Successful in 13s
2026-07-27 04:46:00 +00:00
didericis-codex f33566941b fix(gateway): bound stdlib HTTP request work
test / image-input-builds (pull_request) Successful in 1m2s
test / unit (pull_request) Has started running
test / coverage (pull_request) Blocked by required conditions
test / integration-docker (pull_request) Blocked by required conditions
tracker-policy-pr / check-pr (pull_request) Failing after 12m29s
2026-07-27 04:46:00 +00:00
didericis-codex c7c3a79028 fix(cleanup): revalidate destructive backend plans
test / image-input-builds (pull_request) Failing after 13m11s
test / unit (pull_request) Has started running
test / coverage (pull_request) Blocked by required conditions
test / integration-docker (pull_request) Blocked by required conditions
tracker-policy-pr / check-pr (pull_request) Successful in 13s
2026-07-27 04:45:59 +00:00
didericis-codex bb1776a858 refactor(supervisor): separate MCP dispatch from transport
test / image-input-builds (pull_request) Failing after 13m22s
test / unit (pull_request) Failing after 13m27s
test / coverage (pull_request) Has been skipped
test / integration-docker (pull_request) Has been cancelled
tracker-policy-pr / check-pr (pull_request) Successful in 14s
2026-07-27 04:45:59 +00:00
didericis-codex a24fe0264d refactor(egress): extract outbound DLP request stage
test / integration-docker (pull_request) Has been cancelled
test / image-input-builds (pull_request) Failing after 13m35s
test / unit (pull_request) Failing after 13m41s
test / coverage (pull_request) Has been skipped
tracker-policy-pr / check-pr (pull_request) Failing after 12m43s
2026-07-27 04:45:59 +00:00
didericis-codex 105538d3a6 refactor(egress): extract request policy stages 2026-07-27 04:45:59 +00:00
didericis-codex ffda40abae fix(orchestrator): bound streamed request bodies
test / integration-docker (pull_request) Has been cancelled
test / image-input-builds (pull_request) Successful in 42s
test / unit (pull_request) Failing after 14m6s
test / coverage (pull_request) Has been skipped
tracker-policy-pr / check-pr (pull_request) Failing after 13s
2026-07-27 04:45:59 +00:00
didericis-codex 7dcce2ff12 fix(orchestrator): keep host client dependency-free 2026-07-27 04:45:59 +00:00
didericis-codex 31a7efc0ed refactor(orchestrator): replace manual HTTP dispatch with FastAPI 2026-07-27 04:45:59 +00:00
didericis-codex a25ea7c188 build(orchestrator): pin FastAPI runtime dependencies 2026-07-27 04:45:59 +00:00
didericis-codex 3dbf1780b4 docs(prd): require shared storage permissions
prd-number-check / require-numbered-prds (pull_request) Successful in 11s
test / image-input-builds (pull_request) Successful in 45s
test / integration-docker (pull_request) Successful in 1m8s
test / unit (pull_request) Successful in 2m30s
test / coverage (pull_request) Successful in 25s
tracker-policy-pr / check-pr (pull_request) Failing after 12m49s
2026-07-27 04:45:45 +00:00
didericis-codex ff4da6f41e docs(prd): bound heavy gateway operations
prd-number-check / require-numbered-prds (pull_request) Successful in 10s
test / unit (pull_request) Successful in 54s
test / coverage (pull_request) Has been skipped
test / integration-docker (pull_request) Has been cancelled
test / image-input-builds (pull_request) Successful in 53s
tracker-policy-pr / check-pr (pull_request) Failing after 14m19s
2026-07-27 04:16:56 +00:00
didericis-codex 3bb90da11c docs(prd): require cleanup execution integrity
test / image-input-builds (pull_request) Failing after 14m11s
test / unit (pull_request) Failing after 14m19s
prd-number-check / require-numbered-prds (pull_request) Failing after 14m25s
test / integration-docker (pull_request) Has been cancelled
test / coverage (pull_request) Has been skipped
tracker-policy-pr / check-pr (pull_request) Failing after 12m36s
2026-07-27 03:55:18 +00:00
didericis-codex 0146450951 docs(prd): extend authoritative cleanup boundaries
prd-number-check / require-numbered-prds (pull_request) Successful in 13s
test / image-input-builds (pull_request) Successful in 3m18s
test / unit (pull_request) Successful in 57s
test / integration-docker (pull_request) Successful in 1m9s
test / coverage (pull_request) Successful in 19s
tracker-policy-pr / check-pr (pull_request) Failing after 10m30s
2026-07-27 03:31:46 +00:00
didericis-codex ecaf23cdb5 docs(prd): define authoritative failure boundaries
prd-number-check / require-numbered-prds (pull_request) Successful in 9s
test / unit (pull_request) Successful in 52s
test / integration-docker (pull_request) Waiting to run
test / coverage (pull_request) Blocked by required conditions
test / image-input-builds (pull_request) Successful in 2m42s
tracker-policy-pr / check-pr (pull_request) Failing after 12s
2026-07-27 02:58:07 +00:00
didericis-codex 6e46a9b191 fix(orchestrator): satisfy adapter type contracts
prd-number-check / require-numbered-prds (pull_request) Successful in 12s
lint / lint (push) Successful in 53s
test / image-input-builds (pull_request) Successful in 1m2s
test / integration-docker (pull_request) Successful in 58s
test / unit (pull_request) Successful in 46s
test / coverage (pull_request) Successful in 21s
tracker-policy-pr / check-pr (pull_request) Failing after 7s
2026-07-27 02:19:11 +00:00
didericis-codex a59e495faa refactor(egress): split request policy pipeline stages 2026-07-27 02:19:11 +00:00
didericis-codex b8818948a0 feat(firecracker): enumerate running bottles 2026-07-27 02:19:11 +00:00
didericis-codex b09952045a fix(orchestrator): bound unauthenticated HTTP requests 2026-07-27 02:19:11 +00:00
didericis-codex de192359ee fix(security): authenticate persisted egress secrets 2026-07-27 02:19:11 +00:00
didericis-codex 7d9933edc0 fix(docker): fail closed on network address scan errors 2026-07-27 02:19:11 +00:00
didericis-codex 2bc9ef8ec0 fix(firecracker): abort cleanup on scan errors 2026-07-27 02:19:11 +00:00
didericis-codex 47b6bead69 refactor(backend): type enumeration failures 2026-07-27 02:19:11 +00:00
didericis-codex 15ecada022 fix(docker): surface enumeration failures 2026-07-27 02:19:11 +00:00
didericis-codex e2222bd96b fix(security): prohibit unauthenticated orchestrator 2026-07-27 02:19:11 +00:00
94 changed files with 455 additions and 4069 deletions
+8 -22
View File
@@ -15,7 +15,7 @@
## Features
- **Per-bottle egress allowlist** — TLS-bumped HTTP/HTTPS chokepoint with a per-manifest host allowlist; per-route path/method/header `matches` filtering; outbound DLP scanning for known tokens and secrets, inbound DLP scanning for prompt-injection attempts; DoH and arbitrary hosts blocked by default.
- **Per-route token-match policy** — each egress route picks what happens when the outbound DLP catches a token via `dlp.outbound_on_match`: `supervise` (default) holds the request and surfaces it in `bot-bottle supervise` for approval (an approved value is remembered for the life of the proxy); `redact` scrubs the value and forwards; `block` is a hard `403`. Cuts false-positive friction without weakening default-deny.
- **Per-route token-match policy** — each egress route picks what happens when the outbound DLP catches a token via `dlp.outbound_on_match`: `supervise` (default) holds the request and surfaces it in `./cli.py supervise` for approval (an approved value is remembered for the life of the proxy); `redact` scrubs the value and forwards; `block` is a hard `403`. Cuts false-positive friction without weakening default-deny.
- **Tokens the agent never sees** — host secrets live in a gateway; the agent dials `http://gateway:9099/<path>` and the proxy strips inbound `Authorization` and injects the real token before forwarding. `printenv` in the agent shows proxy URLs only.
- **Gitleaks-scanned push (git-gate)** — `bottle.git` remotes route through a per-bottle `git daemon` that gitleaks-scans incoming refs pre-receive and forwards clean refs upstream over SSH. The agent never holds the upstream credential.
- **Manifest-scoped skills + secrets** — each bottle declares its skills, env, git identity, remotes, and egress routes; unknown keys die at load.
@@ -39,7 +39,7 @@ On the legacy Docker backend, the same logical bottle is two containers per agen
The Docker topology looks like this:
```
host ( bot-bottle )
host ( ./cli.py )
starts │ stops
@@ -71,23 +71,9 @@ When the agent exits, `cli.py` tears down every gateway and both networks; nothi
## Quickstart
```sh
curl -fsSL https://gitea.dideric.is/didericis/bot-bottle/raw/branch/main/install.sh | sh
```
On compatible macOS hosts, the default backend requires Apple's `container` CLI and does not require Docker. The Firecracker backend (Linux) requires Docker on the host for the gateway plus the `firecracker` binary and KVM. The legacy Docker backend requires Docker. Claude bottles also need a long-lived Claude Code OAuth token (`claude setup-token`) exported as `BOT_BOTTLE_CLAUDE_OAUTH_TOKEN`.
The installer is a bootstrapper: it finds a suitable Python, installs bot-bottle with `pipx` (falling back to `pip --user`), creates `~/.bot-bottle`, and runs `bot-bottle doctor`. It is idempotent and never uses `sudo`. Python-native users can skip it entirely with `pipx install bot-bottle` or `uv tool install bot-bottle`.
### Requirements
**Python ≥ 3.11**, and this is the one that trips people up on macOS: the `python3` Apple ships at `/usr/bin/python3` is **3.9.6**, which is too old. Bare `python3` resolves to that stub far more often than people expect. `path_helper` builds a login shell's `PATH` from `/etc/paths` and then appends `/etc/paths.d/*`, and `/usr/bin` sits in the former — so even when `/opt/homebrew/bin` *is* on the `PATH` (via `/etc/paths.d/homebrew`), it comes after `/usr/bin` and loses. Prepending a newer Python is something your shell profile does, and a fresh account, a launchd job, or a CI runner has no such profile. So the installer looks past bare `python3` before giving up: it tries `python3`, then the versioned `python3.11``python3.14` names, then `/opt/homebrew/bin`, `/usr/local/bin`, `~/.local/bin`, and python.org framework builds — and tells you which one it picked when it isn't the obvious one. Point it somewhere specific with `BOT_BOTTLE_PYTHON=/path/to/python3`.
**No `pipx` required.** If `pipx` is present the installer uses it and stays out of the way. If it isn't, bot-bottle installs into a private venv at `~/.bot-bottle/venv` (override with `BOT_BOTTLE_VENV`) and symlinks the entry point into `~/.local/bin`. There is deliberately no `pip install --user` path: Homebrew, python.org and Debian/Ubuntu interpreters are all externally managed (PEP 668), which blocks `--user` outright — so on a Mac it is never the fallback it appears to be. A venv is exempt from PEP 668, and `venv` is stdlib, so unlike `pipx` there is nothing to bootstrap first.
**`git`**, because the default install spec is a `git+` URL. Set `BOT_BOTTLE_INSTALL_SPEC` to a wheel path or index name to avoid it.
**A backend**, which the installer deliberately does *not* install for you — `doctor` reports what's missing afterwards. On compatible macOS hosts, the default backend requires Apple's `container` CLI and does not require Docker. The Firecracker backend (Linux) requires Docker on the host for the gateway plus the `firecracker` binary and KVM. The legacy Docker backend requires Docker. Claude bottles also need a long-lived Claude Code OAuth token (`claude setup-token`) exported as `BOT_BOTTLE_CLAUDE_OAUTH_TOKEN`.
Use `BOT_BOTTLE_BACKEND=docker bot-bottle start <agent>` on hosts where neither Apple Container nor KVM is available and Docker is the desired backend.
Use `BOT_BOTTLE_BACKEND=docker ./cli.py start <agent>` on hosts where neither Apple Container nor KVM is available and Docker is the desired backend.
> **CI (macOS Apple Container):** the advisory `integration-macos` job in `.gitea/workflows/pre-release-test.yml` runs only on manual dispatch. It targets a self-hosted host-mode runner labelled `macos`; Apple Container cannot run inside the Linux pull-request runner. Provision an Apple Silicon host with the `container` CLI running and Python ≥ 3.11 plus `coverage` on the launchd service's explicit `PATH`. The infra container is a singleton (`bot-bottle-mac-infra`), so keep runner concurrency at 1. Its coverage is reported separately and never feeds the required pull-request gate.
@@ -180,10 +166,10 @@ On Linux, a KVM-capable host defaults to the Firecracker backend. It needs:
- **`/dev/kvm`** present and accessible. Load `kvm-intel` or `kvm-amd` (and enable virtualization in BIOS/firmware). The invoking user must be in the `kvm` group: `sudo usermod -aG kvm "$USER"` then re-login. bot-bottle preflights this and reports exactly what's missing.
- **`firecracker`** on `PATH`: grab a release from <https://github.com/firecracker-microvm/firecracker/releases>. Start flows print this pointer when the binary is missing.
- **Docker** for the gateway and image build.
- **A one-time privileged network setup** — the per-bottle TAP pool plus the fail-closed `nftables` isolation table. Run `bot-bottle backend setup --backend=firecracker` for the host-appropriate config (a NixOS module, a `sudo` script elsewhere); `bot-bottle backend status --backend=firecracker` reports what's present, including whether the pool range collides with an existing route. The pool defaults to `10.243.0.0/16` (an obscure RFC-1918 block that dodges docker/libvirt/LAN and, deliberately, Tailscale's `100.64.0.0/10` CGNAT range); override with `BOT_BOTTLE_FC_IP_BASE` if it clashes on your host.
- **A one-time privileged network setup** — the per-bottle TAP pool plus the fail-closed `nftables` isolation table. Run `./cli.py backend setup --backend=firecracker` for the host-appropriate config (a NixOS module, a `sudo` script elsewhere); `./cli.py backend status --backend=firecracker` reports what's present, including whether the pool range collides with an existing route. The pool defaults to `10.243.0.0/16` (an obscure RFC-1918 block that dodges docker/libvirt/LAN and, deliberately, Tailscale's `100.64.0.0/10` CGNAT range); override with `BOT_BOTTLE_FC_IP_BASE` if it clashes on your host.
```sh
BOT_BOTTLE_BACKEND=firecracker bot-bottle start <agent>
BOT_BOTTLE_BACKEND=firecracker ./cli.py start <agent>
```
> **NixOS:** enable `virtualisation.docker`, ensure the KVM module is loaded (`boot.kernelModules = [ "kvm-intel" ];` or `kvm-amd`), and add your user to the `kvm` and `docker` groups. For the network pool, consume the flake module — `imports = [ inputs.bot-bottle.nixosModules.firecracker-netpool ]; services.bot-bottle-firecracker = { enable = true; owner = "you"; };` — then `nixos-rebuild switch` (imperative nft/TAP rules don't survive a rebuild; channel users can `imports = [ <bot-bottle>/nix/firecracker-netpool.nix ]`). `firecracker` isn't in nixpkgs by default as a user binary — install the release binary (pin the version) and put it on `PATH`.
@@ -191,7 +177,7 @@ BOT_BOTTLE_BACKEND=firecracker bot-bottle start <agent>
> **CI:** Firecracker integration runs in the manually dispatched `.gitea/workflows/pre-release-test.yml` on a self-hosted runner labelled `kvm`; privileged KVM hosts never execute unreviewed PR code automatically. Provision it like a normal Firecracker host: `firecracker` on `PATH`, `/dev/kvm`, the cached guest kernel and static dropbear, and the persistent TAP/nft pool. The required pull-request workflow runs unit plus the complete Docker integration suite on `ubuntu-latest`; see `docs/ci.md`.
```sh
bot-bottle start <agent> # builds the image on first run, drops you into claude
./cli.py start <agent> # builds the image on first run, drops you into claude
```
## Manifest
@@ -267,7 +253,7 @@ You help maintain Gitea-hosted projects.
| `dlp.outbound_on_match` | no | What to do when an outbound token is detected: `supervise` (default for manifest routes — hold for operator approval), `redact` (scrub the value and forward), or `block` (hard 403). Agent-provider routes (e.g. `api.anthropic.com`) default to `redact`. |
| `git.fetch` | no | `true` permits smart HTTP clone/fetch (`git-upload-pack`) for this host. Push (`git-receive-pack`) remains blocked. |
When an outbound DLP detector matches a token, the route's `dlp.outbound_on_match` policy decides what happens. Under the default `supervise`, the proxy queues an `egress-token-allow` proposal for the operator's `bot-bottle supervise` TUI and holds the request open until it is answered (or `EGRESS_TOKEN_ALLOW_TIMEOUT_SECONDS`, default 300s, elapses — after which it fails closed). The operator never sees the raw token, only the host, method, path, and a redacted snippet; approving adds the value to an in-memory safelist for the life of the egress proxy. Under `redact`, the matched value is scrubbed from the body, headers, and path and the request is forwarded (failing closed if a match lands somewhere unredactable, like the hostname). Under `block` it stays a hard `403`. Structural blocks (CRLF injection) and not-in-allowlist host blocks are always hard `403`s regardless of policy.
When an outbound DLP detector matches a token, the route's `dlp.outbound_on_match` policy decides what happens. Under the default `supervise`, the proxy queues an `egress-token-allow` proposal for the operator's `./cli.py supervise` TUI and holds the request open until it is answered (or `EGRESS_TOKEN_ALLOW_TIMEOUT_SECONDS`, default 300s, elapses — after which it fails closed). The operator never sees the raw token, only the host, method, path, and a redacted snippet; approving adds the value to an in-memory safelist for the life of the egress proxy. Under `redact`, the matched value is scrubbed from the body, headers, and path and the request is forwarded (failing closed if a match lands somewhere unredactable, like the hostname). Under `block` it stays a hard `403`. Structural blocks (CRLF injection) and not-in-allowlist host blocks are always hard `403`s regardless of policy.
More examples in `examples/`. Full design lives under `docs/prds/`; the trust-boundary rationale is in `docs/prds/0011-per-file-md-manifest.md`.
+1 -1
View File
@@ -226,7 +226,7 @@ class AgentProvider(ABC):
initial task in a non-interactive (headless) session.
Called only when ``--prompt`` is passed to
``bot-bottle start --headless``; the returned args are appended
``./cli.py start --headless``; the returned args are appended
after the provider's ``bypass_args`` and ``startup_args``."""
def provision_ca(self, bottle: "Bottle", plan: "BottlePlan") -> None:
+3 -7
View File
@@ -172,10 +172,6 @@ class BottleCleanupPlan(ABC):
"""True iff there is nothing to clean up; the CLI uses this to
short-circuit before showing the y/N."""
@abstractmethod
def intersect(self, current: "BottleCleanupPlan") -> "BottleCleanupPlan":
"""Resources both displayed to the operator and currently removable."""
@dataclass(frozen=True)
class ExecResult:
@@ -531,7 +527,7 @@ class BottleBackend(ABC, Generic[PlanT, CleanupT]):
host-appropriate config or commands. Prints to stdout/stderr and
returns a shell exit code (0 = nothing to report / success). A
backend that needs no host setup prints a short note and returns
0. Invoked generically by `bot-bottle backend setup [--backend=…]`
0. Invoked generically by `./cli.py backend setup [--backend=…]`
so operators can provision any backend without a
backend-specific command. Classmethod (like `is_available`) —
it's a host query, not per-bottle state."""
@@ -548,14 +544,14 @@ class BottleBackend(ABC, Generic[PlanT, CleanupT]):
stderr. When quiet=True returns the status code silently —
useful for cheap programmatic checks.
Invoked by `bot-bottle backend status [--backend=…]` (quiet=False)
Invoked by `./cli.py backend status [--backend=…]` (quiet=False)
and by is_backend_ready() (caller-controlled)."""
@classmethod
@abstractmethod
def teardown(cls) -> int:
"""Undo `setup()` — the inverse operation, surfaced as
`bot-bottle backend teardown [--backend=…]` (uninstall). Symmetric
`./cli.py backend teardown [--backend=…]` (uninstall). Symmetric
with setup: where setup is advisory (prints the privileged
commands / declarative config to apply), teardown prints the
commands / config change to remove the host prerequisites. A
-60
View File
@@ -1,60 +0,0 @@
"""Shared destructive-cleanup execution and failure accounting."""
from __future__ import annotations
import os
import shutil
import subprocess
from collections.abc import Sequence
from pathlib import Path
class CleanupError(RuntimeError):
"""One or more approved cleanup mutations did not complete."""
class CleanupFailures:
"""Attempt every approved mutation, then fail with complete diagnostics."""
def __init__(self) -> None:
self._messages: list[str] = []
def run(self, argv: Sequence[str], description: str) -> None:
raw_timeout = os.environ.get(
"BOT_BOTTLE_CLEANUP_COMMAND_TIMEOUT_SECONDS", "120",
)
try:
timeout = float(raw_timeout)
except ValueError:
timeout = 120.0
try:
result = subprocess.run(
list(argv), capture_output=True, text=True, check=False,
timeout=max(timeout, 1.0),
)
except (OSError, subprocess.SubprocessError) as exc:
self._messages.append(f"{description}: {exc}")
return
if result.returncode != 0:
detail = (result.stderr or result.stdout).strip()
self._messages.append(
f"{description}: {detail or f'exit {result.returncode}'}"
)
def remove_tree(self, path: Path, description: str) -> None:
try:
shutil.rmtree(path)
except FileNotFoundError:
return
except OSError as exc:
self._messages.append(f"{description}: {exc}")
def record(self, message: str) -> None:
self._messages.append(message)
def raise_if_any(self) -> None:
if self._messages:
raise CleanupError("; ".join(self._messages))
__all__ = ["CleanupError", "CleanupFailures"]
@@ -46,22 +46,6 @@ class DockerBottleCleanupPlan(BottleCleanupPlan):
and not self.orphan_state_dirs
)
def intersect(self, current: BottleCleanupPlan) -> "DockerBottleCleanupPlan":
if not isinstance(current, DockerBottleCleanupPlan):
raise TypeError("cleanup plans must have the same backend type")
return DockerBottleCleanupPlan(
projects=tuple(x for x in self.projects if x in current.projects),
stray_containers=tuple(
x for x in self.stray_containers if x in current.stray_containers
),
stray_networks=tuple(
x for x in self.stray_networks if x in current.stray_networks
),
orphan_state_dirs=tuple(
x for x in self.orphan_state_dirs if x in current.orphan_state_dirs
),
)
def print(self) -> None:
print(file=sys.stderr)
for name in self.projects:
+38 -38
View File
@@ -23,12 +23,11 @@ Active-agent enumeration lives in `backend/docker/enumerate.py`.
from __future__ import annotations
import shutil
import subprocess
from ...paths import bot_bottle_root
from ...log import info
from .. import EnumerationError
from ..cleanup_control import CleanupFailures
from ...log import info, warn
from . import util as docker_mod
from .bottle_cleanup_plan import DockerBottleCleanupPlan
from ...bottle_state import bottle_state_dir, is_preserved
@@ -37,17 +36,15 @@ from .compose import COMPOSE_PROJECT_PREFIX, list_compose_projects
def _list_prefixed_containers() -> list[str]:
"""All bot-bottle-prefixed containers, running or stopped."""
try:
result = subprocess.run(
["docker", "ps", "-a",
"--filter", f"name=^{COMPOSE_PROJECT_PREFIX}",
"--format", "{{.Names}}\t{{.Label \"com.docker.compose.project\"}}"],
capture_output=True, text=True, check=False,
)
except OSError as exc:
raise EnumerationError(f"docker ps failed: {exc}") from exc
result = subprocess.run(
["docker", "ps", "-a",
"--filter", f"name=^{COMPOSE_PROJECT_PREFIX}",
"--format", "{{.Names}}\t{{.Label \"com.docker.compose.project\"}}"],
capture_output=True, text=True, check=False,
)
if result.returncode != 0:
raise EnumerationError(f"docker ps failed: {result.stderr.strip()}")
warn(f"docker ps failed: {result.stderr.strip()}")
return []
out: list[str] = []
for line in (result.stdout or "").splitlines():
if not line:
@@ -66,19 +63,15 @@ def _list_prefixed_networks() -> list[str]:
to a compose project. Compose-managed networks have a
`com.docker.compose.project` label; bare ones (from pre-compose
code paths) don't."""
try:
result = subprocess.run(
["docker", "network", "ls",
"--filter", f"name={COMPOSE_PROJECT_PREFIX}",
"--format", "{{.Name}}\t{{.Label \"com.docker.compose.project\"}}"],
capture_output=True, text=True, check=False,
)
except OSError as exc:
raise EnumerationError(f"docker network ls failed: {exc}") from exc
result = subprocess.run(
["docker", "network", "ls",
"--filter", f"name={COMPOSE_PROJECT_PREFIX}",
"--format", "{{.Name}}\t{{.Label \"com.docker.compose.project\"}}"],
capture_output=True, text=True, check=False,
)
if result.returncode != 0:
raise EnumerationError(
f"docker network ls failed: {result.stderr.strip()}"
)
warn(f"docker network ls failed: {result.stderr.strip()}")
return []
out: list[str] = []
for line in (result.stdout or "").splitlines():
if not line:
@@ -127,10 +120,7 @@ def prepare_cleanup() -> DockerBottleCleanupPlan:
`enumerate_active_agents()` so the orphan-state-dir bucket
doesn't include slugs whose non-docker bottle is still up."""
docker_mod.require_docker()
projects = list_compose_projects(
warn_on_error=False,
raise_on_error=True,
)
projects = list_compose_projects()
project_set = set(projects)
# Late import to avoid a circular at module-load time —
# the backend package's __init__ imports this module.
@@ -150,30 +140,40 @@ def cleanup(plan: DockerBottleCleanupPlan) -> None:
"""Remove everything in the plan. Projects first (whose `compose
down` reaps their containers + networks atomically), then stray
legacy resources, then orphan state dirs."""
failures = CleanupFailures()
for project in plan.projects:
info(f"docker compose down ({project})")
failures.run(
result = subprocess.run(
["docker", "compose", "-p", project, "down", "--volumes"],
f"docker compose down failed for {project}",
capture_output=True, text=True, check=False,
)
if result.returncode != 0:
warn(
f"compose down failed for {project}: "
f"{result.stderr.strip()}"
)
for name in plan.stray_containers:
info(f"removing stray container {name}")
failures.run(
subprocess.run(
["docker", "rm", "-f", name],
f"removing stray container {name}",
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
check=False,
)
for name in plan.stray_networks:
info(f"removing stray network {name}")
failures.run(
subprocess.run(
["docker", "network", "rm", name],
f"removing stray network {name}",
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
check=False,
)
for identity in plan.orphan_state_dirs:
path = bottle_state_dir(identity)
info(f"removing orphan state dir {path}")
failures.remove_tree(path, f"removing orphan state dir {path}")
failures.raise_if_any()
try:
shutil.rmtree(path, ignore_errors=True)
except OSError as e:
warn(f"failed to remove {path}: {e}")
+2 -5
View File
@@ -74,13 +74,10 @@ def list_compose_projects(
result = subprocess.run(
argv, capture_output=True, text=True, check=False,
)
except OSError as exc:
# Not only "not found": docker on PATH but not executable by this
# user raises PermissionError. Either way the query never ran, so an
# enumeration caller must not read the empty result as authoritative.
except FileNotFoundError as exc:
if raise_on_error:
raise EnumerationError(
f"docker compose ls failed: docker unavailable ({exc})"
"docker compose ls failed: docker not found"
) from exc
return []
if result.returncode != 0:
+2 -3
View File
@@ -74,9 +74,8 @@ def _query_services_by_project() -> dict[str, set[str]]:
],
capture_output=True, text=True, check=False,
)
except OSError as exc:
# Missing, or on PATH but not executable by this user (PermissionError).
raise EnumerationError(f"docker ps failed: docker unavailable ({exc})") from exc
except FileNotFoundError as exc:
raise EnumerationError("docker ps failed: docker not found") from exc
if r.returncode != 0:
raise EnumerationError(f"docker ps failed: {r.stderr.strip()}")
return _parse_services_by_project(r.stdout or "")
+3 -3
View File
@@ -8,7 +8,7 @@ pointer, and `status()` reports whether docker is usable.
This is intentionally minimal; a richer version (daemon config checks,
gVisor/runsc install guidance, rootless-docker hints) is tracked
separately. Reached via `DockerBottleBackend.setup` / `.status`, which
the generic `bot-bottle backend {setup,status}` dispatches to.
the generic `./cli.py backend {setup,status}` dispatches to.
"""
from __future__ import annotations
@@ -69,7 +69,7 @@ def teardown() -> int:
sys.stderr.write(
"Docker backend: nothing to undo — it provisions no privileged host "
"state (networks and the gateway are per-launch and are "
"removed by `bot-bottle cleanup`). Docker itself is left installed.\n"
"removed by `./cli.py cleanup`). Docker itself is left installed.\n"
)
return 0
@@ -89,5 +89,5 @@ def status() -> int:
runsc = _docker_on_path() and _util.runsc_available()
sys.stderr.write(f"gVisor runsc runtime: {'registered' if runsc else 'not registered (optional)'}\n")
if not ok:
sys.stderr.write("\nRun: bot-bottle backend setup --backend=docker\n")
sys.stderr.write("\nRun: ./cli.py backend setup --backend=docker\n")
return 0 if ok else 1
@@ -27,11 +27,3 @@ class FirecrackerBottleCleanupPlan(BottleCleanupPlan):
@property
def empty(self) -> bool:
return not (self.vm_pids or self.run_dirs)
def intersect(self, current: BottleCleanupPlan) -> "FirecrackerBottleCleanupPlan":
if not isinstance(current, FirecrackerBottleCleanupPlan):
raise TypeError("cleanup plans must have the same backend type")
return FirecrackerBottleCleanupPlan(
vm_pids=tuple(x for x in self.vm_pids if x in current.vm_pids),
run_dirs=tuple(x for x in self.run_dirs if x in current.run_dirs),
)
+21 -53
View File
@@ -11,9 +11,10 @@ Reaps *orphans* only — resources with no live VM behind them:
— a VMM left lingering after its dir was removed.
A run dir with a *live* firecracker process is a running bottle and is
left strictly alone: it is neither killed nor removed. Active-agent
enumeration uses this same process snapshot, so cleanup and generic
backend consumers agree about which bottles are running.
left strictly alone: it is neither killed nor removed. (The backend's
`enumerate_active` registry is still a stub — #354 — so a live process
is the only reliable "this bottle is in use" signal we have. Once the
registry lands, registry-orphaned-but-running VMs can be reaped too.)
TAP slots free themselves (the flock drops when the launcher exits), so
there is nothing to reclaim there.
@@ -21,15 +22,14 @@ there is nothing to reclaim there.
from __future__ import annotations
from collections.abc import Sequence
import os
import shutil
import signal
import subprocess
from pathlib import Path
from ...log import info
from .. import EnumerationError
from ..cleanup_control import CleanupError, CleanupFailures
from . import lifecycle_lock, util
from .bottle_cleanup_plan import FirecrackerBottleCleanupPlan
@@ -38,7 +38,7 @@ def _run_root() -> Path:
return util.cache_dir() / "run"
def _run_dir_of(args: Sequence[str], run_root: Path) -> Path | None:
def _run_dir_of(cmd: str, run_root: Path) -> Path | None:
"""The bottle run dir a firecracker cmdline belongs to, or None.
A bottle VM is launched with `--config-file <run_root>/<slug>/config.json`,
@@ -46,35 +46,15 @@ def _run_dir_of(args: Sequence[str], run_root: Path) -> Path | None:
the run root. Anything else (a builder VM, the infra VM elsewhere) is
not ours to reap here.
"""
for i, arg in enumerate(args):
if arg == "--config-file" and i + 1 < len(args):
parent = Path(args[i + 1]).parent
toks = cmd.split()
for i, tok in enumerate(toks):
if tok == "--config-file" and i + 1 < len(toks):
parent = Path(toks[i + 1]).parent
if parent.parent == run_root:
return parent
return None
def _decode_cmdline(raw: bytes) -> tuple[str, ...]:
"""Decode Linux's NUL-delimited argv without losing embedded spaces."""
return tuple(
value.decode(errors="surrogateescape")
for value in raw.split(b"\0") if value
)
def _process_args(pid: int) -> tuple[str, ...] | None:
"""Read one process's lossless argv, or None when it exited meanwhile."""
try:
raw = Path(f"/proc/{pid}/cmdline").read_bytes()
except FileNotFoundError:
return None
except OSError as exc:
raise EnumerationError(
f"could not inspect Firecracker pid {pid}: {exc}"
) from exc
return _decode_cmdline(raw)
def _scan_processes(run_root: Path) -> tuple[set[str], list[int]]:
"""Inspect running firecracker VMs under ``run_root``.
@@ -85,7 +65,7 @@ def _scan_processes(run_root: Path) -> tuple[set[str], list[int]]:
"""
try:
result = subprocess.run(
["pgrep", "firecracker"],
["pgrep", "-a", "firecracker"],
capture_output=True, text=True, check=False,
)
except OSError as exc:
@@ -103,14 +83,14 @@ def _scan_processes(run_root: Path) -> tuple[set[str], list[int]]:
live: set[str] = set()
orphan_pids: list[int] = []
for line in result.stdout.splitlines():
parts = line.split(None, 1)
if len(parts) != 2:
continue
try:
pid = int(line.strip())
pid = int(parts[0])
except ValueError:
continue
args = _process_args(pid)
if args is None:
continue
run_dir = _run_dir_of(args, run_root)
run_dir = _run_dir_of(parts[1], run_root)
if run_dir is None:
continue
if run_dir.is_dir():
@@ -151,16 +131,11 @@ def cleanup(plan: FirecrackerBottleCleanupPlan) -> None:
fresh = prepare_cleanup()
approved_pids = set(plan.vm_pids).intersection(fresh.vm_pids)
approved_dirs = set(plan.run_dirs).intersection(fresh.run_dirs)
failures = CleanupFailures()
for pid in sorted(approved_pids):
try:
_terminate_orphan(pid, _run_root())
except CleanupError as exc:
failures.record(str(exc))
_terminate_orphan(pid, _run_root())
for path in sorted(approved_dirs):
info(f"rm -rf {path}")
failures.remove_tree(Path(path), f"removing Firecracker run dir {path}")
failures.raise_if_any()
shutil.rmtree(path, ignore_errors=True)
def _terminate_orphan(pid: int, run_root: Path) -> None:
@@ -182,18 +157,11 @@ def _terminate_orphan(pid: int, run_root: Path) -> None:
raise EnumerationError(
f"could not revalidate Firecracker pid {pid}: {exc}"
) from exc
args = _decode_cmdline(raw)
run_dir = _run_dir_of(args, run_root)
command = raw.replace(b"\0", b" ").decode(errors="replace")
run_dir = _run_dir_of(command, run_root)
if run_dir is None or run_dir.is_dir():
return
info(f"kill firecracker VM pid {pid}")
try:
signal.pidfd_send_signal(pidfd, signal.SIGTERM)
except ProcessLookupError:
return
except OSError as exc:
raise CleanupError(
f"could not signal Firecracker pid {pid}: {exc}"
) from exc
signal.pidfd_send_signal(pidfd, signal.SIGTERM)
finally:
os.close(pidfd)
+8 -102
View File
@@ -19,7 +19,6 @@ from __future__ import annotations
import fcntl
import hashlib
import os
import re
import shlex
import shutil
import subprocess
@@ -42,72 +41,19 @@ _BUILD_TIMEOUT_SECONDS = 900.0
def _dockerfile_hash(dockerfile: Path) -> str:
"""The Dockerfile's content hash. Agent Dockerfiles mostly COPY nothing
from the build context, so their text nearly determines the built image;
any files they *do* COPY are folded into `_rootfs_digest` (so a changed
input busts the cache) and shipped to the VM-side context by
`_send_build_context`."""
"""The Dockerfile's content hash. The shipped agent Dockerfiles COPY
nothing from the build context (see .dockerignore), so their content fully
determines the built image; a Dockerfile that adds COPY will want the
context folded in here too."""
return hashlib.sha256(dockerfile.read_bytes()).hexdigest()[:16]
def _context_copy_sources(dockerfile: Path) -> list[str]:
"""The build-context-relative paths a Dockerfile ``COPY``s in.
Agent Dockerfiles are meant to COPY nothing from the context (the VM-side
build ships only the Dockerfile), but one may pin an input by COPYing a
committed file — a checksum list, an npm lockfile. Return those source
paths so the builder can both ship them to the VM context and fold them
into the cache key. ``COPY --from=<stage>`` reads a build stage, not the
context, so it is excluded; the JSON/exec COPY form is unused by the
shipped images and is skipped rather than mis-parsed."""
joined = re.sub(r"\\\n", " ", dockerfile.read_text(encoding="utf-8"))
sources: list[str] = []
for line in joined.splitlines():
stripped = line.strip()
if not re.match(r"(?i)^COPY\s", stripped):
continue
tokens = stripped.split()[1:]
if any(t.startswith("--from=") for t in tokens):
continue
args = [t for t in tokens if not t.startswith("--")]
if len(args) < 2 or args[0].startswith("["):
continue
sources.extend(args[:-1])
return sources
def _context_files(dockerfile: Path) -> list[tuple[str, Path]]:
"""``(context-relative path, host path)`` for every existing file a
Dockerfile COPYs from the build root — globs expanded, sorted, de-duped.
Absolute or traversing (`..`) sources are dropped: the shipped context
only ever mirrors files under the build root."""
root = resources.build_root()
resolved: dict[str, Path] = {}
for src in _context_copy_sources(dockerfile):
if src.startswith("/") or ".." in Path(src).parts:
continue
if any(ch in src for ch in "*?["):
matches = [p for p in root.glob(src) if p.is_file()]
else:
candidate = root / src
if candidate.is_dir():
matches = [path for path in candidate.rglob("*") if path.is_file()]
else:
matches = [candidate] if candidate.is_file() else []
for path in matches:
resolved[str(path.relative_to(root))] = path
return sorted(resolved.items())
def _rootfs_digest(dockerfile: Path) -> str:
"""Cache key for the built AND boot-injected agent rootfs. Its inputs are
the Dockerfile (the image), the centralized build args, the guest init
injected into it (`util._GUEST_INIT`), and the content of any files the
Dockerfile COPYs from the build context. Folding the init in means a fix to
"""Cache key for the built AND boot-injected agent rootfs. Two inputs
determine the on-disk rootfs: the Dockerfile (the image) and the guest init
injected into it (`util._GUEST_INIT`). Folding the init in means a fix to
it — e.g. making /tmp world-writable — busts the cache instead of silently
reusing a stale rootfs built with the old init; folding the COPYed context
files in means a repinned input (e.g. a changed checksum list) rebuilds
rather than reusing a rootfs baked from the old bytes."""
reusing a stale rootfs built with the old init."""
h = hashlib.sha256()
h.update(_dockerfile_hash(dockerfile).encode())
h.update(b"\0")
@@ -117,11 +63,6 @@ def _rootfs_digest(dockerfile: Path) -> str:
h.update(value.encode())
h.update(b"\0")
h.update(util._GUEST_INIT.encode())
for rel, path in _context_files(dockerfile):
h.update(b"\0")
h.update(rel.encode())
h.update(b"\0")
h.update(path.read_bytes())
return h.hexdigest()[:16]
@@ -213,7 +154,6 @@ def _build_in_infra(
if prep.returncode != 0:
die(f"preparing build dir in the infra VM failed: {prep.stderr.strip()}")
_send_dockerfile(key, ip, dockerfile, ctx)
_send_build_context(key, ip, dockerfile, ctx)
_buildah_build(
key,
ip,
@@ -257,40 +197,6 @@ def _send_dockerfile(private_key: Path, guest_ip: str, dockerfile: Path, ctx: st
f"{proc.stderr.decode(errors='replace').strip()}")
def _send_build_context(private_key: Path, guest_ip: str, dockerfile: Path, ctx: str) -> None:
"""Ship the files ``dockerfile`` COPYs from the build root into the infra
VM's ``{ctx}/ctx``, preserving their build-root-relative paths.
Usually a no-op — agent Dockerfiles COPY nothing — so `{ctx}/ctx` stays the
empty context the build otherwise runs against. It exists so a Dockerfile
that pins an input by COPYing a committed file (a checksum list, an npm
lockfile) still finds that file in the VM-side context. Streamed as a tar
so directories and multiple files land in one round trip."""
files = _context_files(dockerfile)
if not files:
return
root = resources.build_root()
rels = [rel for rel, _ in files]
tar = subprocess.Popen(
["tar", "-C", str(root), "-cf", "-", "--", *rels],
stdout=subprocess.PIPE,
)
try:
proc = subprocess.run(
util.ssh_base_argv(private_key, guest_ip) + [f"tar -C {ctx}/ctx -xf -"],
stdin=tar.stdout, capture_output=True, timeout=120, check=False,
)
finally:
if tar.stdout is not None:
tar.stdout.close()
tar.wait()
if tar.returncode != 0:
die(f"packing the agent build context failed (tar exit {tar.returncode})")
if proc.returncode != 0:
die("sending the agent build context to the infra VM failed: "
f"{proc.stderr.decode(errors='replace').strip() or '<no stderr>'}")
def _buildah_build(
private_key: Path,
guest_ip: str,
@@ -41,8 +41,6 @@ from ... import resources
from ...log import die, info
from . import util
ARTIFACT_HTTP_TIMEOUT_SECONDS = 30.0
# Bump if the on-disk artifact *format* changes (compression, layout) so a new
# scheme can't collide with a cached/published artifact of the old one.
_ARTIFACT_FORMAT = "1"
@@ -166,9 +164,7 @@ def _download(url: str, dest: Path) -> None:
"""Stream `url` to `dest` (atomic via a `.part` sibling)."""
tmp = dest.with_suffix(dest.suffix + ".part")
try:
with urllib.request.urlopen(
_open(url), timeout=ARTIFACT_HTTP_TIMEOUT_SECONDS,
) as resp, open(tmp, "wb") as out:
with urllib.request.urlopen(_open(url)) as resp, open(tmp, "wb") as out:
shutil.copyfileobj(resp, out, _CHUNK)
except urllib.error.HTTPError as e:
tmp.unlink(missing_ok=True)
+1 -1
View File
@@ -204,7 +204,7 @@ def boot_vm(
Records the PID."""
if not netpool.tap_present(slot.iface):
die(f"infra link {slot.iface} not present.\n"
f" bot-bottle backend setup --backend=firecracker")
f" ./cli.py backend setup --backend=firecracker")
run_dir.mkdir(parents=True, exist_ok=True)
rootfs = run_dir / "rootfs.ext4"
@@ -108,7 +108,7 @@ def verify_isolation(private_key: Path, guest_ip: str) -> None:
die(f"ISOLATION FAILURE: the VM reached the host canary "
f"{canary_ip}:{canary_port}. The egress boundary is not in "
f"force — refusing to run the agent (fail-closed). Verify the "
f"nft table with: bot-bottle backend setup --backend=firecracker")
f"nft table with: ./cli.py backend setup --backend=firecracker")
if result.returncode == 2:
die("isolation probe inconclusive: the guest has no python3/bash/nc "
"to run the connectivity test. Refusing to continue "
+6 -14
View File
@@ -3,7 +3,7 @@ config renderers (shell command + NixOS module) shown to operators.
The Firecracker backend needs a privileged one-time network setup:
a pool of point-to-point TAP devices (owned by the invoking user, so
`bot-bottle start` never needs root) and a dedicated nftables table that
`./cli.py start` never needs root) and a dedicated nftables table that
isolates every VM. The pool parameters live in exactly one place —
`netpool.defaults.env`, a plain KEY=VALUE file next to this module —
and every consumer reads *that*: this module (below), the shell script
@@ -177,20 +177,14 @@ def gw_slot() -> Slot:
# --- fail-closed verification ---------------------------------------
def _run_ok(argv: list[str]) -> bool:
"""Run a probe command, treating an unavailable binary as failure
"""Run a probe command, treating a missing binary as failure
(rather than crashing) so callers can stay fail-closed."""
try:
return subprocess.run(
argv, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
check=False,
).returncode == 0
except OSError:
# Not only "missing". A name on PATH that isn't executable by this
# user raises PermissionError, and CPython reports that EACCES in
# preference to the ENOENT from the other PATH entries — which is
# how `doctor` came to die with a traceback on a fresh macOS
# account. Any OSError means the probe couldn't run, which for a
# fail-closed check is indistinguishable from "not present".
except FileNotFoundError:
return False
@@ -250,9 +244,7 @@ def overlapping_routes() -> list[RouteConflict]:
["ip", "-json", "route", "show", "table", "all"],
capture_output=True, text=True, check=False,
)
except OSError:
# Missing, or present-but-not-executable for this user; either way
# there are no routes we can enumerate. See _run_ok.
except FileNotFoundError:
return []
if proc.returncode != 0 or not proc.stdout.strip():
return []
@@ -315,11 +307,11 @@ def allocate(slug: str) -> tuple[Slot, IO[str]]:
return s, handle
die(f"Firecracker TAP pool exhausted ({pool_size()} slots, all in "
f"use). Stop a running bottle or raise BOT_BOTTLE_FC_POOL_SIZE "
f"and re-run `bot-bottle backend setup --backend=firecracker`.")
f"and re-run `./cli.py backend setup --backend=firecracker`.")
raise AssertionError("unreachable")
# --- config renderers (shown by `bot-bottle backend setup`) -----------
# --- config renderers (shown by `./cli.py backend setup`) -----------
# The persistent unit is the portable install: the same systemd oneshot
# on every systemd distro (Debian/Ubuntu/Fedora/RHEL/Arch/…).
@@ -34,7 +34,6 @@ from pathlib import Path
from . import infra_artifact, infra_vm, util
_CHUNK = 1 << 20
_REGISTRY_HTTP_TIMEOUT_SECONDS = 30.0
_GZ_NAME = "rootfs.ext4.gz"
_SHA_NAME = "rootfs.ext4.gz.sha256"
@@ -92,9 +91,7 @@ def _put(url: str, body: "bytes | Path", token: str) -> None:
req.add_header("Authorization", f"token {token}")
req.add_header("Content-Type", "application/octet-stream")
try:
with urllib.request.urlopen(
req, timeout=_REGISTRY_HTTP_TIMEOUT_SECONDS,
) as resp:
with urllib.request.urlopen(req) as resp:
print(f" uploaded {url} (HTTP {resp.status})")
except urllib.error.HTTPError as e:
if e.code == 409:
@@ -115,9 +112,7 @@ def _delete(url: str, token: str) -> None:
if token:
req.add_header("Authorization", f"token {token}")
try:
with urllib.request.urlopen(
req, timeout=_REGISTRY_HTTP_TIMEOUT_SECONDS,
):
with urllib.request.urlopen(req):
pass
except urllib.error.HTTPError as e:
if e.code != 404:
@@ -156,10 +151,7 @@ def _try_download_published(role: str, role_dir: Path) -> str | None:
version = _role_version(role)
sha_url = infra_artifact.artifact_url(version, _SHA_NAME, role=role)
try:
with urllib.request.urlopen(
infra_artifact._open(sha_url),
timeout=_REGISTRY_HTTP_TIMEOUT_SECONDS,
):
with urllib.request.urlopen(infra_artifact._open(sha_url)):
pass
except urllib.error.HTTPError as e:
if e.code == 404:
@@ -203,10 +195,7 @@ def _publish_bundle(role: str, role_dir: Path, token: str) -> str:
# present, a re-publish is a no-op. Otherwise clear any partial upload left
# by an interrupted prior attempt and upload the complete set.
try:
with urllib.request.urlopen(
infra_artifact._open(sha_url),
timeout=_REGISTRY_HTTP_TIMEOUT_SECONDS,
) as resp:
with urllib.request.urlopen(infra_artifact._open(sha_url)) as resp:
remote_sha = resp.read().decode("utf-8").split()[0].strip().lower()
except urllib.error.HTTPError as e:
if e.code != 404:
+6 -9
View File
@@ -8,7 +8,7 @@ bundled setup script. `status()` reports what's present, including
whether the pool range collides with an existing route.
Called through `FirecrackerBottleBackend.setup` / `.status`, which the
generic `bot-bottle backend {setup,status}` command dispatches to.
generic `./cli.py backend {setup,status}` command dispatches to.
"""
from __future__ import annotations
@@ -20,7 +20,6 @@ import subprocess
import sys
from pathlib import Path
from ... import invocation
from ... import resources
from . import netpool
from . import util
@@ -35,7 +34,7 @@ _UNIT_PATH = Path("/etc/systemd/system") / netpool.SYSTEMD_UNIT
def _owner() -> str:
# Under `sudo`, USER is root but SUDO_USER is the real invoker — the
# TAPs must be owned by them so `bot-bottle start` stays rootless.
# TAPs must be owned by them so `./cli.py start` stays rootless.
return os.environ.get("SUDO_USER") or os.environ.get("USER") or "youruser"
@@ -162,7 +161,7 @@ def _setup_systemd() -> None:
if rc == 0:
sys.stderr.write(
f"Installed and started {netpool.SYSTEMD_UNIT}. Verify with "
f"`bot-bottle backend status --backend=firecracker`.\n"
f"`./cli.py backend status --backend=firecracker`.\n"
)
else:
sys.stderr.write(
@@ -180,11 +179,9 @@ def _setup_systemd() -> None:
f"sudo systemctl daemon-reload\n"
f"sudo systemctl enable --now {netpool.SYSTEMD_UNIT}\n"
)
# Absolute path, not `sudo bot-bottle`: sudo's secure_path drops
# ~/.local/bin, where both pipx and install.sh put the entry point.
sys.stderr.write(
f"\n(Or re-run this as root to install it directly:\n"
f" {invocation.sudo_command('backend', 'setup', '--backend=firecracker')})\n"
f"\n(Or re-run this as root to install it directly: "
f"sudo ./cli.py backend setup --backend=firecracker)\n"
)
@@ -337,7 +334,7 @@ def status() -> int:
sys.stderr.write(f"range overlap: none (base {netpool.ip_base()})\n")
_report_persistence()
if not ok:
sys.stderr.write("\nRun: bot-bottle backend setup --backend=firecracker\n")
sys.stderr.write("\nRun: ./cli.py backend setup --backend=firecracker\n")
return 0 if ok else 1
+4 -4
View File
@@ -9,7 +9,7 @@ generation.
The privileged network setup (TAP pool + nft table) is a one-time
operator step — see `netpool.py`, `scripts/firecracker-netpool.sh`,
and `bot-bottle backend setup --backend=firecracker`.
and `./cli.py backend setup --backend=firecracker`.
"""
from __future__ import annotations
@@ -132,18 +132,18 @@ def _require_network_pool() -> None:
f"{netpool.pool_size()} slots) overlaps existing routes: "
f"{detail}. This can shadow or be shadowed by that route; "
f"set BOT_BOTTLE_FC_IP_BASE to a free range and re-run "
f"bot-bottle backend setup --backend=firecracker.")
f"./cli.py backend setup --backend=firecracker.")
missing = netpool.missing_taps()
if missing:
die(f"network pool incomplete — missing TAP devices: "
f"{', '.join(missing)}.\n bot-bottle backend setup --backend=firecracker")
f"{', '.join(missing)}.\n ./cli.py backend setup --backend=firecracker")
if shutil.which("nft") is not None and not netpool.nft_table_present():
# nft is queryable and says the table is absent — that's a
# definite, catchable misconfiguration; fail early.
warn(f"isolation table `inet {netpool.NFT_TABLE}` not found via nft. "
"If this is a permissions issue it will be re-checked "
"empirically after boot; otherwise run: "
"bot-bottle backend setup --backend=firecracker")
"./cli.py backend setup --backend=firecracker")
# --- rootfs pipeline (rootless) -------------------------------------
+2 -2
View File
@@ -43,7 +43,7 @@ class Freezer(ABC):
Calls _freeze for the backend-specific snapshot, then writes the
committed image reference to per-bottle state and marks the bottle
preserved so the next `bot-bottle resume` boots from the snapshot.
preserved so the next `./cli.py resume` boots from the snapshot.
Raises CommitCancelled if the user declines an interactive
confirmation prompt (e.g. the macos-container stop prompt).
@@ -51,7 +51,7 @@ class Freezer(ABC):
image_ref = self._freeze(agent)
write_committed_image(agent.slug, image_ref)
mark_preserved(agent.slug)
info(f"to resume from this snapshot: bot-bottle resume {agent.slug}")
info(f"to resume from this snapshot: ./cli.py resume {agent.slug}")
self._export_hint(agent.slug, image_ref)
@abstractmethod
@@ -25,11 +25,3 @@ class MacosContainerBottleCleanupPlan(BottleCleanupPlan):
@property
def empty(self) -> bool:
return not self.containers and not self.networks
def intersect(self, current: BottleCleanupPlan) -> "MacosContainerBottleCleanupPlan":
if not isinstance(current, MacosContainerBottleCleanupPlan):
raise TypeError("cleanup plans must have the same backend type")
return MacosContainerBottleCleanupPlan(
containers=tuple(x for x in self.containers if x in current.containers),
networks=tuple(x for x in self.networks if x in current.networks),
)
@@ -5,7 +5,6 @@ from __future__ import annotations
import subprocess
from .. import EnumerationError
from ..cleanup_control import CleanupFailures
from ...log import info
from . import util as container_mod
from .bottle_cleanup_plan import MacosContainerBottleCleanupPlan
@@ -54,17 +53,19 @@ def prepare_cleanup() -> MacosContainerBottleCleanupPlan:
def cleanup(plan: MacosContainerBottleCleanupPlan) -> None:
failures = CleanupFailures()
for name in plan.containers:
info(f"container delete --force {name}")
failures.run(
subprocess.run(
["container", "delete", "--force", name],
f"deleting container {name}",
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
check=False,
)
for name in plan.networks:
info(f"container network delete {name}")
failures.run(
subprocess.run(
["container", "network", "delete", name],
f"deleting network {name}",
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
check=False,
)
failures.raise_if_any()
+2 -2
View File
@@ -5,7 +5,7 @@ Like Docker, this backend needs no privileged network-pool provisioning
running. `setup()` points at the install/`container system start` steps;
`status()` reports readiness. Reached via
`MacosContainerBottleBackend.setup` / `.status`, dispatched from the
generic `bot-bottle backend {setup,status}`.
generic `./cli.py backend {setup,status}`.
"""
from __future__ import annotations
@@ -75,5 +75,5 @@ def status() -> int:
if not _service_running():
ok = False
if not ok:
sys.stderr.write("\nRun: bot-bottle backend setup --backend=macos-container\n")
sys.stderr.write("\nRun: ./cli.py backend setup --backend=macos-container\n")
return 0 if ok else 1
+2 -9
View File
@@ -689,13 +689,6 @@ def pinned_local_image_ref(ref: str) -> str:
nested-containers layer. A content-derived tag prevents another concurrent
build from moving the provider's ordinary ``:latest`` tag between those
two builds.
Unlike Docker, `container image tag` only accepts ``image-name[:tag]`` as
its source and rejects a bare image ID ("cannot specify 64 byte hex string
as reference"), so the mutable ``ref`` is what gets tagged here. The
post-tag inspection below is what keeps that fail-closed: if ``ref`` moved
between the two commands, the new tag will not resolve to the ID this call
derived its name from.
"""
image = image_id(ref)
digest = image.removeprefix("sha256:")
@@ -708,14 +701,14 @@ def pinned_local_image_ref(ref: str) -> str:
repository = repository[:last_colon]
pinned_ref = f"{repository}:sha256-{digest}"
result = subprocess.run(
[_CONTAINER, "image", "tag", ref, pinned_ref],
[_CONTAINER, "image", "tag", image, pinned_ref],
capture_output=True,
text=True,
check=False,
)
if result.returncode != 0:
die(
f"could not tag exact local image {image} via {ref!r}: "
f"could not tag exact local image {image}: "
f"{(result.stderr or result.stdout or '').strip() or '<no detail>'}"
)
if image_id(pinned_ref) != image:
+2 -13
View File
@@ -63,23 +63,12 @@ def provision_git_gate(
transport.exec(["chmod", "+x", "/etc/git-gate/access-hook"])
creds = _creds_dir(bottle_id)
transport.exec(["mkdir", "-p", creds])
transport.exec(["chmod", "700", creds])
credential_paths: list[str] = []
for u in plan.upstreams:
if u.identity_file:
key_path = f"{creds}/{u.name}-key"
transport.cp_into(u.identity_file, key_path)
credential_paths.append(key_path)
transport.cp_into(u.identity_file, f"{creds}/{u.name}-key")
known_hosts = str(u.known_hosts_file)
if known_hosts and known_hosts != ".":
known_hosts_path = f"{creds}/{u.name}-known_hosts"
transport.cp_into(known_hosts, known_hosts_path)
credential_paths.append(known_hosts_path)
# Copy-mode behavior differs across Docker, Apple Container, and SSH.
# Apply the security contract inside the gateway so every backend produces
# the same private credential namespace.
if credential_paths:
transport.exec(["chmod", "600", *credential_paths])
transport.cp_into(known_hosts, f"{creds}/{u.name}-known_hosts")
# Init the bare repos + per-repo credential config for this namespace.
script = git_gate_render_provision(bottle_id, plan.upstreams)
transport.exec(["sh", "-c", script])
+1 -1
View File
@@ -86,7 +86,7 @@ def _print_vm_install_instructions() -> None:
info("Then start the service: container system start")
else:
info("Install Firecracker: https://github.com/firecracker-microvm/firecracker/releases")
info("Configure the host: bot-bottle backend setup")
info("Configure the host: ./cli.py backend setup")
def _auto_select_backend(prompt: bool = True) -> str:
+3 -3
View File
@@ -1,9 +1,9 @@
"""`backend` CLI command — generic host setup/status across backends.
`bot-bottle backend setup [--backend=NAME]` provisions (or points at how
`./cli.py backend setup [--backend=NAME]` provisions (or points at how
to provision) the chosen backend's one-time host prerequisites.
`bot-bottle backend status [--backend=NAME]` reports readiness.
`bot-bottle backend teardown [--backend=NAME]` undoes setup (uninstall).
`./cli.py backend status [--backend=NAME]` reports readiness.
`./cli.py backend teardown [--backend=NAME]` undoes setup (uninstall).
All dispatch to the backend's `setup()` / `status()` / `teardown()`
classmethods, so there are no backend-specific commands swapping
+6 -13
View File
@@ -1,7 +1,7 @@
"""cleanup: stop and remove all orphaned bot-bottle resources.
Walks every registered backend (docker, firecracker, macos-container)
so a single `bot-bottle cleanup` reaps every backend's leftovers — a
so a single `./cli.py cleanup` reaps every backend's leftovers — a
firecracker bottle's VM processes and run dirs won't survive a
docker-only cleanup pass (issue addressed alongside #77).
@@ -22,7 +22,6 @@ from __future__ import annotations
import sys
from ...backend import get_bottle_backend, has_backend, known_backend_names
from ...backend.cleanup_control import CleanupError
from ...log import info
from ...util import read_tty_line
@@ -55,18 +54,12 @@ def cmd_cleanup(_argv: list[str]) -> int:
# Confirmation authorizes a fresh authoritative snapshot, not blind use of
# identities that may have changed while the operator reviewed the preview.
failures: list[str] = []
for name, backend, displayed in prepared:
current = backend.prepare_cleanup()
approved = displayed.intersect(current)
if approved.empty:
refreshed = [(name, backend, backend.prepare_cleanup())
for name, backend, _plan in prepared]
for name, backend, plan in refreshed:
if plan.empty:
continue
try:
backend.cleanup(approved)
except CleanupError as exc:
failures.append(f"{name}: {exc}")
if failures:
raise CleanupError("cleanup incomplete: " + "; ".join(failures))
backend.cleanup(plan)
info("cleanup: done")
return 0
+2 -2
View File
@@ -4,7 +4,7 @@ Docker bottles are committed to a local Docker image. Macos-container
bottles are exported and rebuilt as a local Apple Container image.
Firecracker bottles stream the guest rootfs out over SSH and rebuild a
local Docker image. The resulting reference is stored in per-bottle
state so the next `bot-bottle resume <slug>` boots from the snapshot
state so the next `./cli.py resume <slug>` boots from the snapshot
instead of rebuilding from the Dockerfile.
"""
@@ -37,7 +37,7 @@ def cmd_commit(argv: list[str]) -> int:
if slug is None:
active = enumerate_active_agents()
if not active:
die("no active bottles; start one with `bot-bottle start`")
die("no active bottles; start one with `./cli.py start`")
choices = [a.slug for a in active]
slug = tui.filter_select(choices, title="Select bottle to commit")
if slug is None:
+1 -1
View File
@@ -8,7 +8,7 @@ override and transcript snapshot under the same state dir.
Use case: an interrupted or preserved bottle needs to be relaunched;
the operator runs
bot-bottle resume <identity>
./cli.py resume <identity>
to bring up the replacement from the recorded state.
"""
+6 -6
View File
@@ -203,19 +203,19 @@ def _start_headless(
if not os.isatty(stdin_fd):
die(
"--headless requires a PTY on stdin; run via:\n"
" script -q /dev/null bot-bottle start ..."
" script -q /dev/null ./cli.py start ..."
)
agent_name = args.name
if not agent_name:
die("--headless requires an agent name: bot-bottle start <agent> --headless")
die("--headless requires an agent name: ./cli.py start <agent> --headless")
manifest.require_agent(agent_name) # raises ManifestError if unknown
prompt = args.prompt
if not prompt:
die(
"--headless requires --prompt: "
"bot-bottle start <agent> --headless --prompt 'Do the thing'"
"./cli.py start <agent> --headless --prompt 'Do the thing'"
)
if args.bottle:
@@ -319,9 +319,9 @@ def attach_agent(
`resume=True` adds `--continue` so claude picks up its most
recent session non-interactively (no session-picker prompt).
First-attach paths (`bot-bottle start`) leave it False.
First-attach paths (`./cli.py start`) leave it False.
Used as the inner step of `bot-bottle start`."""
Used as the inner step of `./cli.py start`."""
runtime = runtime_for(agent_provider_template)
info(
f"attaching interactive {agent_provider_template} session "
@@ -354,7 +354,7 @@ def settle_state(identity: str) -> None:
if not identity:
return
if is_preserved(identity):
info(f"to resume this bottle: bot-bottle resume {identity}")
info(f"to resume this bottle: ./cli.py resume {identity}")
return
cleanup_state(identity)
+1 -1
View File
@@ -110,7 +110,7 @@ def discover_pending() -> list[QueuedProposal]:
def _approval_status(qp: QueuedProposal, verb: str) -> str:
"""Status-line text after a successful approval."""
base = f"{verb} {qp.proposal.tool} for [{qp.label}]"
return f"{base}; resume: bot-bottle resume {qp.label}"
return f"{base}; resume: ./cli.py resume {qp.label}"
def _detail_lines(
+4 -32
View File
@@ -3,7 +3,6 @@
from __future__ import annotations
import http.server
import io
import socket
import threading
import time
@@ -15,10 +14,6 @@ class Readable(Protocol):
def read(self, size: int = -1, /) -> bytes: ...
class Writable(Protocol):
def write(self, data: bytes, /) -> object: ...
@dataclass(frozen=True)
class BodyReadError(Exception):
status: int
@@ -35,25 +30,6 @@ def read_declared_body(
require_length: bool,
) -> bytes:
"""Validate and read exactly one declared body under a read deadline."""
output = io.BytesIO()
copy_declared_body(
stream, output, connection, raw_length, maximum=maximum,
timeout_seconds=timeout_seconds, require_length=require_length,
)
return output.getvalue()
def copy_declared_body(
stream: Readable,
output: Writable,
connection: socket.socket,
raw_length: str | None,
*,
maximum: int,
timeout_seconds: float,
require_length: bool,
) -> int:
"""Copy one declared body to a sink without retaining it in memory."""
if raw_length is None:
if require_length:
raise BodyReadError(411, "Content-Length required")
@@ -68,6 +44,7 @@ def copy_declared_body(
raise BodyReadError(413, "request body too large")
previous_timeout = connection.gettimeout()
deadline = time.monotonic() + timeout_seconds
chunks: list[bytes] = []
remaining = length
try:
while remaining:
@@ -78,13 +55,13 @@ def copy_declared_body(
chunk = stream.read(min(remaining, 64 * 1024))
if not chunk:
raise BodyReadError(400, "incomplete request body")
output.write(chunk)
chunks.append(chunk)
remaining -= len(chunk)
except TimeoutError as exc:
raise BodyReadError(408, "request body read timed out") from exc
finally:
connection.settimeout(previous_timeout)
return length
return b"".join(chunks)
class BoundedThreadingHTTPServer(http.server.ThreadingHTTPServer):
@@ -129,9 +106,4 @@ class BoundedThreadingHTTPServer(http.server.ThreadingHTTPServer):
self._request_slots.release()
__all__ = [
"BodyReadError",
"BoundedThreadingHTTPServer",
"copy_declared_body",
"read_declared_body",
]
__all__ = ["BodyReadError", "BoundedThreadingHTTPServer", "read_declared_body"]
+20 -38
View File
@@ -21,8 +21,6 @@ from __future__ import annotations
import os
import subprocess
import sys
import tempfile
import threading
import typing
from http.server import BaseHTTPRequestHandler
from pathlib import Path
@@ -32,7 +30,7 @@ from bot_bottle.constants import GIT_GATE_TIMEOUT_SECS, IDENTITY_HEADER
from bot_bottle.gateway.bounded_http import (
BodyReadError,
BoundedThreadingHTTPServer,
copy_declared_body,
read_declared_body,
)
from bot_bottle.gateway.policy_resolver import PolicyResolveError, PolicyResolver
@@ -86,8 +84,6 @@ def resolve_sandbox_root(
MAX_BODY_BYTES = 100 * 1024 * 1024
REQUEST_BODY_TIMEOUT_SECONDS = 30.0
MAX_REQUEST_WORKERS = 16
MAX_BODY_WORKERS = 2
_BODY_WORK_SLOTS = threading.BoundedSemaphore(MAX_BODY_WORKERS)
class GitHttpHandler(BaseHTTPRequestHandler):
@@ -195,40 +191,26 @@ class GitHttpHandler(BaseHTTPRequestHandler):
value = self.headers.get(header)
if value:
env[variable] = value
if not _BODY_WORK_SLOTS.acquire(blocking=False):
self.send_error(503, "git request capacity exhausted")
return
try:
with tempfile.TemporaryFile() as body:
try:
copy_declared_body(
self.rfile,
body,
self.connection,
self.headers.get("content-length"),
maximum=MAX_BODY_BYTES,
timeout_seconds=REQUEST_BODY_TIMEOUT_SECONDS,
require_length=False,
)
except BodyReadError as exc:
self.send_error(exc.status, exc.message)
return
body.seek(0)
try:
proc = subprocess.run(
["git", "http-backend"],
stdin=body,
env=env,
capture_output=True,
check=False,
timeout=GIT_GATE_TIMEOUT_SECS,
)
except (OSError, subprocess.SubprocessError) as exc:
self.log_message("git http-backend unavailable: %s", exc)
self.send_error(503, "git backend unavailable")
return
finally:
_BODY_WORK_SLOTS.release()
body = read_declared_body(
self.rfile,
self.connection,
self.headers.get("content-length"),
maximum=MAX_BODY_BYTES,
timeout_seconds=REQUEST_BODY_TIMEOUT_SECONDS,
require_length=False,
)
except BodyReadError as exc:
self.send_error(exc.status, exc.message)
return
proc = subprocess.run(
["git", "http-backend"],
input=body,
env=env,
capture_output=True,
check=False,
timeout=GIT_GATE_TIMEOUT_SECS,
)
self._write_cgi_response(proc.stdout)
def _repo_dir(self, sandbox_root: Path, path: str) -> Path | None:
+1 -1
View File
@@ -385,7 +385,7 @@ PY
;;
esac
echo "git-gate: queued # gitleaks:allow supervisor approval $proposal_id" >&2
echo "git-gate: approve with 'bot-bottle supervise' to continue this push" >&2
echo "git-gate: approve with './cli.py supervise' to continue this push" >&2
waited=0
while [ "$waited" -lt "$timeout" ]; do
status=$(PYTHONPATH="/app${PYTHONPATH:+:$PYTHONPATH}" python3 - "$proposal_id" <<'PY'
-48
View File
@@ -1,48 +0,0 @@
"""How to tell a user to re-run this CLI.
`bot-bottle ` is the right thing to print for anything the user runs as
themselves it is on their PATH, since that is how they got here.
Under `sudo` it is not. sudo replaces PATH with sudoers' `secure_path`
(`/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin` on Debian and
Ubuntu, similar elsewhere), which deliberately excludes user-writable
directories. Both supported install paths put the entry point in one of those:
pipx uses `~/.local/bin`, and `install.sh`'s venv fallback symlinks there too.
So `sudo bot-bottle ` fails with "command not found" for exactly the users who
followed the documented install, while working for anyone who happened to
install system-wide which is why it survives review so easily.
Naming the absolute path sidesteps secure_path entirely.
"""
from __future__ import annotations
import os
import shutil
import sys
def self_path() -> str:
"""Absolute path to this CLI's entry point.
Falls back to the bare name when the entry point cannot be resolved (an
unusual invocation such as `python -m`), because a slightly wrong hint is
better than a traceback while reporting an unrelated problem.
"""
argv0 = sys.argv[0] or "bot-bottle"
resolved = shutil.which(argv0) or argv0
if not os.path.isabs(resolved):
if os.path.exists(resolved):
resolved = os.path.abspath(resolved)
else:
return "bot-bottle"
return resolved
def sudo_command(*args: str) -> str:
"""A copy-pasteable `sudo …` invocation of this CLI.
>>> sudo_command("backend", "setup", "--backend=firecracker")
'sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker'
"""
return " ".join(["sudo", self_path(), *args])
-14
View File
@@ -7,7 +7,6 @@ import asyncio
import math
import sys
from fastapi import FastAPI, HTTPException
from fastapi.exceptions import RequestValidationError
from fastapi.responses import JSONResponse
from pydantic import BaseModel, ConfigDict, StrictStr
from starlette.types import ASGIApp, Message, Receive, Scope, Send
@@ -209,19 +208,6 @@ def create_app(orch: OrchestratorCore, *, signing_key: str) -> FastAPI:
)
app.add_middleware(ControlPlaneBoundary, signing_key=key)
@app.exception_handler(RequestValidationError)
async def invalid_request(
_request: object, exc: RequestValidationError,
) -> JSONResponse:
errors = exc.errors()
location = errors[0].get("loc", ()) if errors else ()
field = str(location[1]) if len(location) > 1 else ""
suffix = f": {field}" if field else ""
return JSONResponse(
{"error": f"invalid request body{suffix}"},
status_code=400,
)
@app.get("/health")
def health() -> dict[str, str]:
return {"status": "ok"}
+7 -1
View File
@@ -366,7 +366,7 @@ class OrchestratorCore:
value with *env_var_secret*, and restores ``_tokens[bottle_id]``.
Returns True on success, False when no stored secrets exist for this
bottle or decryption fails (wrong key / corrupt data)."""
from .store.secret_store import decrypt_value
from .store.secret_store import decrypt_value, encrypt_value, is_legacy_blob
encrypted = self.registry.get_agent_secrets(bottle_id)
if not encrypted:
return False
@@ -377,6 +377,12 @@ class OrchestratorCore:
except ValueError:
return False
self._tokens[bottle_id] = decrypted
if any(is_legacy_blob(value) for value in encrypted.values()):
migrated = {
key: encrypt_value(env_var_secret, value)
for key, value in decrypted.items()
}
self.registry.store_agent_secrets(bottle_id, migrated)
return True
# --- consolidated gateway ----------------------------------------------
@@ -129,10 +129,6 @@ _MIGRATIONS = TableMigrations(
# v5 — index for fast per-bottle lookups and bulk DELETE on teardown.
"CREATE INDEX IF NOT EXISTS idx_bottled_agent_secrets_id "
"ON bottled_agent_secrets (bottled_agent_id, type)",
# v6 — unauthenticated legacy ciphertext must never be selected by
# attacker-controlled blob contents. Existing local agents are
# intentionally reprovisioned instead of retaining downgrade support.
"DELETE FROM bottled_agent_secrets",
],
)
+32 -5
View File
@@ -18,9 +18,9 @@ value is encrypted independently. New output blobs are:
``version || nonce (16 bytes) || ciphertext || tag (32 bytes)``
encoded as URL-safe base64 (no padding). Unversioned legacy ciphertext is
rejected; the registry migration clears those rows rather than allowing blob
contents to select an unauthenticated decoder.
encoded as URL-safe base64 (no padding). The version marker lets the reader
accept legacy ``nonce || ciphertext`` rows long enough to rewrite them in the
authenticated format after a successful reprovision.
keystream_block_i = HMAC-SHA256(key, nonce || i.to_bytes(4, "big"))
ciphertext_i = plaintext_i XOR keystream_block_i[:len(plaintext_i)]
@@ -84,18 +84,44 @@ def encrypt_value(secret_b64: str, plaintext: str) -> str:
return base64.urlsafe_b64encode(authenticated + tag).rstrip(b"=").decode()
def is_legacy_blob(blob_b64: str) -> bool:
"""Whether *blob_b64* uses the pre-authentication storage format."""
try:
return not _b64dec(blob_b64).startswith(_VERSION)
except (ValueError, TypeError):
return False
def _decrypt_legacy(key: bytes, blob: bytes) -> str:
"""Read the original ``nonce || ciphertext`` format for migration only."""
if len(blob) < _NONCE_BYTES:
raise ValueError("ciphertext blob too short")
nonce, ciphertext = blob[:_NONCE_BYTES], blob[_NONCE_BYTES:]
pt = bytearray()
# The legacy format used the byte offset as the PRF counter.
for i in range(0, len(ciphertext), _BLOCK):
chunk = ciphertext[i : i + _BLOCK]
ks = _keystream(key, nonce, i)[: len(chunk)]
pt.extend(c ^ k for c, k in zip(chunk, ks))
try:
return bytes(pt).decode()
except UnicodeDecodeError as exc:
raise ValueError(f"decryption produced non-UTF-8 output: {exc}") from exc
def decrypt_value(secret_b64: str, blob_b64: str) -> str:
"""Decrypt a blob produced by :func:`encrypt_value`.
Returns the original plaintext string. Raises ``ValueError`` for malformed
input, authentication failure, or a key mismatch."""
input, authentication failure, or a key mismatch. Legacy unauthenticated
rows remain readable so callers can migrate them immediately."""
key = _b64dec(secret_b64)
try:
blob = _b64dec(blob_b64)
except (ValueError, TypeError) as exc:
raise ValueError(f"invalid ciphertext blob: {exc}") from exc
if not blob.startswith(_VERSION):
raise ValueError("unsupported ciphertext format")
return _decrypt_legacy(key, blob)
minimum = len(_VERSION) + _NONCE_BYTES + _TAG_BYTES
if len(blob) < minimum:
raise ValueError("ciphertext blob too short")
@@ -126,4 +152,5 @@ __all__ = [
"new_env_var_secret",
"encrypt_value",
"decrypt_value",
"is_legacy_blob",
]
+9 -57
View File
@@ -2,9 +2,7 @@
from __future__ import annotations
import os
import sqlite3
import stat
from contextlib import contextmanager
from pathlib import Path
@@ -21,62 +19,9 @@ class DbStore:
def __init__(self, db_path: Path, migrations: TableMigrations) -> None:
self.db_path = db_path
self._migrations = migrations
self._secure_parent()
if self.db_path.exists():
self._chmod()
def _secure_parent(self) -> None:
"""Create and verify the private parent directory."""
parent = self.db_path.parent
parent.mkdir(mode=0o700, parents=True, exist_ok=True)
if parent.is_symlink():
raise PermissionError(f"database directory must not be a symlink: {parent}")
parent.chmod(0o700)
parent_stat = parent.lstat()
if not stat.S_ISDIR(parent_stat.st_mode):
raise PermissionError(f"database parent is not a directory: {parent}")
if stat.S_IMODE(parent_stat.st_mode) != 0o700:
raise PermissionError(f"database directory is not mode 0700: {parent}")
def _secure_db_file(self) -> None:
"""Create the database without a permissive filesystem window.
SQLite otherwise creates a missing database using the process umask.
This store contains control-plane identity tokens, so both creation and
repair are fail-closed rather than best-effort.
"""
flags = os.O_RDWR | os.O_CREAT | os.O_NOFOLLOW
fd = os.open(
self.db_path,
flags,
stat.S_IRUSR | stat.S_IWUSR,
)
try:
self._secure_open_file(fd)
finally:
os.close(fd)
def _chmod(self) -> None:
"""Enforce and verify the private database mode after every write."""
fd = os.open(self.db_path, os.O_RDWR | os.O_NOFOLLOW)
try:
self._secure_open_file(fd)
finally:
os.close(fd)
def _secure_open_file(self, fd: int) -> None:
"""Pin, validate, and secure an opened database filesystem object."""
file_stat = os.fstat(fd)
if not stat.S_ISREG(file_stat.st_mode):
raise PermissionError(
f"database must be a regular file: {self.db_path}"
)
os.fchmod(fd, stat.S_IRUSR | stat.S_IWUSR)
if stat.S_IMODE(os.fstat(fd).st_mode) != 0o600:
raise PermissionError(f"database is not mode 0600: {self.db_path}")
self.db_path.parent.mkdir(parents=True, exist_ok=True)
def _connect(self) -> sqlite3.Connection:
self._secure_db_file()
conn = sqlite3.connect(self.db_path)
conn.row_factory = sqlite3.Row
return conn
@@ -106,9 +51,16 @@ class DbStore:
return version == len(self._migrations.migrations)
def migrate(self) -> None:
"""Apply any pending migrations to the already-secured DB file."""
"""Apply any pending migrations and set permissions on the DB file."""
with self._connection() as conn:
self._migrations.apply(conn)
self._chmod()
def _chmod(self) -> None:
try:
self.db_path.chmod(0o600)
except OSError:
pass
__all__ = ["DbStore", "DbVersionError"]
+2 -2
View File
@@ -1,4 +1,4 @@
# VHS tape — drives `bot-bottle start demo` interactively and asks
# VHS tape — drives `./cli.py start demo` interactively and asks
# claude (the AI) to run four probes via natural-language prompts.
# Setup (manifest + dummy SSH key + image pre-warm) and teardown
# happen outside the tape; record via `bash scripts/demo-record.sh`,
@@ -29,7 +29,7 @@ Show
# defaults), one git upstream (unreachable on purpose so gitleaks runs
# before the gate would forward), and a FAKE_TOKEN env var shaped like
# a GitHub PAT.
Type "bot-bottle start demo"
Type "./cli.py start demo"
Enter
Sleep 8s
+1 -1
View File
@@ -132,7 +132,7 @@ Each test runs against a temporary `$HOME` and a temporary `$CWD`:
can revisit, but the v1 of this PRD is one file = one bottle.
- **Hot-reload.** Changes to manifest files take effect at next
`bot-bottle start`; we do not watch the directory.
`./cli.py start`; we do not watch the directory.
## Scope
+2 -2
View File
@@ -290,7 +290,7 @@ After this PRD:
### Cleanup CLI
`bot-bottle cleanup` switches from "list every container with prefix
`./cli.py cleanup` switches from "list every container with prefix
`bot-bottle-` and every network with prefix `bot-bottle-net-`
or `bot-bottle-egress-`" to:
@@ -369,7 +369,7 @@ Sized for one PR each, in order.
`docker compose up -d` + attach + teardown. Per-sidecar `start()`/
`stop()` lifecycle methods deleted in the same chunk. Compose-
log dump on teardown added.
4. **Cleanup CLI on compose.** Switch `bot-bottle cleanup` to
4. **Cleanup CLI on compose.** Switch `./cli.py cleanup` to
`docker compose ls`-based discovery; keep prefix-scan as
fallback for one release.
5. **Dashboard.** Decide on the discovery question (open question
+4 -4
View File
@@ -37,7 +37,7 @@ Two rough edges in the current dashboard:
shows only pending proposals. If no agent has called a tool,
the screen reads "no pending proposals" — even when five
bottles are quietly working. The operator has to `docker
compose ls` (or `bot-bottle cleanup -n` to see the y/N preview)
compose ls` (or `./cli.py cleanup -n` to see the y/N preview)
to find out what's actually live.
2. **`e` / `p` re-discover-and-disambiguate every invocation.**
@@ -82,12 +82,12 @@ the "operator wants to make an unprompted change" case.
global across bottles. Filtering ("show me only this agent's
proposals") might be a follow-up but isn't this PRD.
- **Agent lifecycle from the dashboard.** Starting / stopping
agents stays in `bot-bottle start` / `bot-bottle cleanup`. The
agents stays in `./cli.py start` / `./cli.py cleanup`. The
dashboard reads state; it doesn't change it.
- **Preserved-but-not-running bottles.** The active-agents list
is strictly "what's running now" (cross-referenced from
`docker compose ls`). Preserved state dirs without a live
project don't appear — `bot-bottle resume <identity>` is the
project don't appear — `./cli.py resume <identity>` is the
path for those.
- **A separate per-agent detail view.** The agent rows are
one-line summaries. Pressing Enter on a proposal still drops
@@ -125,7 +125,7 @@ the "operator wants to make an unprompted change" case.
- Changes to proposal handling (`a` / `m` / `r` / Enter all
unchanged).
- Changes to the queue-dir / supervise sidecar protocol.
- New CLI surface beyond what's in `bot-bottle dashboard`.
- New CLI surface beyond what's in `./cli.py dashboard`.
- Touching the manifest, compose renderer, launch lifecycle.
## Proposed design
@@ -14,8 +14,8 @@
Today the dashboard is read-only: it surfaces pending proposals
and active agents (PRD 0019) but can't *start* an agent or
*re-enter* one. The operator's path is split — they launch
agents from one terminal (`bot-bottle start <name>`), and watch
them from another (`bot-bottle dashboard`).
agents from one terminal (`./cli.py start <name>`), and watch
them from another (`./cli.py dashboard`).
This PRD collapses that split. The dashboard becomes the
operator's single surface: pressing a key opens an agent picker,
@@ -31,7 +31,7 @@ claude session AND the dashboard process. Exit claude → back to
dashboard, bottle still running. Start another agent → two
bottles up at once. Quit the dashboard → bottles continue
running. Teardown is **always explicit**: the operator presses
`x` on an agent, or runs `bot-bottle cleanup` later.
`x` on an agent, or runs `./cli.py cleanup` later.
## Problem
@@ -45,7 +45,7 @@ Two real frictions today:
open and the dashboard's "active agents" pane is hopelessly
behind reality because they just spawned three in a row.
2. **`bot-bottle start` ties the bottle to a single claude
2. **`./cli.py start` ties the bottle to a single claude
session.** The start command's `ExitStack` brings the bottle
up, runs claude, and tears down on Ctrl-D — fine for a one-
shot session, wrong for "let me bounce in and out of this
@@ -60,7 +60,7 @@ captures full-merged logs per bottle (PRD 0018). It already
## Goals / Success Criteria
1. From inside `bot-bottle dashboard`, pressing `n` (new) opens
1. From inside `./cli.py dashboard`, pressing `n` (new) opens
an agent picker listing every agent defined in the manifest.
Selecting one runs `prepare → preflight → launch`.
2. The preflight Y/N summary renders cleanly — either as a
@@ -83,7 +83,7 @@ captures full-merged logs per bottle (PRD 0018). It already
state cleanup) without quitting the dashboard.
7. Quitting the dashboard (`q`) leaves every running bottle
running. Bottle teardown is always explicit (per-bottle `x`
or `bot-bottle cleanup`). The next `bot-bottle dashboard`
or `./cli.py cleanup`). The next `./cli.py dashboard`
invocation re-discovers them via `list_active_slugs()` and
surfaces re-attach for any it can reconstruct context for
(see "Cross-dashboard re-attach" below).
@@ -94,13 +94,13 @@ captures full-merged logs per bottle (PRD 0018). It already
embedded-emulator option from the research doc is out of
scope. The handoff (option 1) is the v1; option 2 is a
separate PRD if and when handoff is observably insufficient.
- **Adopting bottles started by an out-of-dashboard `bot-bottle
- **Adopting bottles started by an out-of-dashboard `./cli.py
start` invocation.** Those have their own ExitStack-owner and
the dashboard treats them as read-only-watch (already does
today). Re-attach only applies to bottles the *current
dashboard process* started.
- **Resurrecting an out-of-process bottle into a new dashboard
with full re-attach.** A bottle started by `bot-bottle start`
with full re-attach.** A bottle started by `./cli.py start`
in another terminal — or by a previous dashboard run, now
exited — appears in the agents pane (already does, PRD 0019)
and can be re-attached via `docker exec -it claude` because
@@ -109,12 +109,12 @@ captures full-merged logs per bottle (PRD 0018). It already
context object to drive teardown — e.g., the
ExitStack-tracked CA + state cleanup `_settle_state` performs
today. Cross-dashboard re-attach uses the existing
`bot-bottle cleanup` for teardown, not an `x` keypress (see
`./cli.py cleanup` for teardown, not an `x` keypress (see
open questions).
- **Multi-window UI.** Single curses window, two existing
panes (proposals + agents); the agent picker is a modal, not
a third pane.
- **Removing `bot-bottle start`.** Stays as the script-friendly /
- **Removing `./cli.py start`.** Stays as the script-friendly /
legacy entry point. The dashboard is the new default.
## Scope
@@ -140,7 +140,7 @@ captures full-merged logs per bottle (PRD 0018). It already
### Out of scope
- Changes to `bot-bottle start` itself. It keeps its current
- Changes to `./cli.py start` itself. It keeps its current
shape; the dashboard reuses its internal pieces (backend.
prepare / backend.launch) without reaching through the CLI
layer.
@@ -157,7 +157,7 @@ captures full-merged logs per bottle (PRD 0018). It already
Today's flow:
```
bot-bottle start agent
./cli.py start agent
└─ with backend.launch(plan) as bottle: ← bottle alive while inside `with`
bottle.exec_agent([...], tty=True) ← blocks until claude exits
# context exits → compose down → state cleanup
@@ -166,7 +166,7 @@ bot-bottle start agent
The proposed dashboard-driven flow:
```
bot-bottle dashboard
./cli.py dashboard
└─ bottles: dict[str, tuple[ContextManager, DockerBottle]] = {}
# operator presses `n`, picks agent
@@ -205,7 +205,7 @@ Two shifts:
evaluation, state-dir reap) doesn't fire on a quit-while-
running bottle. It DOES fire when the operator explicitly
stops via `x`, because that calls `cm.__exit__`. For
bottles a previous dashboard quit on, `bot-bottle cleanup`
bottles a previous dashboard quit on, `./cli.py cleanup`
is the path — its compose-down + state-reap logic
already covers the case.
@@ -213,7 +213,7 @@ Two shifts:
When the dashboard discovers a bottle in `discover_active_agents`
that it didn't itself start (a previous-dashboard or external
`bot-bottle start` bottle), Enter still attaches via `docker exec
`./cli.py start` bottle), Enter still attaches via `docker exec
-it … claude` — the agent container is running `sleep infinity`
exactly the same way regardless of who started it. The only
thing the current dashboard lacks for those bottles is the
@@ -221,8 +221,8 @@ launch-context object needed to drive a clean teardown via
`x`.
For v1 we surface this honestly: pressing `x` on a non-owned
agent shows a status hint pointing at `bot-bottle cleanup` (or
`bot-bottle cleanup` targeted at the slug if we add that flag
agent shows a status hint pointing at `./cli.py cleanup` (or
`./cli.py cleanup` targeted at the slug if we add that flag
later). The agent stays alive; the operator handles teardown
out-of-band. Enter (re-attach) works for both owned and
non-owned bottles.
@@ -288,7 +288,7 @@ agents pane.
`x` on a non-owned agent (discovered via `list_active_slugs`
but not in `bottles` dict): no-op with status hint pointing
at `bot-bottle cleanup` (the existing path that tears down
at `./cli.py cleanup` (the existing path that tears down
ANY bot-bottle compose project plus reaps state dirs).
### Dashboard quit
@@ -300,7 +300,7 @@ the `docker compose` project keeps running. The next dashboard
invocation discovers the bottles via `list_active_slugs` and
surfaces re-attach.
This is a real departure from today's `bot-bottle start`
This is a real departure from today's `./cli.py start`
semantics (which couples bottle lifetime to the process via
ExitStack). It's intentional: the dashboard is a watching +
acting surface, not a lifetime owner.
@@ -322,7 +322,7 @@ Sized for one PR each.
dashboard's ExitStack; handoff invokes `attach_agent`.
3. **Re-attach via Enter on owned agents-pane row.** Looks up
the slug in the dashboard's `bottles` map; if present →
handoff; else → status-line hint pointing at `bot-bottle
handoff; else → status-line hint pointing at `./cli.py
resume`.
4. **Explicit per-bottle stop (`x` keybinding).** Pop the
bottle's `close` callback off the stack, call it, refresh.
@@ -369,7 +369,7 @@ Sized for one PR each.
bottles dict goes out of scope without invoking `__exit__`,
so the `docker compose` projects keep running. Bottle
teardown is always explicit: per-bottle `x` (for
dashboard-owned), or `bot-bottle cleanup` (for everything).
dashboard-owned), or `./cli.py cleanup` (for everything).
## Open questions
+3 -3
View File
@@ -46,7 +46,7 @@ window, two panes, no terminal handoff.
## Goals / Success Criteria
1. When the operator runs `bot-bottle dashboard` from inside a
1. When the operator runs `./cli.py dashboard` from inside a
tmux session (`$TMUX` set), the dashboard establishes a
two-pane layout: dashboard in the left pane, an initially-
empty right pane reserved for claude sessions.
@@ -313,7 +313,7 @@ Sized small.
3. **Dashboard launched OUTSIDE tmux but tmux is installed.**
Should the dashboard auto-exec itself inside a fresh tmux
session to get the split-pane experience? Convenient but
surprising (`bot-bottle dashboard` shouldn't silently
surprising (`./cli.py dashboard` shouldn't silently
change what session you're in). v1 leaves this off —
operators who want split-pane mode start tmux themselves
and then run the dashboard.
@@ -332,7 +332,7 @@ Sized small.
PRD-0019 focus indicator?
6. **Concurrent dashboards in different tmux windows.**
Multiple `bot-bottle dashboard` invocations in different
Multiple `./cli.py dashboard` invocations in different
tmux windows would each create their own right pane —
probably fine, each has its own state, but worth
verifying that `tmux list-panes` is scoped to the right
+1 -1
View File
@@ -14,7 +14,7 @@ Today bot-bottle is hard-wired around Claude Code assumptions. When Claude runs
## Goals / Success Criteria
- A Codex agent can be started from the dashboard and via `bot-bottle start` alongside a Claude agent.
- A Codex agent can be started from the dashboard and via `./cli.py start` alongside a Claude agent.
- The manifest can express the agent provider/template and, where needed, a custom agent Dockerfile.
- Claude-specific default egress/auth behavior is no longer implicit; provider-specific auth is expressed through explicit bottle egress routes and roles.
- The launcher preserves required infrastructure behavior for sidecars, egress, pipelock, supervisor MCP, CA handling, git, and shell basics.
@@ -30,7 +30,7 @@ across every bottle spin-up. This has several consequences:
to grant that access.
- **Manual rotation burden.** Operators must manage key files on disk, keeping
them secure, rotating them on a schedule, and distributing them across hosts
that run `bot-bottle start`.
that run `./cli.py start`.
## Goals / Success Criteria
@@ -6,7 +6,7 @@
## Summary
The `bot-bottle dashboard` command has grown from its PRD 0013 roots
The `./cli.py dashboard` command has grown from its PRD 0013 roots
(triage supervise proposals) into a parallel-agent control surface
(PRDs 0019/0020/0021): an active-agents pane, agent picker + start,
re-attach, per-bottle stop, tmux split-pane handoff, operator-
@@ -21,7 +21,7 @@ proposals, approve / modify / reject each one, write audit entries,
deliver the response that unblocks the agent's tool call. Everything
that's about *starting / re-entering / stopping* bottles, or about
*operator-initiated* config edits, comes out. The command is renamed
`bot-bottle supervise` so the name matches what it does after the cut.
`./cli.py supervise` so the name matches what it does after the cut.
Future agent-management UX is explicitly punted: if and when a
control surface for parallel agents resurfaces, the working
@@ -41,9 +41,9 @@ Three concrete pains, all downstream of the dashboard's growth:
ExitStack-free bottle ownership are intricate enough that
shipping the next polish increment costs more than it returns.
2. **No clear ownership of "starts and stops bottles".** Today
that responsibility is split: `bot-bottle start` owns one-shot
that responsibility is split: `./cli.py start` owns one-shot
sessions; the dashboard owns multi-session bottles it started
itself; `bot-bottle cleanup` owns everything else. The dashboard
itself; `./cli.py cleanup` owns everything else. The dashboard
tracking its own `bottles: dict[str, (cm, bottle, identity)]`
that doesn't survive a quit is a confusing third lane.
3. **Wrong target shape for a "manage many agents" UI.** The
@@ -97,12 +97,12 @@ problem is everything that got bolted onto that core after.
dashboard. After this PRD they don't exist anywhere — operators
who need ad-hoc edits use the same path the agents do (call the
supervise tool from inside the bottle) or hand-edit the host-
side files and restart the sidecar. Adding a `bot-bottle routes
side files and restart the sidecar. Adding a `./cli.py routes
edit <slug>` verb is a follow-up if the loss bites.
- **Removing `bot-bottle start` or changing its semantics.** Start
- **Removing `./cli.py start` or changing its semantics.** Start
remains the one-shot launch path. PRD 0020's bottle-outlives-
process model is removed; the only path to a long-running
bottle is `bot-bottle start` (foreground) plus `cli.py cleanup`
bottle is `./cli.py start` (foreground) plus `cli.py cleanup`
for teardown.
- **Removing the supervise-sidecar protocol or any of the three
block-remediation engines.** PRDs 00130016 stay Active. The
@@ -122,8 +122,8 @@ problem is everything that got bolted onto that core after.
### In scope
- **Rename the subcommand.** `bot-bottle dashboard` becomes
`bot-bottle supervise`. The module moves from `bot_bottle/cli/
- **Rename the subcommand.** `./cli.py dashboard` becomes
`./cli.py supervise`. The module moves from `bot_bottle/cli/
dashboard.py` to `bot_bottle/cli/supervise.py`. The dispatcher
in `bot_bottle/cli/__init__.py` and the help text both update.
- **Strip the curses loop to proposal-only.** The remaining
@@ -167,7 +167,7 @@ problem is everything that got bolted onto that core after.
- Any new feature in the supervise TUI. The cut is purely
subtractive (except for the rename).
- Behavior changes in `bot-bottle start`, `cli.py cleanup`,
- Behavior changes in `./cli.py start`, `cli.py cleanup`,
`cli.py resume`, `cli.py list`, `cli.py info`, `cli.py edit`,
`cli.py init` — unchanged.
- Changes to the supervise sidecar (`supervise_server.py`,
@@ -181,7 +181,7 @@ problem is everything that got bolted onto that core after.
### Final shape of the TUI
After this PRD the `bot-bottle supervise` curses surface is:
After this PRD the `./cli.py supervise` curses surface is:
```
bot-bottle supervise (3 pending)
@@ -307,8 +307,8 @@ The PR closes issue #174.
1. **`e` / `p` operator-initiated edits — gone for good or
moved to a separate CLI verb?** The PRD removes them with no
replacement. The simplest replacement is `bot-bottle routes
edit <slug>` and `bot-bottle pipelock edit <slug>`, sharing
replacement. The simplest replacement is `./cli.py routes
edit <slug>` and `./cli.py pipelock edit <slug>`, sharing
the existing `apply_routes_change` / `apply_allowlist_change`
engines. If the loss is felt within the first parallel
run after this lands, that follow-up is a small PR. Leaving
+9 -9
View File
@@ -7,12 +7,12 @@
## Summary
When `bot-bottle start` is run without an agent name, or without a backend
When `./cli.py start` is run without an agent name, or without a backend
explicitly specified, the user currently gets an argparse error (missing
positional) or falls through to the `docker` default silently. This PRD
adds a terminal UI that appears in those gaps: a filter-select screen
built with `curses` that lets the operator pick the agent and/or backend
interactively rather than memorising names or consulting `bot-bottle list`.
interactively rather than memorising names or consulting `./cli.py list`.
## Problem
@@ -29,15 +29,15 @@ visible.
## Goals / Success Criteria
1. `bot-bottle start` (no arguments) shows an interactive agent selector;
1. `./cli.py start` (no arguments) shows an interactive agent selector;
the selected name is used exactly as if it had been passed on the
command line.
2. `bot-bottle start <name>` (no `--backend`, no `BOT_BOTTLE_BACKEND`)
2. `./cli.py start <name>` (no `--backend`, no `BOT_BOTTLE_BACKEND`)
shows an interactive backend selector; the selected backend is used
exactly as if `--backend=<selected>` had been passed.
3. `bot-bottle start <name> --backend=<b>` (both explicit) shows neither
3. `./cli.py start <name> --backend=<b>` (both explicit) shows neither
screen — no behavioural change from today.
4. `bot-bottle start` (no arguments, no env backend) shows the agent
4. `./cli.py start` (no arguments, no env backend) shows the agent
selector first, then the backend selector.
5. The filter-select widget is a standalone utility
(`bot_bottle/cli/tui.py`) shared by both selectors.
@@ -57,7 +57,7 @@ visible.
- No pagination beyond what fits in the terminal window (scroll via
cursor movement is sufficient for typical agent counts).
- No multi-select; exactly one item is chosen per invocation.
- No changes to `bot-bottle resume`, `bot-bottle list`, or any other
- No changes to `./cli.py resume`, `./cli.py list`, or any other
subcommand.
## Design
@@ -83,7 +83,7 @@ def filter_select(
The widget renders to the tty file descriptor opened via `curses.initscr`
(or `curses.newterm` on the tty fd so stdout remains clean for callers
that pipe `bot-bottle`).
that pipe `./cli.py`).
Layout (full-width, minimal):
@@ -140,7 +140,7 @@ agent picker can populate itself from the real manifest. The same
`filter_select` opens `/dev/tty` and feeds it as the input file to
`curses.wrapper`-equivalent code (using `curses.newterm` to avoid
clobbering the caller's stdout/stderr). This keeps the picker
composable — callers can pipe `bot-bottle` output without the curses
composable — callers can pipe `./cli.py` output without the curses
draw sequences contaminating the pipe.
## Implementation chunks
+3 -3
View File
@@ -30,12 +30,12 @@ snapshot before a planned host reboot or hardware migration.
## Goals / Success Criteria
- `bot-bottle commit [<slug>]` takes a snapshot of the running agent and
- `./cli.py commit [<slug>]` takes a snapshot of the running agent and
stores it as a local artifact.
- Without a slug argument the command shows the same interactive picker
as `start` (the list of active slugs).
- The committed artifact reference is stored in per-bottle state so
that the next `bot-bottle resume <slug>` automatically uses the
that the next `./cli.py resume <slug>` automatically uses the
snapshot instead of rebuilding from the Dockerfile.
- `mark_preserved` is called so the state dir survives the normal
session-end cleanup.
@@ -81,7 +81,7 @@ to the committed `.smolmachine` artifact.
### `commit` command
```
bot-bottle commit [<slug>]
./cli.py commit [<slug>]
```
1. Resolve slug (arg or interactive picker from `enumerate_active_agents`).
@@ -37,7 +37,7 @@ egress policy), the operator must duplicate the agent file and change the
selection order, as the effective bottle for the session.
5. Confirming with an empty selection falls back to the agent's `bottle:` field.
If neither is set, a ManifestError is raised pointing the operator at the fix.
6. The ordered bottle list is stored in launch metadata so `bot-bottle resume`
6. The ordered bottle list is stored in launch metadata so `./cli.py resume`
uses the same bottles.
7. The preflight summary (`y/N` screen) shows the effective bottle name(s).
8. The multi-select picker supports incremental filtering, Space/Enter to toggle
@@ -52,7 +52,7 @@ egress policy), the operator must duplicate the agent file and change the
- Reordering the selection list from within the picker (order = insertion order;
drag-and-drop is out of scope).
- Storing bottle selection history / MRU.
- Changes to `bot-bottle edit`, `bot-bottle list`, or `bot-bottle info`.
- Changes to `./cli.py edit`, `./cli.py list`, or `./cli.py info`.
- Removing the `bottle:` key from the agent schema (it stays, now optional).
## Design
+1 -1
View File
@@ -74,7 +74,7 @@ macOS-only for v1. Three concrete blockers:
## Goals / Success Criteria
- `BOT_BOTTLE_BACKEND=smolmachines bot-bottle start <agent>` launches,
- `BOT_BOTTLE_BACKEND=smolmachines ./cli.py start <agent>` launches,
runs, and tears down a bottle on a Linux host with `/dev/kvm`.
- The TSI allowlist is enforced on Linux: PRD 0022's
`tests/integration/test_sandbox_escape.py` passes against
+2 -2
View File
@@ -39,7 +39,7 @@ end-to-end runner that would catch the *next* macOS-only launch regression.)
PR #470 (#414) already made the integration suite backend-agnostic:
`skip_unless_selected_backend_available()` gates on the *selected* backend's
own `is_backend_ready()` rather than `docker_available()`, and each
integration job runs `bot-bottle backend status --backend=<name>` as a preflight
integration job runs `./cli.py backend status --backend=<name>` as a preflight
that fails loudly when the backend is missing. That is the machinery this job
plugs into; this PRD supplies the runner and the job.
@@ -98,7 +98,7 @@ Modeled on `integration-firecracker`:
- `concurrency: { group: integration-macos-infra, cancel-in-progress: false }`
to serialize runs against the singleton.
- **Preflight**`command -v container`, `container system status`, then
`bot-bottle backend status --backend=macos-container`; any failure exits
`./cli.py backend status --backend=macos-container`; any failure exits
non-zero so a misprovisioned runner fails loudly instead of silently
skipping.
- Run the integration suite under coverage with
+1 -1
View File
@@ -16,7 +16,7 @@ verifies host prerequisites after install.
## Problem
There is currently no install path for new users. The only way to run
bot-bottle is to clone the repo and invoke `bot-bottle`. This blocks any
bot-bottle is to clone the repo and invoke `./cli.py`. This blocks any
public demo: readers want `curl | sh` or `pipx install`, not a manual
clone-and-configure flow. There is also no single command that tells a
user whether their host is actually ready to run a bottle.
+8 -293
View File
@@ -32,17 +32,9 @@ not a principled scope exclusion: both are major hosted sandbox platforms and
belong in this landscape even though they target platform builders rather than
bot-bottle's local single-operator workflow.
Updated 2026-07-27 after a scan of recent Show HN launches: **Black LLAB,
Eve, CloudRouter, Nucleus, yolo-cage, and Sandbox Agent SDK** added as a
dated entrant cohort. They sharpen the comparison on three axes the original
table underweighted: the browser/preview loop, parallel-agent operator UX, and
a provider-neutral automation/session API.
## Summary
The main table compares bot-bottle against fifteen canonical
isolation/sandbox tools; a later section evaluates six recent HN entrants
without widening an already unwieldy table.
The main table compares bot-bottle against fifteen isolation/sandbox tools.
Governance/pre-action authorization and credential-only layers are covered
separately because they don't provide VM or container isolation. None
duplicate bot-bottle's combination of local
@@ -550,199 +542,6 @@ them.
framework runtime is not compromised.
- **Maturity**: Specification + reference implementation, 2026.
## Recent HN entrants (added 2026-07-27)
These are grouped by launch date rather than promoted into the main table.
Several are young or sparsely documented, and putting them beside mature
runtime platforms with false precision would obscure the useful comparison.
The HN launch posts are the evidence snapshot; feature claims should be
rechecked against their repositories before relying on them for a security
decision.
### Black LLAB
- **Source**: https://github.com/isaacdear/black-llab ;
HN launch https://news.ycombinator.com/item?id=47402394
- **Isolation/locality**: Local Docker environment, with an isolated container
created for each agent task. Shared host kernel; no stronger boundary is
claimed.
- **Agent integration**: General local/cloud model workspace. Its headline is
dynamic routing of simple prompts to local models and complex prompts to
hosted models, with code execution and web scraping inside the task
container.
- **Network/credentials**: No default-deny egress, payload inspection, or
host-side credential injection documented in the launch.
- **Competitive read**: Superficial overlap ("a container per agent task"),
but not a direct security-policy competitor. Its useful challenge is the
integrated model-selection UX, which bot-bottle intentionally leaves to the
selected agent provider.
- **Maturity**: Early solo project; HN launch received 1 point.
### Eve
- **Source**: https://eve.new/ ;
HN launch https://news.ycombinator.com/item?id=47721255
- **Isolation/locality**: Managed, hosted Linux sandbox per user/session
(claimed 2 vCPU, 4 GB RAM, 10 GB disk), with filesystem, code execution,
headless Chromium, and service connectors.
- **Agent integration**: End-user OpenClaw-style agent product. An orchestrator
routes subtasks to specialist models and can run parallel subagents that
coordinate through a shared filesystem. Web UI and iMessage are primary
interaction surfaces.
- **Network/credentials**: Broad connectors are a product feature; the launch
does not document bot-bottle-style default-deny route policy, content DLP,
or credentials held outside the sandbox.
- **Competitive read**: Adjacent, not direct. Eve sells a managed colleague;
bot-bottle lets an operator run existing coding-agent CLIs under local
containment. Eve nevertheless demonstrates the appeal of background work,
live progress, browser capability, and mobile notification.
- **Maturity**: Commercial hosted product; HN launch received 71 points and
39 comments.
### CloudRouter
- **Source**: https://github.com/manaflow-ai/manaflow/tree/main/packages/cloudrouter ;
HN launch https://news.ycombinator.com/item?id=47006393
- **Isolation/locality**: Claude Code or Codex runs locally and provisions
remote cloud VMs/GPUs for execution. Project files are uploaded to the VM;
each machine exposes auth-protected VNC, VS Code, and Jupyter surfaces.
- **Agent integration**: A skill plus CLI lets the coding agent itself start,
command, inspect, and tear down machines. Browser automation is integrated,
including snapshots and screenshots. Parallel disposable compute is the
central workflow.
- **Network/credentials**: The launch emphasizes remote resource isolation and
authenticated UI endpoints, not default-deny guest egress, payload DLP, or
proxy-held application credentials.
- **Competitive read**: The closest recent workflow competitor. It directly
addresses parallel coding agents, environmental conflict, and closing the
browser/test loop, but trades local custody for elastic cloud compute.
Cloud VMs and GPUs could be a future bot-bottle backend; they do not replace
its manifest/policy layer.
- **Maturity**: Active open-source monorepo project; HN launch received
138 points and 36 comments.
### Nucleus
- **Source**: https://github.com/coproduct-opensource/nucleus ;
HN launch https://news.ycombinator.com/item?id=46855770
- **Isolation/locality**: Firecracker microVM with an enforcing MCP tool proxy.
- **Agent integration/config**: Compositional permission envelope for
read/write/run actions. The envelope is non-escalating and can tighten or
terminate, with scoped approval tokens for gated operations.
- **Network/credentials**: Default-deny egress, DNS allowlist, iptables drift
detection, time/budget caps, and hash-chained audit logging are claimed.
Remote append-only audit storage and attestation were roadmap items at
launch.
- **Competitive read**: Direct on security architecture, especially
non-escalating policy and tamper-evident audit. It is an early execution/tool
proxy rather than a provider-neutral, one-command coding-agent product. Its
tool-level action envelope is semantically finer than bot-bottle's network
boundary; bot-bottle is stronger on turnkey agent/provider integration,
credential custody, Git mediation, and long-running operator workflow.
- **Maturity**: Early OSS experiment; HN launch received 3 points.
### yolo-cage
- **Source**: https://github.com/borenstein/yolo-cage ;
HN launch https://news.ycombinator.com/item?id=46706796
- **Isolation/locality**: Local sandbox for running multiple coding agents in
YOLO mode. The launch discussion describes a VM boundary.
- **Agent integration**: Built around the native Claude Code experience and
motivated by running many agents in parallel without permission-prompt
fatigue.
- **Network/Git/credentials**: Strict egress filtering, configurable HTTP
middleware, and mediated `git`/`gh` dispatch are the main value. The launch
discussion explicitly identifies provider credential handling as unfinished
and difficult because Claude state spans multiple host paths.
- **Competitive read**: The closest new threat-model competitor. It shares
bot-bottle's premise that filesystem isolation alone is insufficient and
that Git plus authorized HTTP channels need mediation. bot-bottle currently
leads on cross-provider support, proxy-held Claude/Codex/forge credentials,
typed per-role manifests, content DLP, and supervision. yolo-cage's simpler
pitch and narrower Claude-first setup may be easier to explain.
- **Maturity**: Early local tool; HN launch received 60 points and 76 comments.
### Sandbox Agent SDK
- **Source**: https://github.com/rivet-dev/sandbox-agent ;
HN launch https://news.ycombinator.com/item?id=46795584
- **Isolation/locality**: Does not provide the isolation primitive. It runs
inside E2B, Daytona, Modal, Cloudflare Containers, Agent Computer, BoxLite,
Docker, or another sandbox provider. Embedded mode can also run locally
without a sandbox.
- **Agent integration**: Provider-neutral Rust server/SDK exposing a common
HTTP/SSE/OpenAPI interface across Claude Code, Codex, OpenCode, Cursor, Amp,
and Pi, plus a universal event/session schema for external storage and
replay. It also exposes filesystem, managed-process, terminal, MCP, skills,
custom-tool, and computer-use APIs. TypeScript is the primary SDK surface.
- **Network/credentials**: Delegated to the chosen sandbox provider.
- **Credential posture**: Its documented convenience command extracts real
OpenAI/Anthropic credentials from local agent configuration and passes them
as environment variables into the sandbox. That is materially weaker than
bot-bottle's host-side credential custody, but it is an integration choice,
not a structural limitation: a sandbox provider could put a credential
proxy underneath the same SDK.
- **Competitive read**: A serious architectural threat despite not supplying
isolation. Sandbox Agent is trying to standardize the boundary *above* the
sandbox: one client protocol, session model, and UI/control surface across
every coding agent and runtime. If that boundary becomes the ecosystem
standard, users and application builders may choose a sandbox provider plus
Sandbox Agent rather than a vertically integrated launcher. bot-bottle's
manifests would then be valuable chiefly as a local policy/backend
implementation unless they expose an equally usable control contract.
- **Maturity**: Apache 2.0, ~1.5k stars and 426 commits at the 2026-07-27
check; HN launch received 41 points.
#### Why the Sandbox Agent architecture is strategically different
The manifest and the universal control protocol solve different layers:
- A bot-bottle manifest is a **trusted launch-time policy composition**. It
selects the agent role, isolation backend, image, skills, egress routes,
credentials, Git mediation, and supervision policy. Crucially, identity and
secret references live on the host side of the trust boundary.
- Sandbox Agent is a **runtime control and observation protocol**. A remote
client creates sessions, sends messages, handles permissions, configures
skills/MCP, manipulates files/processes/desktops, and streams normalized
events. It deliberately delegates sandbox lifecycle, Git management,
storage, network policy, and credential security to other products.
That makes it complementary in a component diagram but competitive in product
architecture. The layer that becomes the stable integration point tends to own
the ecosystem. Three plausible threat paths matter:
1. **Standard control plane, interchangeable runtimes.** Applications integrate
once with Sandbox Agent and treat E2B, Daytona, BoxLite, Docker, or a future
local microVM as replaceable compute. A provider that bundles adequate
egress and credential custody makes bot-bottle's end-to-end launcher less
necessary.
2. **Policy grows upward.** Sandbox Agent already configures permissions,
skills, MCP, custom tools, filesystem/process access, and computer use. If
it adds a declarative, host-verifiable policy document, the overlap with
agent/bottle manifests becomes substantial even if enforcement remains
delegated.
3. **UI and session ownership.** Its universal transcript schema, Inspector,
React components, event replay, and remote terminal/computer APIs can become
the natural basis for desktop, web, and mobile agent managers. bot-bottle's
security layer could remain stronger while losing the operator surface and
distribution channel.
The counter-position is not to claim that manifests and an API are mutually
exclusive. The defensible split is:
- bot-bottle owns the trusted policy and enforcement plane outside the agent;
- a provider-neutral protocol owns agent process control and normalized
events; and
- the operator UI consumes both.
This suggests an explicit compatibility decision rather than parallel,
accidental protocol design: evaluate running Sandbox Agent inside a bottle and
exposing it only through the authenticated bot-bottle control plane. If its
schema is suitable, adopting it could turn a threat into an integration while
keeping manifests as the higher-trust policy source. If it is unsuitable,
bot-bottle should still publish a stable provider-neutral session/event API so
frontends do not depend on Claude/Codex/Pi-specific process behavior.
## Comparison table
*Isolation/sandbox tools only. AGT and OAP are governance layers — see their per-project notes above.*
@@ -817,70 +616,6 @@ would be a *backend* bot-bottle could call, not a competitor to its
manifest layer. endo-familiar is in a different paradigm entirely:
capability passing rather than kernel boundaries.
**Recent entrants change two parts of this read.** yolo-cage is closer to the
actual threat model than agent-safehouse or litterbox: it combines a VM-style
boundary with mediated Git and filtered HTTP specifically for parallel coding
agents. Sandbox Agent SDK is the more important strategic entrant even though
it supplies no isolation. It can become the standard agent-control layer above
all of these runtimes, including a future bot-bottle backend. CloudRouter is
the clearest workflow challenge because its browser/desktop/GPU loop makes
parallel agents visibly more capable, not merely safer.
## Gap evaluation after the 2026-07-27 entrant scan
### Material gaps
1. **A stable provider-neutral control and event protocol.** This is the
largest newly visible gap. bot-bottle normalizes launch/provisioning across
providers, but an external UI or orchestrator still lacks one documented
contract for creating a Claude/Codex/Pi session, sending input, handling
permission/supervision events, streaming normalized output, reconnecting,
and replaying history. Sandbox Agent SDK addresses exactly this layer and
is already portable across many sandbox providers.
2. **Browser/preview closure.** CloudRouter and Eve make a browser or desktop
part of the standard agent environment and expose screenshots/live viewing
to the operator. bot-bottle can run dev servers and supports nested
containers, but it does not present a first-class browser/computer-use
primitive or an auth-protected preview surface. For coding agents expected
to verify UI work, this is a real product gap.
3. **Unified parallel-session operator UX.** Named persistent bottles and
supervision provide the substrate, but the recent products make task
switching, live progress, notifications, terminal attach, diffs, and
session history the product. Security depth will not compensate for a
visibly rougher daily loop.
4. **Normalized transcript persistence and replay.** bot-bottle preserves
provider-specific state for resume; it does not expose a provider-neutral
event record suitable for audit, replay, analytics, or a web/mobile client.
This is both a UX gap and an audit gap.
### Important, but not necessarily bot-bottle features
- **Cloud VM/GPU provisioning.** Valuable for elastic workloads and could be a
backend, but it conflicts with the local-custody default and should not
displace core policy work.
- **Automatic model routing.** Black LLAB and Eve sell task-to-model routing.
bot-bottle's provider-template boundary can host that choice without making
it part of the trusted sandbox policy.
- **A thousand SaaS connectors.** This broadens capability and blast radius.
The bot-bottle-native answer should remain explicit, scoped forge/egress
associations rather than connector count as a goal.
- **SDK-driven sandbox lifecycle as the primary configuration model.** Useful
for platform builders, but not a replacement for reviewable, host-owned
manifests. A control API and a declarative policy source are compatible;
neither should silently become the other.
### Areas where bot-bottle remains ahead
- real provider and forge credentials remain outside the agent process rather
than being extracted into its environment;
- authorized HTTP payloads are scanned, not merely destination-filtered;
- Git writes traverse a distinct gate with secret scanning and host-held
upstream credentials;
- role policy is host-owned, composable, and separate from untrusted repo
content; and
- local Firecracker/Apple Container execution preserves operator custody
without requiring a hosted sandbox platform.
## Borrowable ideas
### Already shipped or otherwise addressed
@@ -907,19 +642,6 @@ parallel agents visibly more capable, not merely safer.
### Still worth considering
- **Sandbox Agent compatibility or an equivalent stable protocol (highest
priority):** spike running its server inside a bottle behind bot-bottle's
authenticated control plane. Compare its session/event schema, permission
model, restore semantics, and provider coverage with current provider
adapters. Adopt compatibility if it preserves the host-owned trust boundary;
otherwise specify bot-bottle's own stable API before building another UI.
- **First-class browser/preview loop** (from CloudRouter and Eve): give a
bottle an optional browser/computer-use capability plus an operator-visible,
authenticated preview/screenshot surface. Treat its network access as part
of the bottle policy, not an implicit bypass.
- **Provider-neutral transcript/event persistence** (from Sandbox Agent SDK):
retain enough normalized structure for replay and audit while preserving the
provider-native state needed for exact resume.
- **Live network activity in the supervisor TUI** (from Docker sbx): show
allowed and blocked connections and let the operator propose policy changes
from the existing supervision surface.
@@ -930,11 +652,10 @@ parallel agents visibly more capable, not merely safer.
closer review. This needs a carefully specified trust model before it can be
more than a heuristic.
Not worth borrowing: SDK-first *policy configuration* as used by boxlite /
microsandbox (cuts against the reviewable declarative-manifest stance), and
the hosted-SaaS custody model of tilde.run (cuts against the "infrastructure I
control" goal). A provider-neutral runtime-control API is a separate concern
and is worth borrowing.
Not worth borrowing: the SDK-first programmatic API style of boxlite /
microsandbox (cuts against the declarative-manifest stance), and the
hosted-SaaS dashboard model of tilde.run (cuts against the
"infrastructure I control" goal).
## Publishing and positioning verdict
@@ -958,15 +679,9 @@ bot-bottle remains unusual in combining:
The practical wedge is “as easy as native yolo, with declarative role policy
and self-hosted custody,” including scoped access to private LAN/Tailnet
services that cloud-first runtimes cannot provide without additional network
plumbing. The main competitive risks are now:
- a local wrapper such as yolo-cage, claudebox, or Docker sbx growing a
role-manifest and credential-custody layer;
- Sandbox Agent SDK becoming the standard control/session boundary and making
the runtime beneath it interchangeable; and
- GUI products such as SuperHQ or CloudRouter adding equivalent policy and
audit depth before bot-bottle closes the browser/preview and
parallel-session UX gaps.
plumbing. The main competitive risks are a local wrapper such as claudebox or
Docker sbx growing a role-manifest layer, and GUI products such as SuperHQ
adding equivalent policy and audit depth.
## Caveats
+6 -6
View File
@@ -6,7 +6,7 @@ Can bot-bottle grow a built-in supervisor — TUI inventory plus PR-feedback rou
## Context
bot-bottle today is a fleet *executor*: `bot-bottle start <agent>` brings up one bottle (agent container + pipelock + optional git-gate + optional cred-proxy on a per-bottle internal network), and `cli.py` tears it down when the session ends. There is no inventory view, no idle-detection, no automated reaction to PR or CI events. In parallel use, a human is the supervisor — opening one terminal per bottle, switching between them, and watching upstream PR state by hand.
bot-bottle today is a fleet *executor*: `./cli.py start <agent>` brings up one bottle (agent container + pipelock + optional git-gate + optional cred-proxy on a per-bottle internal network), and `cli.py` tears it down when the session ends. There is no inventory view, no idle-detection, no automated reaction to PR or CI events. In parallel use, a human is the supervisor — opening one terminal per bottle, switching between them, and watching upstream PR state by hand.
A separate survey of the broader ecosystem ([agent control dashboards research, mid-2026](https://gitea.dideric.is/didericis/consilium-research/src/branch/main/developer-workflow/agent-control-dashboards-2026-05-24.md)) sorts dashboards into five tiers (session managers, parallel runners, Kanban boards, mission-control SPAs, observability backends). The earlier first-pass conclusion was that a full SPA tier conflicts with bot-bottle's isolation model. This doc reconsiders the smaller question: a TUI supervisor in the existing Python CLI.
@@ -25,13 +25,13 @@ A supervisor doesn't have to be heavy. A TUI built into the existing Python CLI,
Three layers, each independently useful, in order of ambition:
### 1. `bot-bottle status` — read-only inventory
### 1. `./cli.py status` — read-only inventory
Reads `docker ps` filtered by a bottle label and tails each bottle's session log. Reports per bottle: name, agent, uptime, last-activity timestamp, token spend if available, associated PR/branch if recorded.
No new daemons. No new ports. No new credentials. ~100 lines.
### 2. `bot-bottle watch` — TUI over the same data
### 2. `./cli.py watch` — TUI over the same data
Same data as `status`, rendered with auto-refresh and keyboard shortcuts that shell out to the existing `cli.py attach / stop / start` commands.
@@ -39,7 +39,7 @@ Library choice: prefer the stdlib `curses` module to stay stdlib-first; fall bac
This is the Claude Squad / tmux-agent-status pattern, applied to bottles instead of tmux sessions. The whole category exists *because* a TUI is the lightweight shape that doesn't require what the SPA tier requires.
### 3. `bot-bottle supervise` — PR feedback router
### 3. `./cli.py supervise` — PR feedback router
The optional, more ambitious layer. The bottle manifest gains an optional field:
@@ -49,7 +49,7 @@ pr_watch:
branch: agent/task-42
```
`bot-bottle supervise` polls the named upstream for new review comments and CI failures on `branch`. When one fires, it surfaces as a desktop notification or a flash in the TUI. The human decides what to do with the feedback — there is no autonomous loop that feeds the comment back into a bottle's next prompt (see "Where to be conservative" for why).
`./cli.py supervise` polls the named upstream for new review comments and CI failures on `branch`. When one fires, it surfaces as a desktop notification or a flash in the TUI. The human decides what to do with the feedback — there is no autonomous loop that feeds the comment back into a bottle's next prompt (see "Where to be conservative" for why).
The polling token is a **host** token (the same `GH_PAT` / Gitea token the host already keeps in shell env), not a bottle credential. The supervisor never holds bottle secrets.
@@ -62,7 +62,7 @@ The load-bearing question is whether the supervisor introduces the privileged-ch
| Reaching into running bottles | Supervisor reads `docker ps` and host-side log files. The host already sees both — Docker is the trust boundary, the supervisor is on the host side of it. |
| Holding bottle credentials | The polling token is a host token. The supervisor never receives `bottle.cred_proxy.routes` entries; it has no path to them. |
| Bridging between bottles | The supervisor does not relay state from bottle A to bottle B. It relays *upstream PR state* to a bottle's next prompt — and only if the manifest opts in. |
| New attack surface | All "control" actions go through `bot-bottle start <agent>`, which already enforces the manifest. The supervisor is an automated caller of the existing CLI, not a parallel control plane. |
| New attack surface | All "control" actions go through `./cli.py start <agent>`, which already enforces the manifest. The supervisor is an automated caller of the existing CLI, not a parallel control plane. |
The boundary stays at the bottle wall. The supervisor looks outward at git/PR state and downward at Docker; it does not look *inward* through pipelock.
@@ -13,11 +13,11 @@ What's the cheapest path to that, and where does it bottom out?
## What "interact" means
Today the flow is bimodal. `bot-bottle start <agent>` brings the
Today the flow is bimodal. `./cli.py start <agent>` brings the
bottle up and immediately drops you into an interactive
`docker exec -it bot-bottle-<slug> claude ...` — claude-code
owns the whole terminal until you Ctrl-D out, at which point the
bottle tears down. The dashboard (`bot-bottle dashboard`) is a
bottle tears down. The dashboard (`./cli.py dashboard`) is a
*separate* invocation that watches across bottles but never
exposes the claude TUI itself.
@@ -107,7 +107,7 @@ What's not good:
This is the v1 the project's existing code-shape strongly
prefers. It clears the bar of "let me talk to claude-code
without quitting `bot-bottle dashboard`."
without quitting `./cli.py dashboard`."
## Option 2: Embedded emulator
+2 -2
View File
@@ -84,7 +84,7 @@ dangerous changes before they left*.
(`supervise_gitleaks_allow`, [`git_gate_render.py`](../../bot_bottle/git_gate_render.py)).
Extend the same flow to **high-risk file classes**: any commit touching
CI/build/deploy scripts, auth/crypto code, egress config, or
adding/changing dependencies → route to `bot-bottle supervise`. This is
adding/changing dependencies → route to `./cli.py supervise`. This is
attribution/policy, not detection, and it's the strongest thing here —
a human on exactly the temporal-escape surfaces.
4. **LLM semantic diff-review — the behavioral backstop.** The only
@@ -146,7 +146,7 @@ preserves the bottom-up distribution funnel.
**governed code-egress review**, not "we resell inference" (the
monetization notes explicitly warn against reselling compute).
- **The web-console supervise/review flow — the strongest anchor.** Turn
the CLI `bot-bottle supervise` approval into a real review surface:
the CLI `./cli.py supervise` approval into a real review surface:
rendered diff + finding context, approve/reject, **who-approved audit
trail, RBAC on approvers, mobile/phone-control** (ties to the
dashboard/vault north star). This is "central enforcement +
@@ -100,7 +100,7 @@ resolver globs each directory.
becomes file ops (mkdir, mv, rm) instead of editing one file.
Power users prefer that; new users may not.
- Discovery requires `ls`, not "grep one file." Tooling helps
(e.g. `bot-bottle list`) but the manifest is no longer a single
(e.g. `./cli.py list`) but the manifest is no longer a single
artifact to email or ship.
- Atomicity: swapping a bottle name across agents touches
multiple files. Git handles this fine; a one-shot text editor
@@ -364,7 +364,7 @@ wins; none of the body-prose or dependency story.
warns / ignores / breaks. If it warns, we'd want a different
field name (e.g. `bot-bottle-bottle`) or a namespaced block.
- **Migration story.** Is the project willing to ship a one-shot
`bot-bottle migrate-manifest` command that does the JSON → MD
`./cli.py migrate-manifest` command that does the JSON → MD
conversion? Or do users just rewrite by hand from the new docs?
- **Bottle file body content.** If most bottle .md files have an
empty body, is the MD-with-frontmatter format still warranted?
+1 -1
View File
@@ -34,7 +34,7 @@ on top of working onboarding.
A first-time user today goes through five steps: install Docker,
install `uv`, set `BOT_BOTTLE_CLAUDE_OAUTH_TOKEN`, write
`bot-bottle.json`, run `bot-bottle start`. One of those is
`bot-bottle.json`, run `./cli.py start`. One of those is
"author a JSON manifest." Polished tools in this category let
users skip that step on day one. The fix is an `init` subcommand
that drops a working `bot-bottle.json` with a default `coder`
+1 -1
View File
@@ -131,7 +131,7 @@ The minimum-viable workflow, no bot-bottle code changes:
3. SSH in.
4. `git clone` bot-bottle on the VM, drop a manifest in place,
inject `BOT_BOTTLE_CLAUDE_OAUTH_TOKEN` via the provider's secrets path.
5. `bot-bottle start <agent>` — the existing launcher handles the rest.
5. `./cli.py start <agent>` — the existing launcher handles the rest.
6. On exit: destroy the VM. No host artifacts persist.
For the "VPN pivot" failure mode, see
@@ -1,536 +0,0 @@
# Sandbox Agent SDK and bot-bottle: protocol versus product
This note asks whether [Sandbox Agent SDK](https://github.com/rivet-dev/sandbox-agent)
and bot-bottle compete for the same architectural layer, whether bot-bottle
can productize the turnkey ecosystem/DX layer above it, and how far the
Docker/OCI analogy actually holds.
Research conducted 2026-07-27. Sandbox Agent SDK was at the `0.4.x` line,
Apache 2.0, and documented support for Claude Code, Codex, OpenCode, Cursor,
Amp, and Pi at the time of review.
## Summary
**The projects are complementary at the component boundary and competitive at
the product boundary.** Sandbox Agent SDK normalizes how software controls a
coding-agent process inside an arbitrary sandbox. bot-bottle decides what
sandbox to create, what trusted role and policy it receives, how credentials
and Git access cross the boundary, how traffic is constrained, and how an
operator launches and supervises the result.
The Docker analogy is useful with one correction:
- Sandbox Agent SDK is not equivalent to Linux container APIs or OCI itself.
It is closer to a **containerd shim plus a portable exec/session API for
coding agents**. It adapts incompatible agent processes to one HTTP/SSE
contract.
- A future independent agent-session specification would be the closer OCI
analogue.
- bot-bottle can credibly occupy the **Docker Engine / Compose / Desktop**
layer: packaging, policy composition, lifecycle, networking, credentials,
storage, operator UX, and a one-command experience above interchangeable
agent adapters and isolation runtimes.
That is a viable position, but “turnkey wrapper” undersells it. A thin wrapper
is replaceable. The valuable product is a **turnkey, policy-first coding-agent
runtime** whose manifest compiles trusted operator intent into multiple
enforcement planes. Sandbox Agent SDK may be one internal process-control
component of that product.
The recommended direction is:
1. Keep the bot-bottle manifest as the host-owned source of trusted policy.
2. Spike Sandbox Agent SDK as the in-bottle provider/session adapter.
3. Expose a stable, provider-neutral bot-bottle control API, compatible with
Sandbox Agent where practical.
4. Keep security decisions and authoritative audit outside the sandbox.
5. Build the ecosystem around policy packs, agent images, skills, backends,
operator UI, and trusted integrations—not around a proprietary transcript
protocol.
## What each project is today
### Sandbox Agent SDK
Sandbox Agent is a Rust server that runs alongside the coding agent. A client
connects over HTTP, streams events over SSE, and uses one API across agent
implementations. Its documented surface includes:
- creating and restoring agent sessions;
- sending messages and streaming normalized events;
- handling permissions;
- configuring MCP servers, skills, and custom tools;
- filesystem and managed-process APIs;
- interactive terminal access;
- computer-use/desktop operations;
- a universal session/transcript schema;
- an Inspector UI, React components, CLI, TypeScript SDK, and OpenAPI spec.
It can run in embedded mode or inside E2B, Daytona, Modal, Cloudflare
Containers, Agent Computer, BoxLite, Docker, and other environments. It
explicitly leaves these concerns to the caller or sandbox provider:
- sandbox creation and lifecycle;
- Git repository management;
- durable session storage;
- network policy;
- isolation strength; and
- secure credential delivery.
Its documented credential convenience path extracts real provider credentials
from local agent configuration and passes them into the sandbox environment.
That is convenient but is not an acceptable security boundary for bot-bottle.
Sources:
- [Sandbox Agent repository and architecture](https://github.com/rivet-dev/sandbox-agent)
- [Sandbox Agent documentation](https://sandboxagent.dev/docs)
- [HTTP API](https://sandboxagent.dev/docs/api-reference)
- [Universal session/transcript schema](https://sandboxagent.dev/docs/session-transcript-schema)
### bot-bottle
bot-bottle is a host-side launch, policy, and enforcement system for existing
coding-agent CLIs. Its current architecture includes:
- agent and bottle manifests with composition via `extends:`;
- a host-only trust boundary for roles, identity, and secret references;
- provider templates and plugins for Claude Code, Codex, Pi, and custom
providers;
- Firecracker on KVM Linux and Apple Container on macOS, with Docker fallback;
- image construction and provider-specific provisioning;
- default-deny inspected egress with path/method/header policy;
- payload DLP on authorized channels;
- real credentials held outside the agent and injected by the gateway;
- Git mediation, upstream credential custody, and gitleaks scanning;
- a per-host authenticated orchestrator and shared gateway;
- named bottle lifecycle, resume, supervision, and audit state; and
- a CLI/TUI intended to make full-permission agents operationally tolerable.
The provider layer currently normalizes launch-time concerns—command, image,
prompt delivery, files, skills, environment, verification, and provider-owned
egress routes. It does **not** yet expose a stable provider-neutral runtime
contract for sessions, messages, transcripts, terminals, or normalized events.
That is the gap Sandbox Agent directly illuminates.
Sources in this repository:
- [`README.md`](../../README.md)
- [`0070-per-host-orchestrator.md`](../prds/0070-per-host-orchestrator.md)
- [`0026-agent-provider-templates.md`](../prds/0026-agent-provider-templates.md)
- [`0053-user-provider-plugins.md`](../prds/0053-user-provider-plugins.md)
- [`agent_provider.py`](../../bot_bottle/agent_provider.py)
## The layer model
The cleanest architecture has four layers:
| Layer | Responsibility | Likely owner |
|---|---|---|
| Operator product | Install, select a role, launch, observe, intervene, resume, review changes | bot-bottle |
| Trusted policy and lifecycle | Compose manifest, choose backend/image, hold credentials, enforce egress/Git, persist authoritative audit | bot-bottle |
| Agent control protocol | Start provider process, create session, send input, stream normalized events, terminal/computer operations | Sandbox Agent or a compatible protocol |
| Isolation primitive | VM/container/process boundary, filesystem, CPU/memory, networking substrate | Firecracker, Apple Container, Docker, E2B, Daytona, BoxLite, etc. |
The important boundary is between trusted policy/lifecycle and agent control.
The agent-control daemon runs in the environment being treated as untrusted.
It can report what the agent says happened, but it cannot authoritatively prove
that policy was enforced. Egress decisions, credential custody, Git scanning,
bottle identity, and security audit must remain outside it.
### Proposed composition
```text
operator UI / CLI / API
|
v
bot-bottle orchestrator (trusted)
- resolves manifest
- owns bottle identity and lifecycle
- stores authoritative audit
- authenticates clients
|
+--------------------------+
| |
v v
isolation backend shared gateway (trusted)
Firecracker / Apple / Docker - egress policy + DLP
| - credential injection
| - Git mediation
v
bottle / guest (untrusted)
- Sandbox Agent server
- Claude Code / Codex / Pi subprocess
- workspace, skills, MCP configuration
```
The bot-bottle manifest would compile into both sides:
- **outside the bottle:** backend, network, egress, credentials, Git,
supervision, identity, and authoritative lifecycle;
- **inside the bottle:** selected provider, prompt, skills, MCP configuration,
startup arguments, and non-secret session metadata.
Sandbox Agent should never receive real secrets merely because its API offers
a credential extraction helper. Provider and forge requests should continue
to use bot-bottle's placeholder/proxy pattern.
## How accurate is the Docker/OCI analogy?
### The useful part
The container ecosystem separates low-level execution from a product that
ordinary developers operate. OCI defines interoperable image, runtime, and
distribution specifications. Docker Engine adds a daemon, API, CLI, object
model, images, networks, volumes, and lifecycle; Docker Desktop and related
products add installation, updates, UI, integrations, policy, and team
workflows.
The same separation can exist for coding agents:
| Container ecosystem | Agent-sandbox ecosystem |
|---|---|
| OCI/runtime contract | A future open agent session/event contract |
| `runc` / runtime adapter | Claude/Codex/Pi adapter |
| containerd shim and task/exec API | Sandbox Agent server and HTTP/SSE session API |
| containerd / CRI-style lifecycle | Sandbox-provider lifecycle APIs |
| Docker Engine / Compose | bot-bottle orchestrator + manifests + backends + gateway |
| Docker Desktop / Hub ecosystem | bot-bottle desktop/mobile UX, policy packs, agent images, skills, trusted integrations |
Sandbox Agent makes coding-agent processes portable in roughly the way a shim
makes runtimes consumable through a common lifecycle interface. bot-bottle can
make the entire safe-agent system usable without asking the operator to
assemble that plumbing.
Official container references:
- [Open Container Initiative](https://opencontainers.org/)
- [OCI Runtime Specification](https://github.com/opencontainers/runtime-spec)
- [Docker Engine architecture](https://docs.docker.com/engine/)
- [Docker alternative runtimes and containerd shims](https://docs.docker.com/engine/daemon/alternative-runtimes/)
### Where the analogy breaks
1. **Sandbox Agent is an implementation, not an independent standard.**
Its OpenAPI document is public, but the project currently owns the server,
adapters, schema, and evolution. OCI is an independently governed set of
specifications with multiple implementations.
2. **It sits above, not below, the isolation boundary.** Linux namespaces,
cgroups, VMs, and OCI runtimes create the boundary. Sandbox Agent controls a
process after some other system has created that boundary.
3. **It reaches into product territory.** Inspector, React components,
computer-use APIs, skills/MCP configuration, transcripts, and restoration
are not merely low-level primitives. Sandbox Agent can continue growing
upward into the same UI and orchestration space bot-bottle might occupy.
4. **Coding agents are semantically uneven.** Normalizing a container
lifecycle is easier than claiming full behavioral parity across Claude
Code, Codex, Cursor, Amp, OpenCode, and Pi. A universal schema can become a
lowest common denominator or accumulate provider-specific escape hatches.
5. **The security contract is not standardized.** An agent-session API says
little about whether credentials are visible, egress is controlled, Git is
mediated, or audit is trustworthy. Those are core bot-bottle concerns.
The positioning should therefore say “Docker-like product layer above an open
agent-control protocol,” not “Sandbox Agent is OCI” or “bot-bottle implements
OCI for agents.”
## Can bot-bottle be the turnkey product layer?
Yes, if it owns substantially more than launch syntax.
The turnkey promise is:
> Choose a trusted role, point it at a project, and run any supported coding
> agent with full permissions. bot-bottle builds the environment, isolates it,
> supplies only the capabilities it needs, keeps credentials outside, mediates
> external writes, and gives the operator one place to watch and intervene.
That product has several defensible jobs:
### 1. Packaging and reproducibility
- provider and toolchain images;
- pinned, verified build inputs;
- skills and MCP configuration;
- role/bottle composition;
- cached startup and portable environment definitions; and
- compatibility testing across agents and backends.
### 2. Trusted policy compilation
The manifest is valuable because one reviewable document compiles into:
- an isolation plan;
- gateway routes and DLP policy;
- credential slots;
- Git-gate repositories and identities;
- provider configuration;
- supervision behavior; and
- operator-facing preflight.
Sandbox Agent's runtime configuration does not replace this. The policy must
be resolved before an untrusted guest or agent-control daemon exists.
### 3. Security enforcement
- dedicated-kernel isolation where available;
- no direct guest route to the internet;
- credentials injected outside the agent;
- content inspection on allowed destinations;
- Git secrets scanning and upstream-key custody;
- fail-closed policy resolution; and
- authoritative host-side audit.
This is the strongest current differentiation from a generic
“Sandbox Agent + Docker/E2B” assembly.
### 4. Lifecycle and operations
- install and host preflight;
- image build/update;
- start, stop, resume, cleanup, and migration;
- concurrent named agents;
- state recovery after crashes;
- live supervision and policy remediation; and
- backend selection without changing the role definition.
### 5. Ecosystem and DX
A product layer can support:
- curated provider images;
- signed policy/bottle packs;
- reusable role templates;
- skills and MCP bundles;
- backend plugins;
- an authenticated desktop/web/mobile operator client;
- browser/preview integration;
- normalized transcripts and change review; and
- team policy distribution and compliance exports.
The analogy to Docker is strongest here: users adopt the coherent workflow and
ecosystem, not because the low-level process API is proprietary.
## Business and product positioning
“Turnkey wrapper” is understandable internally but weak externally. It implies
that the hard work lives underneath and that another wrapper can replace it.
Prefer one of:
- **The policy-first runtime for coding agents**
- **Run any coding agent with full permissions, without giving it your host or
credentials**
- **A turnkey local control plane for isolated coding agents**
- **Docker-like packaging and operations for coding agents, with the security
boundary outside the agent**
The open/product split could resemble the container ecosystem:
### Open foundation
- manifest schema and composition;
- local CLI and core orchestrator;
- provider adapters;
- Firecracker/Apple Container/Docker backends;
- gateway policy format and enforcement;
- Sandbox Agent compatibility;
- local audit and supervision; and
- conformance tests for providers/backends/policy.
### Productizable ecosystem/DX
- polished desktop and mobile clients;
- fleet/remote-host management;
- signed and curated role/image/policy registry;
- team policy distribution and administrative controls;
- durable searchable transcripts and audit exports;
- SSO, RBAC, retention, and tamper-evident audit;
- managed update/compatibility channels;
- remote browser/preview relay;
- enterprise support; and
- optional managed build/cache infrastructure.
OCI itself is not the thing Docker sells. Interoperability expands the market;
the product captures value through reliable packaging, workflow, distribution,
management, and trust. bot-bottle should follow that logic rather than trying
to make its session protocol the moat.
## Strategic threat from Sandbox Agent
Sandbox Agent is a real threat for three reasons:
1. **It can become the integration default.** A frontend or agent platform can
integrate one API and choose among many agents and sandbox vendors.
2. **It can own session data and UI.** The universal event schema, Inspector,
React components, restoration, terminal, and computer-use APIs give it a
natural path toward the operator surface.
3. **Sandbox providers can move upward.** If E2B, Daytona, BoxLite, or another
runtime combines Sandbox Agent with adequate network policy and credential
custody, it can offer much of the turnkey stack.
The threat is not that its manifest syntax is better. It currently has no
equivalent trusted policy composition. The threat is that **the ecosystem may
standardize around its API before bot-bottle has a stable external control
surface**. In that world bot-bottle is evaluated as one sandbox provider,
while the SDK and its consumers own the user relationship.
## Why bot-bottle can still win its layer
Sandbox Agent's scope exclusions align with bot-bottle's deepest work:
- it does not choose or operate the sandbox provider;
- it does not mediate Git;
- it does not own network policy;
- it does not securely deliver credentials;
- it does not durably store sessions; and
- it cannot make guest-generated telemetry authoritative.
Those are not incidental features. Together they define the trusted system
around an untrusted coding agent. bot-bottle also has a narrower and coherent
initial customer: a developer or small operator who wants existing agent CLIs
to run locally with broad permissions and bounded consequences.
The durable advantage is therefore:
> Sandbox Agent makes agents controllable. bot-bottle makes them safe and
> operable.
That sentence remains true only if bot-bottle closes its operator-DX gaps.
Security without a browser/preview loop, stable API, normalized session view,
and good parallel-task UX risks becoming an invisible backend feature.
## Integration options
### Option A — Embed Sandbox Agent inside each bottle
bot-bottle launches Sandbox Agent as the provider process supervisor and
connects it to the host orchestrator through a bottle-scoped authenticated
channel.
**Benefits**
- immediate provider-neutral session API;
- more supported agents;
- normalized streaming and transcripts;
- terminal, filesystem, process, and computer-use primitives;
- Inspector/React ecosystem; and
- less provider-specific reverse engineering in bot-bottle.
**Risks**
- `0.x` API/schema churn;
- extra binary and release-supply-chain dependency;
- lowest-common-denominator normalization;
- conflict with provider-native resume state;
- an in-guest daemon is attacker-controlled after guest compromise;
- duplicate orchestration responsibilities; and
- upstream can move into policy/lifecycle and compete more directly.
**Security rule**
Treat every event and state claim from Sandbox Agent as untrusted telemetry.
Never delegate egress authorization, credential release, bottle identity,
authoritative audit, or Git policy to it.
### Option B — Implement a Sandbox Agent-compatible endpoint
bot-bottle maps the external protocol onto its existing provider adapters and
process model without running the upstream server.
**Benefits**
- ecosystem compatibility with tighter component control;
- no in-guest daemon dependency; and
- room to preserve bot-bottle-native lifecycle semantics.
**Risks**
- large and continuing compatibility burden;
- “full feature coverage” is expensive across all providers;
- accidental protocol fork; and
- effort diverted from policy and UX differentiation.
### Option C — Define an independent bot-bottle session API
Build only the control surface bot-bottle needs.
**Benefits**
- clean fit with the trust model and persistent named bottles;
- no upstream dependency; and
- deliberate support for supervision and security events.
**Risks**
- recreates a fast-growing open-source project;
- no existing client ecosystem;
- slower browser/desktop/mobile work; and
- increases the chance that Sandbox Agent becomes the de facto standard first.
### Recommendation
Start with **Option A as a bounded compatibility spike**, not a product
commitment. Do not begin with a clean-room competing protocol.
The spike should answer:
1. Can Claude Code, Codex, and Pi retain exact native resume behavior?
2. Can Sandbox Agent run without receiving real provider credentials?
3. Can its server be reached through a bottle-scoped authenticated channel
without exposing the orchestrator or broadening guest egress?
4. Which permission events overlap or conflict with bot-bottle supervision?
5. Can normalized events be stored while clearly separating untrusted
transcript telemetry from authoritative gateway/Git audit?
6. Can manifest skills, MCP servers, prompt, and startup arguments compile
deterministically into its configuration?
7. Does its versioning policy permit a compatibility contract bot-bottle can
support?
8. What image-size, startup-time, and update burden does the binary add?
If the answers are favorable, adopt it behind a bot-bottle-owned interface and
pin/test the supported version. If not, implement the smallest compatible
subset needed by external clients before inventing a wholly separate API.
## Product roadmap implications
The competitor scan and this architecture comparison reorder the likely work:
1. **Provider-neutral control/session compatibility spike**
2. **Stable authenticated external bot-bottle API**
3. **Normalized transcript/event persistence**
4. **Parallel-session operator UI**
5. **Browser/preview/computer-use capability**
6. **Policy/image/skill distribution and signing**
7. **Remote host/fleet management**
This does not mean pausing security work. It means exposing the shipped
security work through a product surface that can compete with the SDK-plus-
sandbox ecosystem.
## Decision
Treat Sandbox Agent SDK as a potentially standard **agent process-control
layer**, not as a sandbox replacement and not as a minor complementary
library. Position bot-bottle one layer above it:
- manifests express trusted role and environment policy;
- bot-bottle compiles and enforces that policy across host, gateway, Git, and
isolation backends;
- Sandbox Agent or a compatible protocol controls the selected agent process;
and
- bot-bottle owns the turnkey operator experience.
The Docker analogy is strategically sound when stated as:
> Sandbox Agent can be the portable task/exec protocol; bot-bottle can be the
> opinionated engine, Compose-like policy layer, and Desktop-like operator
> product.
It is not sound when stated as:
> Sandbox Agent is OCI and bot-bottle is Docker.
There is no independent OCI-equivalent agent specification yet, and Sandbox
Agent already reaches into UI/session territory. Compatibility should be
pursued quickly, while the trusted manifest/enforcement plane and operator
experience remain the parts bot-bottle deliberately owns.
@@ -1,183 +0,0 @@
# Testing a clean bot-bottle install on Linux
How do you exercise `install.sh` the way a brand-new user would — on a
pristine Linux environment you can throw away afterward — *without*
polluting your daily-driver host, and across the several package-management
regimes Linux fragments into? This is the Linux counterpart to
[`testing-clean-install-on-macos.md`](testing-clean-install-on-macos.md);
the conclusion is different because Linux gives us a boundary macOS doesn't.
## Summary
On macOS the honest options were a throwaway user or a VM, and the throwaway
user won on pragmatics (nested virtualization is gated to M3+). On Linux the
calculus flips: a **disposable KVM virtual machine, booted from a distro
cloud image and deleted per run, is both the cleanest boundary and the one
that lets a single harness cover Ubuntu, Fedora, Arch, Alpine, and NixOS**.
The host already requires KVM for the Firecracker backend, so the VM is cheap
here.
The harness lives at [`scripts/linux-install-test.sh`](../../scripts/linux-install-test.sh).
Per run it caches one read-only base image, boots a throwaway copy-on-write
overlay (`qemu-img create -b base`), installs the distro's prerequisites,
pipes *this checkout's* `install.sh` into the guest exactly as `curl … | sh`
would, asserts the CLI installed, and deletes the overlay — the Linux
equivalent of `docker run --rm`, for a whole machine.
## Why a VM, not a container or a throwaway user
| Mechanism | Why it's the wrong boundary here |
|---|---|
| **Container** (`docker run --rm`) | Shares the host kernel and ships a deliberately minimal userland — no systemd, a stubbed-out package manager story, and (crucially) it doesn't reproduce the *externally-managed Python* (PEP 668) that real desktop/server installs put in front of the user. It tests "does install.sh run in a container," not "does it run on a real distro." |
| **Throwaway user** (`useradd`/`userdel`) | The macOS pick, but weaker on Linux: it reaches the real host, yet every system package it installs (python, pipx, git via `apt`/`dnf`/…) stays behind, and it can only ever test the *one* distro the host runs. The whole Linux-specific value is the cross-distro matrix. |
| **Disposable KVM VM** (this harness) | A genuine kernel + userland + package-manager boundary that wipes to nothing on teardown, and swaps freely between distro cloud images. The one real cost — nested virtualization for the *backend* — doesn't apply, because we gate the installer, not the runtime (below). |
## Two variants: `test` (bare host) and `test-ready` (prepared host)
`install.sh` never installs a backend, and never installs its own toolchain
prerequisites (python3, git, pipx) — it installs the `bot-bottle` package and
runs `doctor`, which *reports* what's missing
([`install.sh`](../../install.sh) header,
[`bot_bottle/cli/commands/doctor.py`](../../bot_bottle/cli/commands/doctor.py)).
That leaves two distinct things worth testing, split into two subcommands that
mirror the macOS harness's `test` / `test-ready` convention (there the split is
the backend service; here it is the toolchain the installer needs):
- **`test`** — `install.sh` runs on the **bare cloud image**, prerequisites and
all left as the vendor ships them. This exercises `install.sh`'s own
prerequisite-guard logic — the entire first half of the script (python
version gate, git-for-git-specs gate, pipx/pip PEP-668 handling).
- **`test-ready`** — the harness installs python3 + git + pipx first (the
`prereqs` step), then runs `install.sh`. This is the *prepared-host happy
path*: does a clean install actually land and produce a working CLI?
**Pass criteria differ by variant:**
| Variant | PASS when |
|---|---|
| `test` | `install.sh` **either** installs cleanly (the image already carried enough) **or** declines with one of its own recognized, actionable prerequisite errors (missing python3/git, no usable pip, PEP 668). A crash or an *unrecognized* failure is a FAIL. |
| `test-ready` | `install.sh` actually lands: the `bot-bottle` entry point is present and runs, and `doctor` reports a usable python and config without crashing. A graceful decline is no longer good enough. |
Neither variant requires a green `doctor`: inside the VM there is no nested KVM
or Docker, so **the backend is correctly reported not-ready** — install.sh does
not install a backend and cannot regress one, and this harness does not
provision the Docker backend. This is where Linux necessarily diverges from the
macOS `test-ready`, which reaches the host backend; `BB_TEST_REQUIRE_BACKEND=1`
makes readiness fatal anyway, for a nested-virt host that can satisfy it. The
verdict instead classifies `doctor`'s output the way the macOS harness does — a
`Traceback` is an install defect (fail), a missing `python`/`config` line is a
fail, a not-ready backend is reported — so a genuine installer regression (a
broken shim, an import error, a botched PATH) stays visible.
`test-all` runs the full matrix — every distro × both variants — each cell in
its own throwaway VM, and prints a per-cell PASS/FAIL summary.
## The distro matrix is the point
Each distro exercises a different corner of the installer:
| Distro | Cloud image | What it stresses |
|---|---|---|
| **Ubuntu** (noble) | `cloud-images.ubuntu.com` | The common case; `apt`'s `pipx`, externally-managed Python (PEP 668) → install.sh's pipx path. |
| **Fedora** | Fedora Cloud Base Generic | `dnf` packaging, a different default Python, BSD-style checksum file. |
| **Arch** | `geo.mirror.pkgbuild.com/images/latest` | Rolling / newest Python; `python-pipx`. |
| **Alpine** | Alpine `nocloud_` (cloudinit) image | musl libc + BusyBox `sh` — the harshest POSIX-`sh` host for a `#!/bin/sh` installer. |
| **NixOS** | locally built with `nixos-generators` | No FHS `~/.local` on PATH by default; `nix profile install` prereqs; pipx laying a self-contained venv on a non-FHS host. |
### Validation run (2026-07-27) — full green
Full matrix on the delphi KVM host, QEMU 11.0.2, both variants × all five
distros passing:
| Distro | `test` (bare) | `test-ready` (prepared) |
|---|---|---|
| Ubuntu 24.04 | ✅ declines at git gate | ✅ installs, doctor python+config green |
| Fedora 44 | ✅ declines at git gate | ✅ installs |
| Arch (latest) | ✅ declines at git gate | ✅ installs |
| Alpine 3.21 | ✅ declines at git gate | ✅ installs |
| NixOS 24.11 | ✅ declines (no python3) | ✅ installs |
The bare `test` sees `install.sh` decline soundly — exit 1 at the
git-for-git-specs gate on the Debian/Fedora/Arch/Alpine images (they ship
python3 but not git), and at the python3 gate on NixOS (no python3 on PATH) —
and `test-ready` installs cleanly with `doctor` reporting a usable python and
config (backends all not-ready, as expected in a plain VM).
Getting to green surfaced and fixed a series of real defects:
- **Fedora 41 was EOL/404** → bumped to 44.
- The liveness probe used `bot-bottle --version`, which the CLI does not
implement (unknown args die non-zero), so every *successful* install was
misreported as failed → switched to `bot-bottle --help`.
- **Alpine** needed three fixes: the `generic_` image ignores a NoCloud seed
(switched to the `nocloud_` variant); OpenRC does not auto-start sshd after
cloud-init injects the key (start it via `runcmd`); and Alpine's non-PAM
sshd refuses pubkey auth for a cloud-init-*locked* account (give it a
throwaway password). It also has no `sudo` by default (install it via
cloud-init `packages:`).
- **NixOS** publishes no downloadable cloud qcow2 (its cloud images are
Hydra-built AMIs), so the harness builds one with `nixos-generators`
([`linux-install-test-nixos.nix`](../../scripts/linux-install-test-nixos.nix)):
cloud-init for the key, flakes enabled, deliberately no python/git/pipx. The
`test-ready` prereq install pins `nixpkgs/nixos-24.11` because the guest's
default unstable registry builds pipx from source (and its test suite
currently fails to build).
- Two harness-hygiene bugs also fixed: `cmd_down` left `serial.log` behind
(orphaned run dirs), and the teardown trap was armed after `cmd_up`, leaking
a VM when `wait_for_ssh` timed out.
In `test-ready` the harness installs `python3 + git + pipx` first on each
distro (install.sh installs none of them), so all five drive the recommended
pipx path. `test` then removes that scaffolding and lets each distro's bare
image collide with install.sh's guards — on most cloud images python3 is
present (cloud-init needs it) but git and pipx are not, so install.sh is
expected to decline at the git-for-git-specs gate or the PEP-668 pip check with
an actionable message. Both are legitimate, and the two variants together cover
the whole first half of the installer as well as the happy path.
## What a clean install touches (the footprint that decides "wipeable")
| Artifact | Location | In the guest's `$HOME`? | Survives VM teardown? |
|---|---|---|---|
| Config / state / db | `~/.bot-bottle/{agents,bottles,contrib,…}` ([`install.sh`](../../install.sh)) | ✅ | ❌ overlay deleted |
| pipx venv + shim | `~/.local/pipx/venvs/bot-bottle`, shim in `~/.local/bin` | ✅ | ❌ overlay deleted |
| pip `--user` fallback | `~/.local/lib` + `~/.local/bin` | ✅ | ❌ overlay deleted |
| **Distro prerequisites** (python/git/pipx, `test-ready` only) | system paths via `apt`/`dnf`/`pacman`/`apk`/`nix profile` | ❌ | ❌ **overlay deleted** |
Unlike the macOS throwaway user (whose Homebrew / Apple-Container / Rosetta
footprint *survives*), **every row here dies with the overlay** — that is the
VM's whole advantage. The cached base image is read-only backing and is the
only thing that persists between runs, on purpose.
## Design notes baked into the harness
- **User-mode networking** (`-netdev user,hostfwd=tcp:127.0.0.1:PORT-:22`):
no root, no bridge, no host network state touched. Only SSH is forwarded.
- **cloud-init seed ISO** (`cloud-localds`) injects an ephemeral SSH keypair
and a passwordless-sudo login. The keypair is generated per run and deleted
on teardown; the guest can't be logged into after it's gone.
- **Copy-on-write overlay**: the cached base is never mutated, so a corrupt or
interrupted run can't poison the cache; downloads land at `*.partial` and
are renamed only after checksum verification.
- **Checksums**: verified against each vendor's published sums file at
download time (GNU `hash file`, bare-hash, and Fedora's BSD
`SHA256 (file) = hash` formats are all handled). Alpine ships `.sha512` only
(this verifier is sha256) so it is skipped; NixOS is built locally, not
downloaded, so there is nothing to verify.
- **NixOS is built, not downloaded**: `ensure_base_image` runs
`nixos-generate -f qcow` against
[`linux-install-test-nixos.nix`](../../scripts/linux-install-test-nixos.nix)
once and caches the result; the per-run seed/overlay flow is otherwise
identical to the downloaded distros.
- **`test-all`** runs every distro × both variants (`test` and `test-ready`),
each cell in its own subshell on its own forwarded port, so one cell's
failure (or teardown trap) can't abort the matrix; it prints a per-cell
PASS/FAIL summary.
## Not wired into PR CI
Like the macOS harness, the runtime is host-specific (needs `/dev/kvm`,
`qemu`, and `cloud-localds`) and is not exercised by the Linux pull-request
runner. It is validated statically (`bash -n`, `shellcheck`) and run by hand
on a KVM-capable host. The cloud-image URLs in the `DISTRO` table are the one
place to bump when a distro cuts a newer build.
@@ -1,305 +0,0 @@
# Testing a clean bot-bottle install on macOS
How do you exercise `install.sh` (and, ideally, a first `bot-bottle start`)
the way a brand-new user would — on a pristine macOS environment you can
throw away afterward — *without* permanently polluting your daily-driver
Mac? The user's framing: is there a VM or boundary that avoids creating a
separate account, or is spinning up and tearing down a throwaway macOS
user on the CLI easy enough to just do that?
## Summary
There is no lightweight, in-place macOS sandbox that hands you a clean home
directory and wipeable system state without *either* a VM or a separate
user account. `sandbox-exec` (Seatbelt) is deprecated and confines a
process, not an environment; App Sandbox is for shipping apps, not for
provisioning a fresh dev host. So the real choice is exactly the two the
user named: **a disposable macOS VM** or **a throwaway user account**
and which one is right turns on a detail specific to *this* project.
bot-bottle's default macOS backend is Apple's `container`, which runs each
container in its own lightweight VM via `Virtualization.framework`
([`README.md:27`](../README.md), [`apple-container-backend.md`](apple-container-backend.md)).
That means a full end-to-end test — install *and* `bot-bottle start`
needs virtualization to work wherever bot-bottle runs. Inside a macOS guest
VM that requires **nested virtualization, which Apple gates to M3 or newer
chips on macOS 15+**. On M1/M2 you cannot run the Apple Container backend
(or Docker Desktop, same reason) inside a macOS VM at all.
The recommendation splits on what you're testing and what silicon you have:
- **Install-script correctness only** (does `curl | sh` → pipx → config dir
`doctor`'s Python/config checks pass?): a **disposable Tart VM** is the
cleanest boundary and works on any Apple Silicon Mac. `doctor` will report
the backend as not-ready inside the VM on M1/M2, which is fine — you're
testing the installer, not the runtime.
- **Full runtime** (actually launch a bottle) on **M3/M4**: a **disposable
Tart VM from a golden base image, cloned per run** is the gold standard —
a genuine kernel/state boundary that wipes to nothing.
- **Full runtime** on **M1/M2**, or when you'd rather not fight nested virt:
a **throwaway admin user via `sysadminctl`** is the pragmatic pick. It
tests the real backend because the backend runs on the host hypervisor —
but it is a *hygiene* boundary, not a security one, and it does **not**
clean the system-level footprint (see below).
Prefer the VM. Reach for the throwaway user only when nested virt is off the
table and you accept an imperfect wipe.
## Why "a boundary without a separate user" doesn't really exist on macOS
macOS has no namespace/overlay story like Linux `unshare` + tmpfs. The
options that sound like in-place sandboxes don't fit:
| Mechanism | Why it doesn't give you a clean, wipeable env |
|---|---|
| `sandbox-exec` / Seatbelt | Officially deprecated; confines *one process's* syscalls against a profile. It cannot present a fresh `$HOME` or a pristine `/usr/local`, and it won't let the Apple Container system service work. |
| App Sandbox | Entitlement-based confinement for signed `.app` bundles, not a provisioning tool for a CLI dev environment. |
| A second `$HOME` via `HOME=/tmp/foo` | Redirects only what honors `$HOME`. `install.sh` mostly does (it writes `~/.bot-bottle` and pipx/pip `--user` paths), but the Apple `container` install lands in `/usr/local` + a **system service**, and Homebrew lands in `/opt/homebrew` — all outside any `$HOME` you set. You'd get a false sense of "clean." |
| APFS snapshot rollback (`tmutil localsnapshot`) | You can't roll the live boot volume back to a local snapshot without booting to Recovery; it's not a per-run userspace undo. |
So the honest answer to "is there some boundary that avoids a separate
user?": yes — a **VM** — and it's the *stronger* boundary anyway. The only
lighter-weight option is the separate user, with the caveats below.
## What a clean install actually touches (the footprint that decides "wipeable")
Grounding the teardown story in what `install.sh` and the backend create:
| Artifact | Location | In `$HOME`? | Survives user deletion? |
|---|---|---|---|
| Config / state / db | `~/.bot-bottle/{agents,bottles,contrib,state,db}` ([`install.sh:80-83`](../install.sh), [`bot_bottle/paths.py:59`](../bot_bottle/paths.py)) | ✅ | ❌ removed with home |
| pipx venv + shim | `~/.local/pipx/venvs/bot-bottle`, shim in `~/.local/bin` ([`install.sh:87-89`](../install.sh)) | ✅ | ❌ removed with home |
| private venv fallback (no pipx) | `~/.bot-bottle/venv` + symlink in `~/.local/bin` ([`install.sh`](../install.sh)) | ✅ | ❌ removed with home |
| PATH / token exports | shell profile (`~/.zprofile`, etc.); `BOT_BOTTLE_CLAUDE_OAUTH_TOKEN` ([`README.md:74`](../README.md)) | ✅ | ❌ removed with home |
| **Apple `container` install** | `/usr/local/...` + notarized `.pkg` receipts | ❌ | ✅ **stays** |
| Apple `container` **service state** | per-user: `~/Library/Application Support/com.apple.container/` (`container system start`) | ✅ | ❌ removed with home |
| **Homebrew** (if used for `container`/python) | `/opt/homebrew` | ❌ | ✅ **stays** |
| Rosetta 2 (needed for image builds) | system | ❌ | ✅ **stays** |
The bold rows are the crux: **deleting the throwaway user does not uninstall
the Apple Container runtime, Homebrew, or Rosetta.** A VM, by contrast, wipes
100% of the above by definition — that's its entire advantage for this task.
The service row is the exception, and a live harness run corrected it: the
Apple `container` service is **per-user**, not a host-wide launchd service.
`container system status` reports an `appRoot` under
`~/Library/Application Support/com.apple.container/`, and a freshly created
account sees `container system service: NOT running` even while the creating
admin's is running. That cuts both ways — the state is genuinely removed with
the home, so the reset is *more* complete than this table first claimed, but
it also means **no brand-new user can run a bottle until they run
`container system start` once**. `doctor` correctly fails until they do, which
is why `test` judges backend readiness separately from install correctness.
## Option A — Disposable Tart VM (recommended)
[Tart](https://tart.run) is a CLI-first macOS/Linux VM manager built on
`Virtualization.framework`, purpose-built for exactly this "does it work on
a clean macOS, without my settings/permissions/data" workflow. Keep one
pristine *golden* image, clone a throwaway per run, delete it after.
```sh
brew install cirruslabs/cli/tart
# One-time: build a golden base (either a prebuilt image or a vanilla IPSW).
tart clone ghcr.io/cirruslabs/macos-tahoe-base:latest golden # ~25 GB pull
# — or a truly vanilla install you click through once —
# tart create golden --from-ipsw latest --disk-size 60
# Per test run: clone → boot → test → destroy.
tart clone golden test-run
tart run test-run &
ssh admin@"$(tart ip test-run)"
# inside the guest:
# curl -fsSL https://gitea.dideric.is/didericis/bot-bottle/raw/branch/main/install.sh | sh
# bot-bottle doctor
tart stop test-run
tart delete test-run # back to pristine; golden is untouched
```
Cloning is cheap (sparse files), so the golden image is your reset button —
every `tart clone` is a fresh macOS. This is the closest thing to a Linux
`docker run --rm` for a whole Mac.
**The nested-virt caveat (read before relying on it for runtime tests).**
The Apple Container backend inside the guest needs
`Virtualization.framework` to work *inside* the VM. Apple enables nested
virtualization only on **M3 or newer**, on **macOS 15 (Sequoia) or later**;
M2 and earlier are excluded by Apple, confirmed by Apple DTS. Consequences:
- **M3/M4 host:** full runtime works in the guest. `bot-bottle doctor`
reports the backend ready and `start` can launch a bottle. Gold standard.
- **M1/M2 host:** the guest can install bot-bottle and pass the Python /
config-dir checks, but `doctor`'s backend check will fail and you cannot
launch a bottle in the VM. Still perfectly good for testing *the
installer*; not for the runtime.
- **M4-specific:** a known bug blocks pre-Ventura guests on M4; use a
current macOS guest (which you want anyway, since Apple `container`
targets macOS 26 Tahoe).
UTM is the GUI equivalent on the same framework (and was first to expose
nested virt) if you'd rather click; Tart wins for a scriptable
spin-up/tear-down loop.
## Option B — Throwaway user via `sysadminctl` (pragmatic fallback)
Creating and deleting a user from the CLI is genuinely a two-liner, and it
tests the **real** backend on any Apple Silicon Mac because the backend runs
on the host hypervisor — no nested virt needed.
```sh
# Create a self-contained admin user (admin needed for the container service).
sudo sysadminctl -addUser bbtest -fullName "bot-bottle test" \
-password 'throwaway' -admin
# Log into that account (fast-user-switch or the login window), then run the
# installer as bbtest exactly as a new user would. When done:
sudo sysadminctl -deleteUser bbtest -secure # -secure erases the home dir
```
Honest accounting of what this does and doesn't buy you:
- **Boundary strength:** it's a *hygiene / fresh-`$HOME`* boundary, **not a
security boundary.** Same kernel, same admin group; an admin test user can
touch system state. If the point is "clean environment," fine. If the point
is "contain something untrusted," this is the wrong tool — use a VM.
- **Wipe completeness:** `-secure` erases the home dir (so `~/.bot-bottle`,
the pipx venv, and profile exports go away), but as the footprint table
shows, the **Apple Container runtime, its launchd system service,
Homebrew, and Rosetta persist.** For a truly repeatable "did a *system with
nothing installed* work?" test, that residue defeats the purpose — the
second run isn't clean.
- **Operational gotchas:** don't pass real passwords on the command line (they
land in `ps` and history — this is a throwaway credential, so it's
tolerable here). Deletion must run as root from a normally-booted, admin-
logged-in session; the Terminal needs **Full Disk Access** or you'll hit
error `-14120` and a half-deleted account. Prefer letting the system place
the home dir (don't pass `-home`), or deletion can orphan it.
Use this when you're on M1/M2, you specifically want to exercise the live
backend, and you can tolerate the system-level runtime staying installed
between runs (or you uninstall Apple `container` / brew by hand to reset).
## Honorable mentions
- **External bootable macOS volume.** A fresh macOS on an external SSD (or a
separate APFS volume) is bare-metal disposable: no nested-virt limit, real
backend works, and you `diskutil` the volume away to reset. Cost is reboot
friction per run — good for an occasional thorough pass, poor for a tight
loop.
- **Rented / cloud Mac.** AWS EC2 Mac (dedicated Mac minis), Scaleway Apple
silicon, or MacStadium give a genuinely throwaway host you release when
done. Overkill for local iteration, but this is essentially what the
project's own advisory `integration-macos` CI job needs — a self-hosted
Apple Silicon runner with the `container` CLI, Python ≥ 3.11, and coverage
on the launchd service's PATH ([`README.md:78`](../README.md)). If you end
up standing up a cloud Mac for install testing, it doubles as that runner.
## Recommendation
Default to a **disposable Tart VM** — it's the only option that wipes the
*entire* footprint (including the Apple Container system service that a user
deletion leaves behind), it's a real boundary, and the spin-up/tear-down
loop is a two-command `tart clone` / `tart delete`. Confirm your chip first:
on **M3/M4** it tests install *and* runtime end-to-end; on **M1/M2** it still
cleanly tests `install.sh` + `doctor`'s Python/config path, and you fall back
to a **throwaway `sysadminctl` admin user** for live-backend testing —
accepting that it's a hygiene boundary and that you'll manually uninstall the
Apple Container runtime / Homebrew between runs to get back to truly clean.
There is no third, lighter-weight "in-place boundary without a user" that
actually delivers a clean, wipeable macOS — the VM *is* that answer, and it's
the better one.
## Harness
The throwaway-user loop is scripted in
[`scripts/macos-install-test.sh`](../../scripts/macos-install-test.sh):
`up` creates the account, `run` pipes *this checkout's* `install.sh` into it
headlessly (so a PR is verifiable before it lands) and lets the installer run
`doctor`, `down` deletes the account and its home (the full reset), and
`deep-reset` additionally uninstalls the host `container` runtime. It leans on
the footprint analysis above — the reset is just user deletion because
everything `install.sh` writes is user-home-local.
There are two one-shot cycles, because "does the installer work" and "can a new
user actually run a bottle" are different questions:
```sh
sudo ./scripts/macos-install-test.sh test # up → run → status → down
sudo ./scripts/macos-install-test.sh test-ready # ... + prereqs before status
```
`test` models a macOS system **without** the prerequisites set up for this user
— which is the default state of every new account, since the `container`
service is per-user. It asserts the install is sound and *reports* backend
readiness without failing on it, because install.sh provides no backend and so
cannot regress one.
`test-ready` models a system **with** them, then demands doctor go fully green,
backend included. Both variants pass as of this writing.
### The per-user prerequisite is two steps, not one
Running it revealed that "set up the backend for this account" is more than
starting a service:
1. **`container system start`** — the service is per-user. The run confirms it
directly: the throwaway account's `appRoot` is
`/Users/bbtest/Library/Application Support/com.apple.container/`, its own,
and starting it left the admin's service untouched.
2. **A guest kernel**, which also lives in that per-user app root. A fresh
account has none, so `container system start` prompts to download one —
and *only* prompts, since the flags default to asking. Headless callers
must pass `--enable-kernel-install` or the command dies on
`failed to read user input`.
Neither step is done by `bot-bottle backend setup --backend=macos-container`,
which only checks and then tells you to run `container system start` yourself.
So a new account's real path to a working backend is:
`container system start --enable-kernel-install`.
The harness must also enter the user's launchd domain via
`launchctl asuser <uid>` to do any of this. `container system start` registers
`com.apple.container.apiserver` as a per-user launchd agent and talks to it
over XPC; from plain `sudo -u` the caller is still in root's bootstrap
namespace, the lookup crosses domains, and the apiserver answers
`invalidState: "unauthorized request"` even though the agent started fine.
It refuses to start against an existing account (a reused home is not a clean
install), and it tears the account down from an `EXIT`/`INT` trap armed the
moment the account exists, so a failed or Ctrl-C'd run still leaves the machine
clean. Its verdict is deliberately stricter than the installer's own: note that
`install.sh` exits **0** when it finishes but `doctor` reports unmet
prerequisites, so "the installer succeeded" is not the assertion — `test` fails
if the install fails, if `bot-bottle` never reached the new user's `PATH`, or if
`doctor` is unhappy. `BB_TEST_KEEP=1` skips the teardown to poke at a failure.
### What a fresh account actually inherits
Expect the first honest run on a developer Mac to fail at the *Python* gate,
and expect that to be correct. A new account's `PATH` is just `/etc/paths`
(`/usr/local/bin:/System/Cryptexes/App/usr/bin:/usr/bin:/bin:/usr/sbin:/sbin`),
which notably does **not** include `/opt/homebrew/bin`. Homebrew's `shellenv`
line lives in the *installing* user's `~/.zprofile` and is not inherited, so a
throwaway user resolves `python3` to `/usr/bin/python3` — the Command Line
Tools stub, still **3.9.6** on macOS 26 — and `install.sh` correctly dies on its
`3.11+` requirement. Your own shell resolving `python3` to a 3.14 Homebrew
build says nothing about what a new user sees; that gap is exactly what this
harness exists to expose.
## Sources
- [Apple Containers on macOS: technical comparison with Docker — The New Stack](https://thenewstack.io/apple-containers-on-macos-a-technical-comparison-with-docker/)
- [How to Set Up Apple Containerization on macOS 26 — Stéphane Paquet](https://spaquet.medium.com/how-to-set-up-apple-containerization-on-macos-26-f870cc8c26cd)
- [Install Apple Container CLI (macOS 15/26) — 4sysops](https://4sysops.com/archives/install-apple-container-cli-running-containers-natively-on-macos-15-sequoia-and-macos-26-tahoe/)
- [Nested virtualization on Apple Silicon (M3+, macOS 15) — UTM issue #6700](https://github.com/utmapp/UTM/issues/6700)
- [macOS 15 Sequoia nested virtualization for M3+ — Parallels Forums](https://forum.parallels.com/threads/macos-15-sequoia-nested-virtualization-for-m3-macs.364397/)
- [M2 nested virtualization restriction (Apple DTS) — Apple Developer Forums](https://developer.apple.com/forums/thread/756723)
- [M4 can't virtualize older macOS — Yahoo/Tech](https://tech.yahoo.com/computing/articles/m4-mac-computers-cant-virtualize-175122301.html)
- [Tart — macOS/Linux VMs on Apple Silicon (Cirrus Labs)](https://tart.run/quick-start/)
- [Tart GitHub](https://github.com/cirruslabs/tart)
- [macOS VMs in a single command — frr.dev](https://www.frr.dev/posts/tart-macos-vms-from-terminal/)
- [sysadminctl reference — SS64](https://ss64.com/mac/sysadminctl.html)
- [User management from the macOS command line — macnotes](https://macnotes.wordpress.com/2019/03/28/user-management-create-remove-change-password-secure-token-from-macos-command-line/)
+51 -131
View File
@@ -8,20 +8,14 @@
# pipx install bot-bottle # from a checkout or a published index
# uv tool install bot-bottle
#
# This script is a thin bootstrapper: it finds a Python 3.11+ interpreter,
# installs the package with pipx (falling back to a private venv), creates the
# config dir, and runs `bot-bottle doctor`. It is idempotent (safe to re-run)
# and never uses sudo. It does NOT install Docker or a VM backend for you —
# `doctor` reports what's missing after install.
#
# Env:
# BOT_BOTTLE_PYTHON interpreter to install with (skips the search)
# BOT_BOTTLE_INSTALL_SPEC pip/git spec to install instead of the default
# BOT_BOTTLE_VENV where the non-pipx install lives (~/.bot-bottle/venv)
# This script is a thin bootstrapper: it checks prerequisites, installs the
# package with pipx (falling back to pip --user), creates the config dir, and
# runs `bot-bottle doctor`. It is idempotent (safe to re-run) and never uses
# sudo. It does NOT install Docker or a VM backend for you — `doctor` reports
# what's missing after install.
set -eu
PACKAGE_SPEC="${BOT_BOTTLE_INSTALL_SPEC:-git+https://gitea.dideric.is/didericis/bot-bottle.git}"
VENV="${BOT_BOTTLE_VENV:-${HOME}/.bot-bottle/venv}"
MIN_PYTHON_MAJOR=3
MIN_PYTHON_MINOR=11
@@ -34,97 +28,17 @@ die() {
exit 1
}
# --- prerequisites: find an interpreter new enough ----------------------------
# --- prerequisites -----------------------------------------------------------
# Is $1 an interpreter that exists and meets the floor?
python_ok() {
[ -n "${1:-}" ] || return 1
command -v "$1" >/dev/null 2>&1 || return 1
"$1" - "$MIN_PYTHON_MAJOR" "$MIN_PYTHON_MINOR" <<'PY' >/dev/null 2>&1
command -v python3 >/dev/null 2>&1 \
|| die "python3 ${MIN_PYTHON_MAJOR}.${MIN_PYTHON_MINOR}+ is required but was not found"
python3 - "$MIN_PYTHON_MAJOR" "$MIN_PYTHON_MINOR" <<'PY' || die "python3 ${MIN_PYTHON_MAJOR}.${MIN_PYTHON_MINOR} or newer is required"
import sys
want = (int(sys.argv[1]), int(sys.argv[2]))
raise SystemExit(0 if sys.version_info[:2] >= want else 1)
PY
}
python_version() {
"$1" -c 'import sys; print("%d.%d.%d" % sys.version_info[:3])' 2>/dev/null
}
# `python3` on PATH is often NOT the newest interpreter installed, and on macOS
# it is usually the oldest: a fresh login shell's PATH is just /etc/paths, so
# python3 resolves to the Command Line Tools stub (3.9.x) while the usable
# 3.11+ build sits in /opt/homebrew/bin or a python.org framework directory,
# reachable only via a line in the *installing* user's shell profile. A new
# account inherits none of that. Look past PATH before giving up, so the common
# case installs instead of dead-ending on a version error.
find_python() {
for candidate in \
"${BOT_BOTTLE_PYTHON:-}" \
python3 \
python3.14 python3.13 python3.12 python3.11 \
/opt/homebrew/bin/python3 \
/usr/local/bin/python3 \
"${HOME}/.local/bin/python3" \
/Library/Frameworks/Python.framework/Versions/*/bin/python3
do
# An unmatched glob arrives here literally; python_ok rejects it.
if python_ok "$candidate"; then
command -v "$candidate"
return 0
fi
done
return 1
}
# An explicit choice that doesn't work is an error, not a reason to quietly
# search elsewhere and install somewhere the caller didn't ask for.
if [ -n "${BOT_BOTTLE_PYTHON:-}" ] && ! python_ok "${BOT_BOTTLE_PYTHON}"; then
if command -v "${BOT_BOTTLE_PYTHON}" >/dev/null 2>&1; then
die "BOT_BOTTLE_PYTHON=${BOT_BOTTLE_PYTHON} is $(python_version "${BOT_BOTTLE_PYTHON}"), "\
"below the ${MIN_PYTHON_MAJOR}.${MIN_PYTHON_MINOR} floor. Unset it to search for a newer one."
fi
die "BOT_BOTTLE_PYTHON=${BOT_BOTTLE_PYTHON} is not an executable interpreter."
fi
PYTHON="$(find_python || true)"
if [ -z "${PYTHON}" ]; then
path_python="$(command -v python3 2>/dev/null || true)"
if [ -n "${path_python}" ]; then
found="the python3 on your PATH is ${path_python} ($(python_version "${path_python}")), which is too old"
else
found="no python3 was found on your PATH"
fi
case "$(uname -s)" in
Darwin) fix=" brew install python@3.12
# or install from https://www.python.org/downloads/macos/
# macOS itself ships only /usr/bin/python3, which is too old" ;;
*) fix=" sudo apt install python3.12 # Debian/Ubuntu
sudo dnf install python3.12 # Fedora/RHEL" ;;
esac
die "bot-bottle needs python3 ${MIN_PYTHON_MAJOR}.${MIN_PYTHON_MINOR} or newer, and none was found.
${found}.
Also checked: python3.11-3.14, /opt/homebrew/bin, /usr/local/bin,
~/.local/bin, and python.org framework builds.
Install a newer Python, then re-run this installer:
${fix}
Already have one somewhere? Point at it directly:
BOT_BOTTLE_PYTHON=/path/to/python3 sh install.sh"
fi
# Be explicit when the interpreter isn't the obvious one, so nobody is left
# wondering which Python their install ended up on.
path_python="$(command -v python3 2>/dev/null || true)"
if [ "${PYTHON}" != "${path_python}" ]; then
say "using ${PYTHON} ($(python_version "${PYTHON}"))"
if [ -n "${path_python}" ]; then
say "note: 'python3' on your PATH is ${path_python} ($(python_version "${path_python}")), which is below the ${MIN_PYTHON_MAJOR}.${MIN_PYTHON_MINOR} floor"
fi
fi
# Installing a `git+` spec (the default) shells out to git under the hood,
# whether via pipx or pip. Fail early with a clear message rather than deep
@@ -137,6 +51,30 @@ case "${PACKAGE_SPEC}" in
;;
esac
# The pip fallback needs a usable pip. Externally-managed interpreters
# (PEP 668, common on Debian/Ubuntu/Homebrew) reject `pip install --user`;
# pipx sidesteps that, so recommend it when pip can't be used.
if ! command -v pipx >/dev/null 2>&1; then
python3 -m pip --version >/dev/null 2>&1 || die \
"neither pipx nor a usable 'python3 -m pip' was found. Install pipx "\
"(recommended): 'python3 -m pip install --user pipx' or your OS package manager."
if python3 - <<'PY'
import os
import sys
import sysconfig
# PEP 668: an EXTERNALLY-MANAGED marker in the stdlib dir means pip refuses
# to install into this interpreter without --break-system-packages.
marker = os.path.join(sysconfig.get_path("stdlib"), "EXTERNALLY-MANAGED")
raise SystemExit(0 if os.path.exists(marker) else 1)
PY
then
die "this Python is externally managed (PEP 668), so 'pip install --user' is "\
"blocked. Install pipx and re-run: 'python3 -m pip install --user --break-system-packages pipx', "\
"then 'pipx ensurepath'."
fi
fi
# --- config directories ------------------------------------------------------
mkdir -p \
@@ -146,50 +84,32 @@ mkdir -p \
# --- install -----------------------------------------------------------------
BIN_DIR="${HOME}/.local/bin"
if command -v pipx >/dev/null 2>&1; then
# --python pins the venv to the interpreter we vetted. Without it pipx uses
# whichever Python it was itself installed with, which is not necessarily
# the one that passed the version check above.
say "installing with pipx (python: ${PYTHON})"
pipx install --python "${PYTHON}" --force "${PACKAGE_SPEC}"
# Ask pipx where it puts entry points rather than assuming ~/.local/bin.
pipx_bin="$(pipx environment --value PIPX_BIN_DIR 2>/dev/null || true)"
[ -n "${pipx_bin}" ] && BIN_DIR="${pipx_bin}"
say "installing with pipx"
pipx install --force "${PACKAGE_SPEC}"
else
# No `pip install --user` fallback: PEP 668 makes it unusable on nearly
# every interpreter a Mac offers (Homebrew and python.org are both
# externally managed), and on Debian/Ubuntu too. A private venv sidesteps
# that entirely — PEP 668 does not apply inside a venv — and `venv` is
# stdlib, so unlike pipx there is nothing to bootstrap first.
say "pipx not found; installing into a managed venv at ${VENV}"
"${PYTHON}" -m venv --clear "${VENV}" || die \
"could not create a virtualenv at ${VENV} using ${PYTHON}. On Debian/Ubuntu "\
"the venv module ships separately: 'sudo apt install python3-venv'."
"${VENV}/bin/python" -m pip install --upgrade "${PACKAGE_SPEC}"
# Expose the entry point outside the venv, the way pipx would.
mkdir -p "${BIN_DIR}"
ln -sf "${VENV}/bin/bot-bottle" "${BIN_DIR}/bot-bottle"
say "pipx not found; installing with 'python3 -m pip install --user'"
python3 -m pip install --user --upgrade "${PACKAGE_SPEC}"
fi
# --- locate the entry point --------------------------------------------------
# The pip --user scripts directory is platform-specific: ~/.local/bin on Linux,
# but ~/Library/Python/<X.Y>/bin on a python.org macOS interpreter. Ask the
# interpreter for its own user-scheme scripts dir instead of hardcoding.
USER_SCRIPTS="$(python3 - <<'PY'
import sysconfig
print(sysconfig.get_path("scripts", sysconfig.get_preferred_scheme("user")))
PY
)"
if command -v bot-bottle >/dev/null 2>&1; then
BOT_BOTTLE_BIN="bot-bottle"
elif [ -x "${BIN_DIR}/bot-bottle" ]; then
BOT_BOTTLE_BIN="${BIN_DIR}/bot-bottle"
# Name the file the user's own login shell actually reads. ~/.profile is
# the safe default for non-zsh: bash falls back to it, and suggesting
# ~/.bash_profile could shadow an existing ~/.profile.
case "${SHELL:-}" in
*/zsh) profile="~/.zprofile" ;;
*) profile="~/.profile" ;;
esac
say "note: add ${BIN_DIR} to your PATH to run 'bot-bottle' directly:"
say " echo 'export PATH=\"${BIN_DIR}:\$PATH\"' >> ${profile}"
elif [ -n "${USER_SCRIPTS}" ] && [ -x "${USER_SCRIPTS}/bot-bottle" ]; then
BOT_BOTTLE_BIN="${USER_SCRIPTS}/bot-bottle"
say "note: add ${USER_SCRIPTS} to your PATH to run 'bot-bottle' directly"
else
die "bot-bottle was installed but no entry point turned up in ${BIN_DIR}"
die "bot-bottle was installed but is not on PATH; add ${USER_SCRIPTS:-your user scripts dir} to PATH and re-run"
fi
# --- verify ------------------------------------------------------------------
+1 -1
View File
@@ -9,7 +9,7 @@
#
# Why a pool + one-time setup: creating a TAP and assigning it an IP
# needs CAP_NET_ADMIN. Pre-creating user-owned, pre-addressed TAPs
# means `bot-bottle start` never needs root. The nft table is static
# means `./cli.py start` never needs root. The nft table is static
# (keyed on the `bbfc*` interface wildcard), so it covers every slot
# without per-launch changes.
#
-24
View File
@@ -1,24 +0,0 @@
# NixOS image for scripts/linux-install-test.sh.
#
# NixOS publishes no downloadable cloud qcow2 (its cloud images are Hydra-built
# AMIs), so the harness BUILDS this one with nixos-generators (-f qcow). It is
# deliberately minimal — no python3/git/pipx — so the bare `test` variant is
# genuinely under-provisioned and exercises install.sh's guards; `test-ready`
# provisions them with `nix profile install` (hence flakes below).
{ lib, ... }:
{
# Consume the same NoCloud seed the other distros use: cloud-init injects the
# per-run ephemeral SSH key for root. Leave networking to NixOS's default
# dhcpcd (QEMU user-mode NAT) — enabling cloud-init's networkd here conflicts
# with dhcpcd and can drop the guest's network.
services.cloud-init.enable = true;
services.cloud-init.network.enable = false;
services.openssh.enable = true;
services.openssh.settings.PermitRootLogin = lib.mkForce "prohibit-password";
# Flakes so the `test-ready` prereq step can `nix profile install nixpkgs#...`.
nix.settings.experimental-features = [ "nix-command" "flakes" ];
system.stateVersion = "24.11";
}
-753
View File
@@ -1,753 +0,0 @@
#!/usr/bin/env bash
# Clean-install test harness for the Linux path.
#
# Exercises install.sh the way a brand-new user would, inside a THROWAWAY
# QEMU/KVM virtual machine that is booted from a distro cloud image and
# deleted afterward. install.sh's entire footprint is user-home-local (the
# pipx venv under ~/.local, the ~/.bot-bottle config dir, and a printed PATH
# hint), so a fresh VM's fresh $HOME is the clean surface we want — and unlike
# a throwaway user account, tearing the VM down also wipes any OS-level
# prerequisites installed into it, so the reset is total. Full rationale in
# docs/research/testing-clean-install-on-linux.md.
#
# Why a VM and not a container or a throwaway user: a container shares the
# host kernel and cannot exercise a genuinely pristine OS (systemd, the distro
# package manager, PEP 668 externally-managed Python) the way a real guest
# does, and a throwaway user leaves every system package it installs behind.
# The host already needs KVM for the Firecracker backend, so a per-run,
# copy-on-write VM is cheap here: one cached base image, a throwaway overlay
# per run (`qemu-img create -b base`), deleted on teardown — the Linux
# equivalent of `docker run --rm`, but for a whole machine.
#
# Usage:
# ./scripts/linux-install-test.sh test # up -> run -> verdict -> down (bare host)
# ./scripts/linux-install-test.sh test-ready # ... with prerequisites installed first
# ./scripts/linux-install-test.sh test-all # every distro × both variants, with a summary
# ./scripts/linux-install-test.sh up # fetch base image, boot a fresh VM
# ./scripts/linux-install-test.sh prereqs # install python3 + git + pipx in the VM
# ./scripts/linux-install-test.sh run # pipe install.sh into the VM
# ./scripts/linux-install-test.sh status # VM reachable? is the install sound?
# ./scripts/linux-install-test.sh down # kill the VM, delete overlay + seed (the reset)
# ./scripts/linux-install-test.sh ssh # open an interactive shell in the running VM
#
# TWO TEST VARIANTS, because "does install.sh handle an unprepared host" and
# "does a prepared host get a clean install" are different questions (mirrors
# the macOS harness's test / test-ready split):
#
# test A Linux system WITHOUT the prerequisites set up — the default
# state of a stock cloud image (python3 is usually present for
# cloud-init, but git and pipx are not). This exercises
# install.sh's own prerequisite-guard logic. It is SOUND — a
# PASS — when install.sh EITHER installs cleanly (the image
# already had enough) OR declines with one of its own recognized,
# actionable errors (missing python3/git, no usable pip, PEP 668
# externally-managed). A crash or an unrecognized failure fails.
#
# test-ready A Linux system WITH the prerequisites satisfied — the harness
# installs python3 + git + pipx first (see the DISTRO table),
# then runs install.sh. A graceful decline is no longer good
# enough here: the install MUST land, the entry point must run,
# and doctor must report a usable python and config without
# crashing.
#
# Neither variant requires a green backend. install.sh does not install a
# backend and cannot regress one, and inside a plain VM there is no nested KVM
# for Firecracker; the harness does not provision the Docker backend either. So
# backend readiness is reported, not required (this is where Linux necessarily
# diverges from the macOS test-ready, which reaches the host backend).
# BB_TEST_REQUIRE_BACKEND=1 makes it fatal anyway, for a nested-virt host.
#
# Config via env:
# BB_TEST_DISTRO ubuntu | fedora | arch | alpine | nixos (default: ubuntu)
# BB_TEST_SSH_PORT host port forwarded to the guest's :22 (default: 2222)
# BB_TEST_CACHE_DIR where base images are cached (default: ~/.cache/bot-bottle-install-test)
# BB_TEST_RUN_DIR per-run scratch (overlay, seed, key, …) (default: a mktemp dir)
# BB_TEST_MEM_MB guest RAM (default: 2048)
# BB_TEST_CPUS guest vCPUs (default: 2)
# BB_TEST_DISK overlay virtual size (default: 12G)
# BB_TEST_BOOT_TIMEOUT seconds to wait for SSH after boot (default: 300)
# BB_TEST_KEEP 1 = the test cycles skip teardown, to poke at a failure
# BB_TEST_REQUIRE_BACKEND 1 = make a not-ready backend fatal (needs nested virt)
# BB_TEST_INSTALL_URL curl this install.sh in the guest instead of piping the local checkout
# BB_TEST_SKIP_VERIFY 1 = skip base-image checksum verification (not recommended)
# BOT_BOTTLE_INSTALL_SPEC passed through to install.sh (pip / git spec)
#
# Notes:
# * Needs /dev/kvm, qemu-system-x86_64, and cloud-localds (cloud-image-utils
# / cloud-utils); the nixos distro additionally needs nixos-generate. On the
# NixOS host: nix shell nixpkgs#qemu nixpkgs#cloud-utils nixpkgs#nixos-generators
# * Networking is user-mode (`-netdev user,hostfwd`) so the harness needs no
# root, no bridge, and touches no host network state. Only SSH is forwarded.
# * The cloud-image URLs in the DISTRO table are the one place to bump when a
# distro cuts a new build; each is verified against the vendor's published
# checksum at download time (guarding against truncated/corrupt pulls).
set -euo pipefail
DISTRO="${BB_TEST_DISTRO:-ubuntu}"
SSH_PORT="${BB_TEST_SSH_PORT:-2222}"
CACHE_DIR="${BB_TEST_CACHE_DIR:-${XDG_CACHE_HOME:-$HOME/.cache}/bot-bottle-install-test}"
MEM_MB="${BB_TEST_MEM_MB:-2048}"
CPUS="${BB_TEST_CPUS:-2}"
DISK="${BB_TEST_DISK:-12G}"
BOOT_TIMEOUT="${BB_TEST_BOOT_TIMEOUT:-300}"
_SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
_REPO_ROOT="$(cd "$_SCRIPT_DIR/.." && pwd)"
# Per-run scratch. Persisted across sub-commands (up/prereqs/run/status/down)
# via a marker file so `up` in one invocation and `down` in the next find the
# same VM; the test cycles set RUN_DIR themselves and never write the marker.
RUN_DIR="${BB_TEST_RUN_DIR:-}"
# Set by the test cycles, which chain the steps and suppress the per-step "next
# command" hints. _STEPS / _PASS_CLAIM / _REQUIRE_INSTALL are the per-variant
# knobs the shared cycle and teardown read.
IN_TEST=0
_STEPS=4
_PASS_CLAIM=""
_REQUIRE_INSTALL=0
# --- distro table ----------------------------------------------------
# For each distro: cloud-image URL | checksum-file URL | default SSH user |
# prerequisite-install command (run in the guest; uses sudo when user != root).
#
# The prerequisite command installs python3 + git + pipx — install.sh installs
# none of them — so `test-ready` (and the `prereqs` sub-command) drive
# install.sh down its recommended pipx path. Bump the URLs here when a distro
# publishes a newer build.
declare -A IMAGE_URL SUM_URL SSH_USER PREREQ
IMAGE_URL[ubuntu]="https://cloud-images.ubuntu.com/noble/current/noble-server-cloudimg-amd64.img"
SUM_URL[ubuntu]="https://cloud-images.ubuntu.com/noble/current/SHA256SUMS"
SSH_USER[ubuntu]="ubuntu"
PREREQ[ubuntu]="sudo apt-get update && sudo DEBIAN_FRONTEND=noninteractive apt-get install -y python3 git pipx"
IMAGE_URL[fedora]="https://download.fedoraproject.org/pub/fedora/linux/releases/44/Cloud/x86_64/images/Fedora-Cloud-Base-Generic-44-1.7.x86_64.qcow2"
SUM_URL[fedora]="https://download.fedoraproject.org/pub/fedora/linux/releases/44/Cloud/x86_64/images/Fedora-Cloud-44-1.7-x86_64-CHECKSUM"
SSH_USER[fedora]="fedora"
PREREQ[fedora]="sudo dnf install -y python3 git pipx"
IMAGE_URL[arch]="https://geo.mirror.pkgbuild.com/images/latest/Arch-Linux-x86_64-cloudimg.qcow2"
SUM_URL[arch]="https://geo.mirror.pkgbuild.com/images/latest/Arch-Linux-x86_64-cloudimg.qcow2.SHA256"
SSH_USER[arch]="arch"
PREREQ[arch]="sudo pacman -Sy --noconfirm python git python-pipx"
# Use the *nocloud_* Alpine variant, not generic_: the generic image probes
# network datasources and ignores the local NoCloud seed, so cloud-init never
# runs and the SSH key is never injected. (Only published at .0 patch levels.)
IMAGE_URL[alpine]="https://dl-cdn.alpinelinux.org/alpine/v3.21/releases/cloud/nocloud_alpine-3.21.0-x86_64-bios-cloudinit-r0.qcow2"
SUM_URL[alpine]="" # Alpine cloud images ship .sha512 only; this verifier is sha256.
SSH_USER[alpine]="alpine"
PREREQ[alpine]="sudo apk add --no-cache python3 git pipx"
# NixOS publishes no downloadable cloud qcow2 (its cloud images are Hydra-built
# AMIs), so the harness BUILDS one with nixos-generators — see build_nixos_image
# and linux-install-test-nixos.nix. It's externally-managed in its own way (no
# FHS ~/.local on PATH by default); `nix profile install` provisions the
# prerequisites into root's profile (the image enables flakes for this).
IMAGE_URL[nixos]="nix:build" # sentinel: ensure_base_image builds instead of downloading
SUM_URL[nixos]=""
SSH_USER[nixos]="root"
# Pin to a stable release: the guest's default `nixpkgs` registry is unstable,
# where pipx isn't in the binary cache and builds from source (its test suite
# currently fails to build). nixos-24.11 has these cached as substitutes.
PREREQ[nixos]="nix profile install nixpkgs/nixos-24.11#python3 nixpkgs/nixos-24.11#git nixpkgs/nixos-24.11#pipx"
ALL_DISTROS=(ubuntu fedora arch alpine nixos)
# --- guards ----------------------------------------------------------
require_linux() {
[ "$(uname -s)" = "Linux" ] \
|| { echo "error: this harness is Linux-only (uname is $(uname -s))" >&2; exit 1; }
}
require_kvm() {
[ -e /dev/kvm ] && [ -r /dev/kvm ] && [ -w /dev/kvm ] \
|| { echo "error: /dev/kvm is missing or not accessible (add yourself to the 'kvm' group)" >&2; exit 1; }
}
require_tools() {
local missing=()
command -v qemu-system-x86_64 >/dev/null 2>&1 || missing+=(qemu-system-x86_64)
command -v qemu-img >/dev/null 2>&1 || missing+=(qemu-img)
command -v cloud-localds >/dev/null 2>&1 || missing+=(cloud-localds)
command -v ssh >/dev/null 2>&1 || missing+=(ssh)
command -v curl >/dev/null 2>&1 || missing+=(curl)
if [ "${#missing[@]}" -ne 0 ]; then
echo "error: missing required tools: ${missing[*]}" >&2
echo " on NixOS: nix shell nixpkgs#qemu nixpkgs#cloud-utils nixpkgs#openssh nixpkgs#curl" >&2
exit 1
fi
}
known_distro() {
[ -n "${IMAGE_URL[$DISTRO]:-}" ] \
|| { echo "error: unknown distro '$DISTRO' (known: ${ALL_DISTROS[*]})" >&2; exit 1; }
}
# --- run-dir bookkeeping ---------------------------------------------
# The marker lets prereqs/run/status/down in separate invocations find the VM
# that `up` started. The test cycles set RUN_DIR themselves and never write it.
_marker() { echo "${TMPDIR:-/tmp}/bot-bottle-install-test.$DISTRO.run"; }
_ensure_run_dir() {
if [ -z "$RUN_DIR" ]; then
RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/bb-install-test.$DISTRO.XXXXXX")"
fi
mkdir -p "$RUN_DIR"
}
_load_run_dir() {
if [ -z "$RUN_DIR" ] && [ -f "$(_marker)" ]; then
RUN_DIR="$(cat "$(_marker)")"
fi
[ -n "$RUN_DIR" ] && [ -d "$RUN_DIR" ]
}
_ssh_key() { echo "$RUN_DIR/id_ed25519"; }
_overlay() { echo "$RUN_DIR/overlay.qcow2"; }
_seed() { echo "$RUN_DIR/seed.iso"; }
_pidfile() { echo "$RUN_DIR/qemu.pid"; }
_serial() { echo "$RUN_DIR/serial.log"; }
# --- ssh helpers -----------------------------------------------------
_ssh_opts() {
# No host-key pinning: the guest is thrown away every run.
printf '%s\0' \
-i "$(_ssh_key)" \
-p "$SSH_PORT" \
-o StrictHostKeyChecking=no \
-o UserKnownHostsFile=/dev/null \
-o LogLevel=ERROR \
-o ConnectTimeout=8 \
-o BatchMode=yes
}
guest() {
local -a opts
mapfile -d '' -t opts < <(_ssh_opts)
ssh "${opts[@]}" "${SSH_USER[$DISTRO]}@127.0.0.1" "$@"
}
wait_for_ssh() {
local deadline=$(( SECONDS + BOOT_TIMEOUT ))
echo "== waiting for SSH on 127.0.0.1:$SSH_PORT (up to ${BOOT_TIMEOUT}s) =="
while [ "$SECONDS" -lt "$deadline" ]; do
if guest true 2>/dev/null; then
echo " guest is up"
return 0
fi
# Bail early if QEMU has died — no point waiting out the timeout.
if [ -f "$(_pidfile)" ] && ! kill -0 "$(cat "$(_pidfile)")" 2>/dev/null; then
echo "error: QEMU exited before SSH came up; see $(_serial)" >&2
return 1
fi
sleep 3
done
echo "error: timed out waiting for SSH; see $(_serial)" >&2
return 1
}
# --- image cache -----------------------------------------------------
_base_image() {
# One cached file per distro. NixOS is built (not downloaded), so it has a
# fixed cache name; the rest are keyed by the image's basename so a URL bump
# lands as a new cache entry rather than a stale hit.
if [ "$DISTRO" = nixos ]; then
echo "$CACHE_DIR/nixos-built.qcow2"
return
fi
local url="${IMAGE_URL[$DISTRO]}"
echo "$CACHE_DIR/$DISTRO-$(basename "$url")"
}
# NixOS has no upstream cloud qcow2; build one with nixos-generators and copy it
# out of the (immutable, GC-able) store into the cache.
build_nixos_image() {
local out; out="$(_base_image)"
[ -f "$out" ] && { echo "== nixos base image cached: $out =="; return 0; }
command -v nixos-generate >/dev/null 2>&1 || {
echo "error: 'nixos-generate' is required to build the NixOS image" >&2
echo " run inside: nix shell nixpkgs#nixos-generators nixpkgs#qemu nixpkgs#cloud-utils" >&2
return 1
}
echo "== building NixOS cloud image with nixos-generators (first run is slow) =="
local link="$CACHE_DIR/nixos-result"
nixos-generate -f qcow --system x86_64-linux \
-c "$_SCRIPT_DIR/linux-install-test-nixos.nix" -o "$link"
cp -L "$link"/*.qcow2 "$out"
rm -f "$link"
echo " built + cached: $out"
}
verify_checksum() {
local file="$1" sums_url="${SUM_URL[$DISTRO]}" base
base="$(basename "${IMAGE_URL[$DISTRO]}")"
if [ "${BB_TEST_SKIP_VERIFY:-0}" = "1" ] || [ -z "$sums_url" ]; then
echo " checksum: SKIPPED (${sums_url:+set BB_TEST_SKIP_VERIFY=0 to enable}${sums_url:-no sums URL for $DISTRO})" >&2
return 0
fi
local want
# Vendors publish either "HASH filename" tables or a bare "HASH" (or a
# "SHA256 (file) = HASH" BSD line, e.g. Fedora). Cover all three.
local sums; sums="$(curl -fsSL "$sums_url")"
want="$(printf '%s\n' "$sums" | awk -v f="$base" '
$0 ~ f && $1 ~ /^[0-9a-fA-F]{64}$/ { print $1; exit } # GNU "hash file"
$1=="SHA256" && $0 ~ f { gsub(/[()]/,""); print $NF; exit } # BSD "SHA256 (file) = hash"
')"
[ -z "$want" ] && want="$(printf '%s\n' "$sums" | awk '/^[0-9a-fA-F]{64}$/ {print $1; exit}')"
[ -n "$want" ] || { echo "error: could not find a sha256 for $base in $sums_url" >&2; return 1; }
local got; got="$(sha256sum "$file" | awk '{print $1}')"
if [ "$want" != "$got" ]; then
echo "error: checksum mismatch for $base" >&2
echo " want $want" >&2
echo " got $got" >&2
return 1
fi
echo " checksum: OK"
}
ensure_base_image() {
mkdir -p "$CACHE_DIR"
if [ "$DISTRO" = nixos ]; then
build_nixos_image
return
fi
local base; base="$(_base_image)"
if [ -f "$base" ]; then
echo "== base image cached: $base =="
return 0
fi
echo "== downloading $DISTRO cloud image =="
echo " ${IMAGE_URL[$DISTRO]}"
# Download to a temp name and rename on success so an interrupted pull
# never poisons the cache with a truncated image.
local tmp="$base.partial"
curl -fSL --retry 3 -o "$tmp" "${IMAGE_URL[$DISTRO]}"
verify_checksum "$tmp"
mv "$tmp" "$base"
echo " cached: $base"
}
# --- cloud-init seed -------------------------------------------------
make_seed() {
ssh-keygen -t ed25519 -N '' -f "$(_ssh_key)" -q
local pub; pub="$(cat "$(_ssh_key).pub")"
local user="${SSH_USER[$DISTRO]}"
local user_data="$RUN_DIR/user-data"
if [ "$user" = "root" ]; then
# NixOS' cloud-init lands the key straight on root; no sudo needed.
cat > "$user_data" <<EOF
#cloud-config
ssh_authorized_keys:
- $pub
EOF
elif [ "$DISTRO" = alpine ]; then
# Alpine needs three things the systemd distros don't: cloud-init locks
# the account (lock_passwd), but Alpine's non-PAM sshd then refuses
# pubkey auth for a locked account — so give it a throwaway password;
# and OpenRC does not auto-start sshd after the key is injected, so
# start it via runcmd.
cat > "$user_data" <<EOF
#cloud-config
users:
- name: $user
sudo: ALL=(ALL) NOPASSWD:ALL
shell: /bin/sh
lock_passwd: false
ssh_authorized_keys:
- $pub
chpasswd:
expire: false
list: |
$user:bbtest
packages:
- sudo
runcmd:
- [ sh, -c, "rc-service sshd start 2>/dev/null || true" ]
EOF
else
cat > "$user_data" <<EOF
#cloud-config
users:
- name: $user
sudo: ALL=(ALL) NOPASSWD:ALL
shell: /bin/sh
lock_passwd: true
ssh_authorized_keys:
- $pub
EOF
fi
# NoCloud wants a meta-data with an instance-id, or cloud-init may not treat
# the seed as a new instance (the Alpine nocloud image is strict about this).
printf 'instance-id: bbtest-%s\nlocal-hostname: bbtest-%s\n' "$DISTRO" "$DISTRO" \
> "$RUN_DIR/meta-data"
cloud-localds "$(_seed)" "$user_data" "$RUN_DIR/meta-data"
}
# --- commands --------------------------------------------------------
cmd_up() {
require_linux; require_kvm; require_tools; known_distro
_ensure_run_dir
ensure_base_image
# Throwaway copy-on-write overlay: the cached base is read-only backing,
# all guest writes land in the overlay, and `down` deletes it. Resize so
# pipx + a git build have headroom (cloud-init grows the rootfs to fit).
qemu-img create -q -f qcow2 -F qcow2 -b "$(_base_image)" "$(_overlay)" "$DISK"
make_seed
echo "== booting $DISTRO VM (mem=${MEM_MB}M cpus=$CPUS, ssh -> :$SSH_PORT) =="
qemu-system-x86_64 \
-machine accel=kvm -cpu host -smp "$CPUS" -m "$MEM_MB" \
-display none -daemonize \
-pidfile "$(_pidfile)" \
-serial "file:$(_serial)" \
-drive "file=$(_overlay),if=virtio,format=qcow2" \
-drive "file=$(_seed),if=virtio,format=raw" \
-netdev "user,id=n0,hostfwd=tcp:127.0.0.1:$SSH_PORT-:22" \
-device virtio-net-pci,netdev=n0
# Only publish the marker (so a later prereqs/run/down finds this VM) when
# we aren't inside a test cycle, which manages its own RUN_DIR + teardown.
[ "$IN_TEST" = 1 ] || echo "$RUN_DIR" > "$(_marker)"
wait_for_ssh
if [ "$IN_TEST" != 1 ]; then
echo "== VM is up. Prepare it with: BB_TEST_DISTRO=$DISTRO $0 prereqs (or go straight to 'run') =="
fi
}
# Install install.sh's toolchain prerequisites (python3 + git + pipx) into the
# running VM. This is what separates `test-ready` from `test`, and it is a
# distinct sub-command so a manual up/prereqs/run/down cycle is possible.
cmd_prereqs() {
require_linux
_load_run_dir || { echo "error: no running VM for $DISTRO; run '$0 up' first" >&2; return 1; }
echo "== installing prerequisites (python3 + git + pipx) on $DISTRO =="
# Runs via the guest login shell; PREREQ is a client-side table value.
guest "${PREREQ[$DISTRO]}"
}
cmd_run() {
require_linux
_load_run_dir || { echo "error: no running VM for $DISTRO; run '$0 up' first" >&2; return 1; }
local spec_env=""
[ -n "${BOT_BOTTLE_INSTALL_SPEC:-}" ] \
&& spec_env="BOT_BOTTLE_INSTALL_SPEC='$BOT_BOTTLE_INSTALL_SPEC' "
echo "== installing bot-bottle as ${SSH_USER[$DISTRO]} =="
# Capture install.sh's exit code and full output rather than aborting on
# non-zero: on a bare host a clean prerequisite *decline* is sound, so the
# verdict step — not set -e — decides the outcome.
local rc
set +e
if [ -n "${BB_TEST_INSTALL_URL:-}" ]; then
guest "curl -fsSL '$BB_TEST_INSTALL_URL' | ${spec_env}sh" 2>&1 | tee "$RUN_DIR/install.log"
else
# Feed THIS checkout's install.sh in over stdin — the same `curl … | sh`
# shape a real user runs, and nothing is staged in the guest to leak.
guest "${spec_env}sh -s" < "$_REPO_ROOT/install.sh" 2>&1 | tee "$RUN_DIR/install.log"
fi
rc="${PIPESTATUS[0]}"
set -e
printf '%s\n' "$rc" > "$RUN_DIR/install.rc"
echo "== install.sh exited $rc =="
[ "$IN_TEST" = 1 ] \
|| echo "== verdict anytime with: BB_TEST_DISTRO=$DISTRO $0 status =="
}
# Quietly report whether a runnable bot-bottle entry point exists for the
# guest user, checking the pipx/pip locations install.sh may leave off PATH.
entry_point_runnable() {
# shellcheck disable=SC2016 # expand in the GUEST shell.
guest '
for bb in "$HOME/.local/bin/bot-bottle" "$HOME/.bot-bottle/venv/bin/bot-bottle" "$(command -v bot-bottle 2>/dev/null)"; do
[ -n "$bb" ] && [ -x "$bb" ] || continue
# --help exits 0 before any DB/migration/network work; it is the
# cheapest proof the package imports and the shim runs. (bot-bottle
# has no --version: an unknown arg would die non-zero.)
"$bb" --help >/dev/null 2>&1 && exit 0
done
exit 1
' >/dev/null 2>&1
}
# `bot-bottle doctor` in the guest, classified. doctor's own exit code
# conflates "is the install sound" with "is a backend ready to run a bottle" —
# and inside a plain VM no backend can be ready (no nested KVM/Docker), so the
# raw exit code is non-zero by design. This separates the two: an unhandled
# traceback, or a missing python/config line, is an install defect and fails;
# a not-ready backend is reported, not fatal (unless BB_TEST_REQUIRE_BACKEND=1,
# for a nested-virt host that can actually satisfy it).
doctor_in_guest() {
local out rc=0 bad=0
out="$(mktemp "${TMPDIR:-/tmp}/bb-doctor.XXXXXX")"
# shellcheck disable=SC2016 # $HOME/$bb must expand in the GUEST shell.
guest '
for bb in bot-bottle "$HOME/.local/bin/bot-bottle" "$HOME/.bot-bottle/venv/bin/bot-bottle"; do
if command -v "$bb" >/dev/null 2>&1; then
case "$bb" in
bot-bottle) : ;;
*) echo " (not on PATH — running $bb directly, as install.sh advises)" ;;
esac
exec "$bb" doctor
fi
done
echo " no bot-bottle entry point found for this user" >&2
exit 1
' >"$out" 2>&1 || rc=$?
cat "$out"
# An unhandled exception is always an install/product defect, never an
# environment fact — doctor's non-zero exit alone would not distinguish it.
if grep -q 'Traceback (most recent call last)' "$out"; then
echo " doctor crashed (traceback above) — a defect, not a missing prerequisite" >&2
bad=1
fi
grep -qE '^ok: +python:' "$out" \
|| { echo " doctor never reported a usable python" >&2; bad=1; }
grep -qE '^ok: +config:' "$out" \
|| { echo " doctor never reported a usable config dir" >&2; bad=1; }
if [ "$rc" -ne 0 ] && ! grep -qE '^(fail|warn): +backend' "$out"; then
echo " doctor failed for something other than backend readiness" >&2
bad=1
fi
local backend_ready=1
grep -qE '^fail: +backend' "$out" && backend_ready=0
rm -f "$out"
[ "$bad" -eq 0 ] || return 1
if [ "${BB_TEST_REQUIRE_BACKEND:-0}" = "1" ] && [ "$backend_ready" -eq 0 ]; then
echo " backend is not ready and BB_TEST_REQUIRE_BACKEND=1 — failing" >&2
return 1
fi
if [ "$backend_ready" -eq 1 ]; then
echo " doctor: install sound; a backend is ready"
else
echo " doctor: install sound; no backend ready (expected in a plain VM — install gate only)"
fi
return 0
}
# The prerequisite-decline messages install.sh prints via die(). On a bare host
# ANY of these means install.sh correctly refused rather than half-installing —
# a sound outcome for `test`.
PREREQ_ERR_RE='is required but was not found|or newer is required|git is required to install from|neither pipx nor a usable|externally managed \(PEP 668\)|is not on PATH'
# The verdict: is the install sound? Reads install.sh's captured exit code and
# output (from cmd_run) plus the guest's resulting state.
# - installed & runnable -> the verdict is doctor's soundness classification.
# - no entry point, but a recognized prerequisite decline, and declines are
# allowed (bare `test`, _REQUIRE_INSTALL=0) -> sound.
# - anything else -> not sound.
install_verdict() {
local rc=""
[ -f "$RUN_DIR/install.rc" ] && rc="$(cat "$RUN_DIR/install.rc")"
if [ "${rc:-1}" = 0 ] && entry_point_runnable; then
echo "doctor (in guest):"
doctor_in_guest
return $?
fi
if [ "${_REQUIRE_INSTALL:-0}" != "1" ] \
&& [ -n "$rc" ] && [ "$rc" != 0 ] \
&& [ -f "$RUN_DIR/install.log" ] \
&& grep -Eiq "$PREREQ_ERR_RE" "$RUN_DIR/install.log"; then
echo " install.sh declined with an actionable prerequisite error (rc=$rc)"
echo " — sound on a bare host; run 'test-ready' (or 'prereqs') to install."
return 0
fi
if [ "${_REQUIRE_INSTALL:-0}" = "1" ]; then
echo " prerequisites were provisioned, but install.sh left no runnable entry point (rc=${rc:-?})" >&2
else
echo " install.sh neither installed nor gave a recognized prerequisite error (rc=${rc:-?})" >&2
fi
return 1
}
cmd_status() {
require_linux
if ! _load_run_dir; then
echo "vm: no running VM for $DISTRO"
return 0
fi
if [ -f "$(_pidfile)" ] && kill -0 "$(cat "$(_pidfile)")" 2>/dev/null; then
echo "vm: $DISTRO running (pid $(cat "$(_pidfile)"), ssh :$SSH_PORT)"
else
echo "vm: $DISTRO run-dir present but QEMU not alive"
return 1
fi
if install_verdict; then
echo "OK[$DISTRO]: the install is sound"
return 0
fi
echo "FAIL[$DISTRO]: the install is not sound (see above)" >&2
return 1
}
cmd_down() {
require_linux
if ! _load_run_dir; then
echo "$DISTRO: nothing running"
return 0
fi
if [ -f "$(_pidfile)" ]; then
local pid; pid="$(cat "$(_pidfile)")"
if kill -0 "$pid" 2>/dev/null; then
kill "$pid" 2>/dev/null || true
for _ in 1 2 3 4 5; do kill -0 "$pid" 2>/dev/null || break; sleep 1; done
kill -9 "$pid" 2>/dev/null || true
fi
fi
# Deleting the overlay + seed is the reset; the read-only base stays cached.
rm -f "$(_overlay)" "$(_seed)" "$(_ssh_key)" "$(_ssh_key).pub" \
"$RUN_DIR/user-data" "$RUN_DIR/meta-data" \
"$RUN_DIR/install.rc" "$RUN_DIR/install.log" "$(_serial)"
# Only remove a scratch dir we created (leave a user-provided one alone).
[ -n "${BB_TEST_RUN_DIR:-}" ] || rmdir "$RUN_DIR" 2>/dev/null || true
rm -f "$(_marker)"
echo "removed $DISTRO VM and its overlay — install surface is clean."
}
cmd_ssh() {
require_linux
_load_run_dir || { echo "error: no running VM for $DISTRO" >&2; return 1; }
local -a opts
mapfile -d '' -t opts < <(_ssh_opts)
exec ssh -t "${opts[@]}" "${SSH_USER[$DISTRO]}@127.0.0.1"
}
# Teardown half of the test cycles, armed the moment the VM exists so a failure
# or a Ctrl-C still leaves nothing running.
_test_teardown() {
local rc=$?
trap - EXIT INT TERM
if [ "${BB_TEST_KEEP:-0}" = "1" ]; then
echo
echo "== [$_STEPS/$_STEPS] down: SKIPPED (BB_TEST_KEEP=1) =="
echo " VM still up; remove with: BB_TEST_DISTRO=$DISTRO $0 down"
echo " ssh in with: BB_TEST_RUN_DIR=$RUN_DIR BB_TEST_DISTRO=$DISTRO $0 ssh"
exit "$rc"
fi
echo
echo "== [$_STEPS/$_STEPS] down =="
cmd_down || rc=1
if [ "$rc" -eq 0 ]; then
echo
echo "PASS[$DISTRO]: $_PASS_CLAIM"
else
echo
echo "FAIL[$DISTRO]: see above (the VM was torn down regardless)." >&2
fi
exit "$rc"
}
# The two variants differ only in whether the prerequisites get installed
# before install.sh runs, which is exactly the question each one asks:
#
# test a bare host — assert install.sh is SOUND (installs cleanly, or
# declines with an actionable prerequisite error).
# test-ready prerequisites satisfied — assert install.sh actually LANDS.
_test_cycle() {
local with_prereqs="$1"
require_linux; require_kvm; require_tools; known_distro
IN_TEST=1
RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/bb-install-test.$DISTRO.XXXXXX")"
# Arm teardown BEFORE cmd_up: its wait_for_ssh can fail after QEMU is
# already running (e.g. a guest that never opens SSH), and without the trap
# in place that would leak the VM.
trap _test_teardown EXIT INT TERM
echo "== [1/$_STEPS] up ($DISTRO) =="
cmd_up
local step=2
if [ "$with_prereqs" = 1 ]; then
echo
echo "== [$step/$_STEPS] prereqs =="
cmd_prereqs || { echo "error: could not install prerequisites (see above)." >&2; return 1; }
step=$(( step + 1 ))
fi
echo
echo "== [$step/$_STEPS] run =="
cmd_run
step=$(( step + 1 ))
echo
echo "== [$step/$_STEPS] verdict =="
# install.sh exits 0 even when doctor reports unmet prerequisites, so the
# install succeeding is not the verdict — this is.
cmd_status || {
echo "error: the install is not sound for $DISTRO (see above)." >&2
echo " re-run with BB_TEST_KEEP=1 to keep the VM and dig in." >&2
return 1
}
}
cmd_test() {
_STEPS=4
_PASS_CLAIM="on a bare $DISTRO host, install.sh behaves soundly."
_test_cycle 0
}
cmd_test_ready() {
_STEPS=5
_PASS_CLAIM="a $DISTRO host with prerequisites satisfied installs bot-bottle cleanly."
# Prerequisites are provisioned, so a graceful decline is no longer an
# acceptable outcome — the install must actually land.
_REQUIRE_INSTALL=1
_test_cycle 1
}
cmd_test_all() {
require_linux; require_kvm; require_tools
local -a passed=() failed=()
local port="$SSH_PORT"
local script
script="$_SCRIPT_DIR/$(basename "${BASH_SOURCE[0]}")"
# Every distro × both variants. Each cell gets its own forwarded port so a
# leftover from a prior cell can't collide, and its own subshell so one
# cell's failure (or teardown trap) doesn't abort the matrix.
for sub in test test-ready; do
for d in "${ALL_DISTROS[@]}"; do
echo
echo "########################################################"
echo "# $d ($sub)"
echo "########################################################"
if ( DISTRO="$d" SSH_PORT="$port" \
BB_TEST_DISTRO="$d" BB_TEST_SSH_PORT="$port" \
bash "$script" "$sub" ); then
passed+=("$d/$sub")
else
failed+=("$d/$sub")
fi
port=$(( port + 1 ))
done
done
echo
echo "== matrix summary =="
echo " PASS: ${passed[*]:-(none)}"
echo " FAIL: ${failed[*]:-(none)}"
[ "${#failed[@]}" -eq 0 ]
}
case "${1:-}" in
test) cmd_test ;;
test-ready) cmd_test_ready ;;
test-all) cmd_test_all ;;
up) cmd_up ;;
prereqs) cmd_prereqs ;;
run) cmd_run ;;
status) cmd_status ;;
down) cmd_down ;;
ssh) cmd_ssh ;;
*) echo "usage: $0 {test|test-ready|test-all|up|prereqs|run|status|down|ssh} (distro via BB_TEST_DISTRO)" >&2; exit 2 ;;
esac
-531
View File
@@ -1,531 +0,0 @@
#!/usr/bin/env bash
# Clean-install test harness for the macOS (Apple `container`) path.
#
# Exercises install.sh the way a brand-new user would, inside a throwaway
# macOS account you create and delete from the CLI. install.sh's entire
# footprint is user-home-local — the pipx venv under ~/.local, or the private
# venv at ~/.bot-bottle/venv plus a ~/.local/bin symlink, and the ~/.bot-bottle
# config dir. It writes no shell-profile PATH line, and never installs the
# backend (see
# the header of install.sh), so deleting the user is a complete,
# deterministic reset of everything the installer touched. The Apple
# `container` runtime is a HOST prerequisite installed once and kept;
# `deep-reset` is the rare escape hatch that also removes it.
#
# Why a throwaway user and not a disposable VM: bot-bottle's default macOS
# backend is Apple `container`, which runs each container in its own
# Virtualization.framework microVM. Running that backend inside a macOS
# guest VM needs nested virtualization, which Apple gates to M3+ silicon.
# On M1/M2 a separate user account is the only way to get a clean $HOME
# while still reaching the real host backend. Full rationale in
# docs/research/testing-clean-install-on-macos.md.
#
# Usage:
# sudo ./scripts/macos-install-test.sh test # up -> run -> status -> down
# sudo ./scripts/macos-install-test.sh test-ready # ... with prereqs satisfied
# sudo ./scripts/macos-install-test.sh up # create the throwaway user
# sudo ./scripts/macos-install-test.sh run # run install.sh (+doctor) as it
# sudo ./scripts/macos-install-test.sh prereqs # start the per-user backend svc
# ./scripts/macos-install-test.sh status # user present? backend ready?
# sudo ./scripts/macos-install-test.sh down # delete user + home (the reset)
# sudo ./scripts/macos-install-test.sh deep-reset # ALSO uninstall host `container`
#
# TWO TEST VARIANTS, because "does the installer work" and "can a new user
# actually run a bottle" are different questions with different answers:
#
# test A macOS system WITHOUT the prerequisites set up for this user —
# the default state of any brand-new account, since the Apple
# `container` service is per-user. Asserts the INSTALL is sound
# (entry point present, doctor runs without crashing, python and
# config check out). Backend readiness is reported, not required:
# install.sh does not provide a backend and cannot regress one,
# so failing on it would make this permanently red.
#
# test-ready A macOS system WITH the prerequisites satisfied — it starts the
# per-user `container` service (and installs its guest kernel)
# for the throwaway user first, then demands doctor go fully
# green, backend included. This is the end-to-end claim: a new
# user can actually run a bottle. Note it downloads a guest
# kernel per run, because the previous run's went with the
# deleted home; BB_TEST_KERNEL_INSTALL=0 skips that.
#
# Both refuse to start if the account already exists (a reused home is not a
# clean install), and both tear the account down on the way out however they
# exit, so a failed run never leaves an orphan behind. install.sh itself exits
# 0 when doctor reports unmet prerequisites, so either variant is a stricter
# gate than running the installer by hand.
#
# Config via env:
# BB_TEST_USER account short name (default: bbtest)
# BB_TEST_FULLNAME account full name (default: "bot-bottle install test")
# BB_TEST_ADMIN 1=admin (reach container svc), 0=standard (default: 1)
# BB_TEST_INSTALL_URL curl this install.sh instead of piping the local checkout
# BB_TEST_KEEP 1=`test` skips its teardown, to poke at a failure
# BB_TEST_REQUIRE_BACKEND 1=`test` also fails when no backend is ready
# BB_TEST_KERNEL_INSTALL 0=`test-ready` skips the guest-kernel download
# BB_TEST_REPO_URL https repo the throwaway user clones (default: this one)
# BOT_BOTTLE_INSTALL_SPEC passed through to install.sh (pip / git spec)
#
# Notes:
# * Run from a normally-booted admin session. Grant Terminal *Full Disk
# Access* (System Settings -> Privacy & Security) or `down` half-fails
# with error -14120 and leaves an orphaned account.
# * `sysadminctl` always exits 0 even on failure, so `up`/`down` verify
# the result with `dscl` and fail loudly on a mismatch.
# * The account is created without a password: `run` drives it headlessly
# via `sudo -u`, which never needs the target's password. The account
# cannot GUI-login, which this harness does not require.
# * `run` covers the installer + `bot-bottle doctor`. Actually launching a
# bottle from the throwaway user may need a full launchd user session
# (`launchctl asuser`); on M1/M2 the backend can't run under nested virt
# anyway, so this harness stops at install + doctor.
set -euo pipefail
USER_NAME="${BB_TEST_USER:-bbtest}"
FULL_NAME="${BB_TEST_FULLNAME:-bot-bottle install test}"
ADMIN="${BB_TEST_ADMIN:-1}"
# https, not the ssh `origin`: the throwaway user has no key of ours.
BB_TEST_REPO_URL="${BB_TEST_REPO_URL:-https://gitea.dideric.is/didericis/bot-bottle.git}"
_SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
_REPO_ROOT="$(cd "$_SCRIPT_DIR/.." && pwd)"
# Set by `test`, which chains the steps itself and so suppresses the
# "here's the next command to run" hints the individual steps print.
IN_TEST=0
# --- guards ----------------------------------------------------------
require_macos() {
[ "$(uname -s)" = "Darwin" ] \
|| { echo "error: this harness is macOS-only (uname is $(uname -s))" >&2; exit 1; }
}
require_root() {
if [ "$(id -u)" -ne 0 ]; then
echo "error: '$1' needs root; re-run under sudo" >&2
exit 1
fi
}
user_exists() { dscl . -read "/Users/$USER_NAME" >/dev/null 2>&1; }
user_uid() {
dscl . -read "/Users/$USER_NAME" UniqueID 2>/dev/null | awk '{print $2}'
}
# Whether we can enter the target user's launchd domain.
#
# This matters more than it sounds. `container system start` registers
# com.apple.container.apiserver as a per-USER launchd agent and then talks to
# it over XPC. Under plain `sudo -u` the caller is still in root's bootstrap
# namespace, so the lookup crosses domains and the apiserver answers
# invalidState: "unauthorized request"
# even though it launched fine. `launchctl asuser <uid>` puts the whole
# invocation in that user's domain, which is also what a real user gets from
# Terminal — so it is the more faithful way to run *everything* here, not just
# the service start. Falls back to plain sudo when unavailable, since a
# never-logged-in account may have no bootstrappable domain at all.
_SESSION_OK=""
user_session_available() {
if [ -z "$_SESSION_OK" ]; then
local uid
uid="$(user_uid)"
if [ -n "$uid" ] && launchctl asuser "$uid" true >/dev/null 2>&1; then
_SESSION_OK=1
else
_SESSION_OK=0
fi
fi
[ "$_SESSION_OK" = "1" ]
}
# Run a shell snippet as the throwaway user in a fresh login shell.
#
# The snippet goes in on stdin, never as an argument. `sudo -i` joins its argv
# into a single string and hands that to the login shell's -c, so quoting in an
# argument is not preserved and a newline ends the command outright ("sh: -c:
# line 1: syntax error: unexpected end of file"). Feeding `sh -s` from stdin
# sidesteps the joining entirely and lets snippets be multi-line.
run_as_user() {
if user_session_available; then
printf '%s\n' "$1" | launchctl asuser "$(user_uid)" sudo -u "$USER_NAME" -i sh -s
else
printf '%s\n' "$1" | sudo -u "$USER_NAME" -i sh -s
fi
}
# `bot-bottle doctor` as the throwaway user. Non-zero when no entry point was
# installed at all, or when doctor itself is unhappy.
#
# Deliberately does NOT require `bot-bottle` on PATH: install.sh prints the
# PATH line rather than editing a shell profile, so on a fresh account the
# entry point is installed and working but not on PATH. Demanding PATH here
# would fail every run for a reason the installer intends.
#
# Doctor's own exit code conflates two unrelated things: whether the install
# works, and whether this user has a backend ready to run a bottle. Those need
# different verdicts here. install.sh explicitly does not install a backend,
# and the Apple `container` service is per-USER — its state lives in
# ~/Library/Application Support/com.apple.container — so a brand-new account
# never has one running, no matter how correct the install is. Failing on that
# would leave `test` permanently red for something the installer cannot fix
# and cannot regress. So: assert the install, report the backend.
# BB_TEST_REQUIRE_BACKEND=1 makes backend readiness fatal too.
doctor_as_user() {
local out rc=0 bad=0
out="$(mktemp "${TMPDIR:-/tmp}/bb-doctor.XXXXXX")"
# shellcheck disable=SC2016 # $HOME/$bb must expand in the *target* user's
# shell, not in this one — that's the whole point of the single quotes.
run_as_user '
for bb in bot-bottle "$HOME/.local/bin/bot-bottle" "$HOME/.bot-bottle/venv/bin/bot-bottle"; do
if command -v "$bb" >/dev/null 2>&1; then
case "$bb" in
bot-bottle) : ;;
*) echo " (not on PATH — running $bb directly, as install.sh advises)" ;;
esac
exec "$bb" doctor
fi
done
echo " no bot-bottle entry point found for this user" >&2
exit 1
' >"$out" 2>&1 || rc=$?
cat "$out"
# An unhandled exception is always an install/product defect, never an
# environment fact — this is exactly how the PermissionError on `ip` showed
# up, and doctor's non-zero exit alone would not have distinguished it.
if grep -q 'Traceback (most recent call last)' "$out"; then
echo " doctor crashed (traceback above) — a defect, not a missing prerequisite" >&2
bad=1
fi
grep -qE '^ok: +python:' "$out" \
|| { echo " doctor never reported a usable python" >&2; bad=1; }
grep -qE '^ok: +config:' "$out" \
|| { echo " doctor never reported a usable config dir" >&2; bad=1; }
if [ "$rc" -ne 0 ] && ! grep -qE '^(fail|warn): +backend' "$out"; then
echo " doctor failed for something other than backend readiness" >&2
bad=1
fi
local backend_ready=1
grep -qE '^fail: +backend' "$out" && backend_ready=0
rm -f "$out"
[ "$bad" -eq 0 ] || return 1
if [ "$backend_ready" -eq 0 ]; then
if [ "${BB_TEST_REQUIRE_BACKEND:-0}" = "1" ]; then
echo " no backend is ready and BB_TEST_REQUIRE_BACKEND=1" >&2
return 1
fi
echo " note: install is sound; no backend ready for this user."
echo " Expected on a fresh account — the Apple 'container' service is"
echo " per-user and needs one 'container system start'. Set"
echo " BB_TEST_REQUIRE_BACKEND=1 to treat this as a failure."
fi
return 0
}
# --- commands --------------------------------------------------------
cmd_up() {
require_macos
require_root up
if user_exists; then
echo "$USER_NAME already exists; nothing to do (run 'down' first to reset)"
return 0
fi
local admin_flag=()
[ "$ADMIN" = "1" ] && admin_flag=(-admin)
# No -password: the account is only ever driven headlessly via `sudo -u`,
# which doesn't need one. sysadminctl warns about FileVault here; that's
# irrelevant to a headless test account.
sysadminctl -addUser "$USER_NAME" -fullName "$FULL_NAME" "${admin_flag[@]}" || true
# sysadminctl exits 0 regardless of outcome, so confirm the account landed.
user_exists || { echo "error: failed to create $USER_NAME" >&2; return 1; }
if [ "$IN_TEST" = 1 ]; then
echo "created $USER_NAME (admin=$ADMIN)"
else
echo "created $USER_NAME (admin=$ADMIN). Install into it with: sudo $0 run"
fi
}
# The package spec to install. install.sh's own default is the repo's DEFAULT
# BRANCH, which would mean piping this checkout's installer into the throwaway
# user and then installing main's code — verifying the PR's install.sh against
# a package that doesn't contain the PR. Pin to the branch under test instead.
#
# The branch has to be pushed: the throwaway user clones over https and cannot
# read this checkout (mode-700 home, and no SSH key for the ssh remote), so
# what it gets is whatever the remote has. Say so when that differs from here.
resolve_install_spec() {
if [ -n "${BOT_BOTTLE_INSTALL_SPEC:-}" ]; then
echo "$BOT_BOTTLE_INSTALL_SPEC"
return 0
fi
local branch
branch="$(git -C "$_REPO_ROOT" rev-parse --abbrev-ref HEAD 2>/dev/null)" || branch=""
if [ -z "$branch" ] || [ "$branch" = "HEAD" ]; then
echo "note: not on a named branch; installing the repo default" >&2
echo "git+${BB_TEST_REPO_URL}"
return 0
fi
if [ -n "$(git -C "$_REPO_ROOT" status --porcelain 2>/dev/null)" ]; then
echo "warning: working tree is dirty — the throwaway user installs" >&2
echo " $branch as pushed, not what is on disk here." >&2
elif ! git -C "$_REPO_ROOT" diff --quiet "@{upstream}" 2>/dev/null; then
echo "warning: $branch differs from its upstream — push first, or the" >&2
echo " throwaway user installs stale code." >&2
fi
echo "git+${BB_TEST_REPO_URL}@${branch}"
}
cmd_run() {
require_macos
require_root run
user_exists || { echo "error: $USER_NAME does not exist; run 'sudo $0 up' first" >&2; return 1; }
local spec
spec="$(resolve_install_spec)"
echo "== installing bot-bottle as $USER_NAME =="
echo "== spec: $spec =="
if [ -n "${BB_TEST_INSTALL_URL:-}" ]; then
run_as_user "BOT_BOTTLE_INSTALL_SPEC='$spec' curl -fsSL '$BB_TEST_INSTALL_URL' | sh"
else
# Test THIS checkout's install.sh, not the published one, so a PR is
# verifiable before it lands. Feed it in on stdin rather than staging a
# copy somewhere the throwaway user can read: the redirect is opened by
# root before sudo drops privileges, so the tester's mode-700 home is a
# non-issue, there's no temp file to leak if the run is interrupted, and
# `sh -s` is the same shape as the documented `curl … | sh` install.
#
# The spec is prepended to the stream as an export rather than passed
# as an argument, for the same argv-joining reason as run_as_user.
{
printf "export BOT_BOTTLE_INSTALL_SPEC='%s'\n" "$spec"
cat "$_REPO_ROOT/install.sh"
} | sudo -u "$USER_NAME" -i sh -s
fi
[ "$IN_TEST" = 1 ] \
|| echo "== install.sh runs 'doctor' itself; re-check anytime with: $0 status =="
}
# Informational, with one teeth-bearing case: when it can actually reach
# doctor (root, account present) its exit status is doctor's, so `test` and
# any other caller can use it as the post-install assertion.
cmd_status() {
require_macos
local rc=0
if user_exists; then
echo "user: $USER_NAME present"
if [ "$(id -u)" -eq 0 ]; then
echo "doctor (as $USER_NAME):"
doctor_as_user || rc=1
else
echo " (re-run under sudo to run 'bot-bottle doctor' as $USER_NAME)"
fi
else
echo "user: $USER_NAME absent"
fi
if command -v container >/dev/null 2>&1; then
echo "backend: apple 'container' present ($(container --version 2>/dev/null | head -1))"
else
echo "backend: apple 'container' NOT on PATH (host prerequisite; install once)"
fi
return "$rc"
}
cmd_down() {
require_macos
require_root down
if ! user_exists; then
echo "$USER_NAME not present; nothing to remove"
return 0
fi
# A plain -deleteUser removes the home dir, which is the whole reset.
# -secure is a no-op on modern macOS (secure erase of the home folder
# was removed in Sierra), so it buys nothing here.
sysadminctl -deleteUser "$USER_NAME" || true
if user_exists; then
echo "error: $USER_NAME still present after delete." >&2
echo " - grant Terminal Full Disk Access (System Settings > Privacy & Security), or" >&2
echo " - it may hold the last Secure Token (won't happen while another admin exists)" >&2
return 1
fi
echo "removed $USER_NAME and its home — install surface is clean."
}
# Satisfy the throwaway user's backend prerequisites.
#
# `bot-bottle backend setup --backend=macos-container` deliberately does NOT do
# this: it only checks, then tells you to run `container system start` yourself
# (see backend/macos_container/setup.py — "no privileged host setup required").
# So the harness runs the command the product asks for, as the user who needs
# it, which is the whole prerequisite for the macOS backend given the CLI is
# already installed host-wide.
cmd_prereqs() {
require_macos
require_root prereqs
user_exists || { echo "error: $USER_NAME does not exist" >&2; return 1; }
command -v container >/dev/null 2>&1 || {
echo "error: the Apple 'container' CLI is not on PATH. It is a HOST" >&2
echo " prerequisite this harness does not install; see the header." >&2
return 1
}
echo "starting the per-user Apple 'container' service for $USER_NAME"
user_session_available || {
echo "warning: no launchd session for $USER_NAME; the service start is" >&2
echo " likely to fail with 'unauthorized request'. See user_session_available." >&2
}
# The kernel flag is not optional for a headless run. `container system
# start` PROMPTS for the default Linux kernel when neither flag is given
# ("default: prompt user"), and a throwaway account has no kernel because
# it lives in the per-user app root — so the prompt always fires here and
# dies on "failed to read user input" with no TTY.
#
# Enabling it is also the honest choice for what this variant claims: the
# backend cannot actually run a bottle without a guest kernel. The cost is
# that every test-ready run downloads one afresh, since the previous run's
# copy went with the deleted home. BB_TEST_KERNEL_INSTALL=0 skips the
# download when you only care that the service comes up.
local kernel_flag=--enable-kernel-install
[ "${BB_TEST_KERNEL_INSTALL:-1}" = "0" ] && kernel_flag=--disable-kernel-install
echo " ($kernel_flag)"
run_as_user "container system start $kernel_flag" || {
echo "error: 'container system start' failed for $USER_NAME" >&2
echo " invalidState/\"unauthorized request\" means the CLI could not reach" >&2
echo " its per-user apiserver over XPC — a never-GUI-logged-in account may" >&2
echo " have no usable launchd domain; log into $USER_NAME once to get one." >&2
echo " A download failure means the guest kernel could not be fetched; retry" >&2
echo " with network, or BB_TEST_KERNEL_INSTALL=0 to skip it." >&2
return 1
}
run_as_user 'container system status' || {
echo "error: the service is still not running for $USER_NAME" >&2
return 1
}
}
# Teardown half of the test cycles, installed as an EXIT trap the moment the
# account exists so a failure — or a Ctrl-C — still leaves the machine clean.
_test_teardown() {
local rc=$?
trap - EXIT INT TERM
if [ "${BB_TEST_KEEP:-0}" = "1" ]; then
echo
echo "== [$_STEPS/$_STEPS] down: SKIPPED (BB_TEST_KEEP=1) =="
echo " $USER_NAME is still around; remove it with: sudo $0 down"
exit "$rc"
fi
echo
echo "== [$_STEPS/$_STEPS] down =="
cmd_down || rc=1
if [ "$rc" -eq 0 ]; then
echo
echo "PASS: $_PASS_CLAIM"
else
echo
echo "FAIL: see above (the throwaway account was torn down regardless)." >&2
fi
exit "$rc"
}
# The two variants differ only in whether the backend prerequisites get
# satisfied before doctor runs, which is exactly the question each one asks:
#
# test a brand-new user on a machine where nothing has been set up for
# them. Asserts the INSTALL is sound; backend readiness is
# reported, not required, because install.sh does not provide a
# backend and cannot regress one.
# test-ready the same user with the documented prerequisites satisfied.
# Asserts doctor goes fully green, backend included — the
# end-to-end "a new user can actually run a bottle" claim.
_test_cycle() {
local with_prereqs="$1"
require_macos
require_root test
# A pre-existing account means a pre-existing home, which is the one thing
# this harness exists to rule out. Don't silently test a dirty install.
if user_exists; then
echo "error: $USER_NAME already exists, so this would not be a clean install." >&2
echo " reset first: sudo $0 down" >&2
return 1
fi
IN_TEST=1
echo "== [1/$_STEPS] up =="
cmd_up
trap _test_teardown EXIT INT TERM
echo
echo "== [2/$_STEPS] run =="
cmd_run
if [ "$with_prereqs" = 1 ]; then
echo
echo "== [3/$_STEPS] prereqs =="
cmd_prereqs || {
echo "error: could not satisfy the backend prerequisites (see above)." >&2
return 1
}
fi
echo
echo "== [$((_STEPS - 1))/$_STEPS] status =="
# install.sh exits 0 even when doctor reports unmet prerequisites, so the
# install succeeding is not the verdict — this is.
cmd_status || {
echo "error: doctor is unhappy for a freshly installed user (see above)." >&2
echo " re-run with BB_TEST_KEEP=1 to keep $USER_NAME around and dig in." >&2
return 1
}
}
cmd_test() {
_STEPS=4
_PASS_CLAIM="a brand-new user can install bot-bottle; the install is sound."
_test_cycle 0
}
cmd_test_ready() {
_STEPS=5
_PASS_CLAIM="a brand-new user with the prerequisites met passes doctor outright."
# With the prerequisites satisfied there is no excuse for an unready
# backend, so this variant demands the thing `test` only reports.
BB_TEST_REQUIRE_BACKEND=1
_test_cycle 1
}
cmd_deep_reset() {
require_macos
require_root deep-reset
# Remove the user first (idempotent), then the HOST-level container
# runtime that a user deletion leaves behind under /usr/local + launchd.
cmd_down || true
if command -v container >/dev/null 2>&1; then
# The service can run in more than one launchd context (the invoking
# user's and root's), so stop both, best-effort.
[ -n "${SUDO_USER:-}" ] && sudo -u "$SUDO_USER" container system stop 2>/dev/null || true
container system stop 2>/dev/null || true
if [ -x /usr/local/bin/uninstall-container.sh ]; then
/usr/local/bin/uninstall-container.sh -d || true
echo "uninstalled the host Apple 'container' runtime"
else
echo "note: /usr/local/bin/uninstall-container.sh not found; runtime left as-is" >&2
fi
else
echo "no 'container' runtime on PATH; nothing further to remove"
fi
}
case "${1:-}" in
test) cmd_test ;;
test-ready) cmd_test_ready ;;
up) cmd_up ;;
run) cmd_run ;;
prereqs) cmd_prereqs ;;
status) cmd_status ;;
down) cmd_down ;;
deep-reset) cmd_deep_reset ;;
*) echo "usage: $0 {test|test-ready|up|run|prereqs|status|down|deep-reset}" >&2 ; exit 2 ;;
esac
+2 -2
View File
@@ -95,7 +95,7 @@ BOT_BOTTLE_RUN_CANARIES=1 python -m scripts.unittest_gate \
```
4. Skip guards live in `tests._backend` and gate on the backend's own
readiness check, `bot_bottle.backend.has_backend` — the same probe
behind `bot-bottle backend status`:
behind `./cli.py backend status`:
- Backend-agnostic tests (go through `get_bottle_backend()`) decorate
the class with `@skip_unless_selected_backend_available()` — the test
runs against whichever backend `BOT_BOTTLE_BACKEND` selects and skips
@@ -105,7 +105,7 @@ BOT_BOTTLE_RUN_CANARIES=1 python -m scripts.unittest_gate \
`backend.docker.*`, …) decorate with `@skip_unless_backend("docker")`
so they no-op under a run targeting a different backend.
Each CI integration job runs `bot-bottle backend status --backend=<name>`
Each CI integration job runs `./cli.py backend status --backend=<name>`
as a preflight, which prints a clear per-check summary and exits non-zero
when the backend is missing — so absent infrastructure fails the job
instead of hiding among per-test `unittest.skip` lines.
+1 -1
View File
@@ -2,7 +2,7 @@
Each integration test targets the backend named by ``BOT_BOTTLE_BACKEND``
(default ``docker``) and gates on that backend's full readiness check —
``is_backend_ready()`` (equivalent to ``bot-bottle backend status``), not just
``is_backend_ready()`` (equivalent to ``./cli.py backend status``), not just
a binary-on-PATH probe. When the backend is not ready, diagnostic output is
printed during test discovery so the operator sees a concrete reason for each
skip.
+1 -1
View File
@@ -1,7 +1,7 @@
"""Tests for the backend-aware skip guards in ``tests/_backend.py``.
The guards delegate their readiness check to
``bot_bottle.backend.is_backend_ready`` (the probe behind ``bot-bottle backend
``bot_bottle.backend.is_backend_ready`` (the probe behind ``./cli.py backend
status``); here that probe is mocked so the unit job asserts the
selection/skip logic without either backend present on the runner.
"""
+1 -1
View File
@@ -1,6 +1,6 @@
"""Unit: the generic `backend` CLI command.
`bot-bottle backend {setup,status} [--backend=NAME]` resolves a backend
`./cli.py backend {setup,status} [--backend=NAME]` resolves a backend
and dispatches to its `setup()` / `status()` classmethods no
backend-specific commands. Also covers the docker backend's own
setup/status logic (the firecracker path is exercised by
+2 -6
View File
@@ -17,7 +17,6 @@ from bot_bottle.cli.commands import cleanup as cmd
def _make_backend(empty: bool = True):
backend = MagicMock()
plan = MagicMock(empty=empty)
plan.intersect.return_value = plan
backend.prepare_cleanup.return_value = plan
backend.cleanup = MagicMock()
return backend, plan
@@ -136,12 +135,10 @@ class TestCmdCleanup(unittest.TestCase):
docker.cleanup.assert_called_once_with(docker_plan)
fc.cleanup.assert_not_called()
def test_executes_only_displayed_resources_still_current(self):
def test_executes_refreshed_plan_after_confirmation(self):
backend = MagicMock()
preview = MagicMock(empty=False)
refreshed = MagicMock(empty=False)
approved = MagicMock(empty=False)
preview.intersect.return_value = approved
backend.prepare_cleanup.side_effect = [preview, refreshed]
with patch.object(
@@ -155,8 +152,7 @@ class TestCmdCleanup(unittest.TestCase):
):
self.assertEqual(0, cmd.cmd_cleanup([]))
preview.intersect.assert_called_once_with(refreshed)
backend.cleanup.assert_called_once_with(approved)
backend.cleanup.assert_called_once_with(refreshed)
if __name__ == "__main__":
-11
View File
@@ -44,17 +44,6 @@ class TestProjectNaming(unittest.TestCase):
class TestComposeProjectListing(unittest.TestCase):
def test_compose_ls_empty_when_docker_unusable(self):
# Missing is the obvious case; present-but-not-executable raises
# PermissionError instead, which must not escape as a crash.
for exc in (FileNotFoundError, PermissionError(13, "Permission denied", "docker")):
with self.subTest(exc=type(exc).__name__):
with mock.patch(
"bot_bottle.backend.docker.compose.subprocess.run",
side_effect=exc,
):
self.assertEqual([], list_compose_projects())
def test_compose_ls_error_warns_by_default(self):
with (
mock.patch(
-44
View File
@@ -3,11 +3,9 @@
from __future__ import annotations
import sqlite3
import stat
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch
from bot_bottle.store.db_store import DbStore
from bot_bottle.store.migrations import TableMigrations
@@ -24,48 +22,6 @@ class TestDbStoreIsMigrated(unittest.TestCase):
store = _store(Path(d))
self.assertFalse(store.is_migrated())
def test_creates_private_directory_and_database_before_first_open(self):
with tempfile.TemporaryDirectory() as d:
parent = Path(d) / "store"
store = _store(parent)
self.assertEqual(0o700, stat.S_IMODE(parent.stat().st_mode))
self.assertFalse(store.db_path.exists())
store.migrate()
self.assertEqual(0o600, stat.S_IMODE(store.db_path.stat().st_mode))
def test_repairs_existing_permissions(self):
with tempfile.TemporaryDirectory() as d:
parent = Path(d) / "store"
parent.mkdir(mode=0o755)
db_path = parent / "test.db"
db_path.touch(mode=0o644)
store = _store(parent)
self.assertEqual(db_path, store.db_path)
self.assertEqual(0o700, stat.S_IMODE(parent.stat().st_mode))
self.assertEqual(0o600, stat.S_IMODE(db_path.stat().st_mode))
def test_permission_repair_failure_is_not_suppressed(self):
with tempfile.TemporaryDirectory() as d:
parent = Path(d) / "store"
parent.mkdir()
(parent / "test.db").touch()
with patch("os.fchmod", side_effect=OSError("denied")):
with self.assertRaisesRegex(OSError, "denied"):
_store(parent)
def test_rejects_database_symlink_without_changing_target(self):
with tempfile.TemporaryDirectory() as d:
parent = Path(d) / "store"
parent.mkdir()
target = Path(d) / "target"
target.touch(mode=0o644)
(parent / "test.db").symlink_to(target)
with self.assertRaises(OSError):
_store(parent)
self.assertEqual(0o644, stat.S_IMODE(target.stat().st_mode))
def test_returns_false_when_schema_versions_missing(self):
# DB file exists but has no schema_versions table → OperationalError → False.
with tempfile.TemporaryDirectory() as d:
-35
View File
@@ -14,12 +14,9 @@ from __future__ import annotations
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch
from tests.unit import use_bottle_root
from bot_bottle import bottle_state
from bot_bottle.backend import EnumerationError
from bot_bottle.backend.docker import cleanup
from bot_bottle.backend.docker.cleanup import _list_orphan_state_dirs
@@ -119,37 +116,5 @@ class TestOrphanStateDirs(_FakeHomeMixin, unittest.TestCase):
)
class TestAuthoritativeDiscovery(unittest.TestCase):
def test_prepare_requires_authoritative_compose_projects(self):
with patch.object(cleanup.docker_mod, "require_docker"), \
patch.object(
cleanup, "list_compose_projects",
side_effect=EnumerationError("compose unavailable"),
) as projects, self.assertRaisesRegex(
EnumerationError, "compose unavailable",
):
cleanup.prepare_cleanup()
projects.assert_called_once_with(
warn_on_error=False,
raise_on_error=True,
)
def test_container_query_failure_raises(self):
failed = cleanup.subprocess.CompletedProcess(
[], 1, stdout="", stderr="daemon unavailable",
)
with patch.object(cleanup.subprocess, "run", return_value=failed), \
self.assertRaisesRegex(EnumerationError, "daemon unavailable"):
cleanup._list_prefixed_containers()
def test_network_query_failure_raises(self):
failed = cleanup.subprocess.CompletedProcess(
[], 1, stdout="", stderr="daemon unavailable",
)
with patch.object(cleanup.subprocess, "run", return_value=failed), \
self.assertRaisesRegex(EnumerationError, "daemon unavailable"):
cleanup._list_prefixed_networks()
if __name__ == "__main__":
unittest.main()
+15 -47
View File
@@ -32,19 +32,15 @@ class TestProcessScan(unittest.TestCase):
self.assertEqual(
Path("/cache/run/dev-a"),
fc_cleanup._run_dir_of(
("firecracker", "--config-file",
f"{run_root}/dev-a/config.json"), run_root
f"firecracker --config-file {run_root}/dev-a/config.json", run_root
),
)
# infra/builder VMs elsewhere, or nested paths, are not ours.
self.assertIsNone(
fc_cleanup._run_dir_of(
("firecracker", "--config-file", "/elsewhere/config.json"),
run_root,
)
fc_cleanup._run_dir_of("firecracker --config-file /elsewhere/config.json", run_root)
)
self.assertIsNone(
fc_cleanup._run_dir_of(("firecracker", "--no-config"), run_root)
fc_cleanup._run_dir_of("firecracker --no-config", run_root)
)
def test_scan_splits_live_dirs_from_orphan_pids(self):
@@ -52,40 +48,17 @@ class TestProcessScan(unittest.TestCase):
run_root = Path(tmp)
(run_root / "live-a").mkdir() # dir present -> live VM, protected
# "gone-b" dir intentionally absent -> lingering VMM, orphan pid
args = {
111: ("firecracker", "--config-file",
f"{run_root}/live-a/config.json"),
222: ("firecracker", "--config-file",
f"{run_root}/gone-b/config.json"),
333: ("firecracker", "--config-file", "/elsewhere/config.json"),
}
with patch.object(
fc_cleanup.subprocess, "run",
return_value=_proc("111\n222\n333\nnotanint\n"),
), patch.object(
fc_cleanup, "_process_args", side_effect=args.get,
):
out = (
f"111 firecracker --config-file {run_root}/live-a/config.json\n"
f"222 firecracker --config-file {run_root}/gone-b/config.json\n"
"333 firecracker --config-file /elsewhere/config.json\n"
"notanint firecracker --config-file x\n"
)
with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)):
live, orphan_pids = fc_cleanup._scan_processes(run_root)
self.assertEqual({str(run_root / "live-a")}, live)
self.assertEqual([222], orphan_pids)
def test_scan_preserves_spaces_in_config_path(self):
with tempfile.TemporaryDirectory(prefix="fc cache ") as tmp:
run_root = Path(tmp)
live = run_root / "live bottle"
live.mkdir()
with patch.object(
fc_cleanup.subprocess, "run", return_value=_proc("111\n"),
), patch.object(
fc_cleanup, "_process_args",
return_value=("firecracker", "--config-file",
str(live / "config.json")),
):
self.assertEqual(
({str(live)}, []),
fc_cleanup._scan_processes(run_root),
)
def test_scan_empty_when_pgrep_finds_no_processes(self):
with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(returncode=1)):
self.assertEqual((set(), []), fc_cleanup._scan_processes(Path("/x")))
@@ -144,14 +117,9 @@ class TestProcessScan(unittest.TestCase):
run_root = Path(tmp)
(run_root / "live-a").mkdir()
(run_root / "dead-b").mkdir()
out = f"111 firecracker --config-file {run_root}/live-a/config.json\n"
with patch.object(fc_cleanup, "_run_root", return_value=run_root), \
patch.object(fc_cleanup.subprocess, "run",
return_value=_proc("111\n")), \
patch.object(
fc_cleanup, "_process_args",
return_value=("firecracker", "--config-file",
str(run_root / "live-a/config.json")),
):
patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)):
plan = fc_cleanup.prepare_cleanup()
self.assertEqual((), plan.vm_pids)
self.assertEqual((str(run_root / "dead-b"),), plan.run_dirs)
@@ -167,11 +135,11 @@ class TestCleanupRemoval(unittest.TestCase):
with patch.object(fc_cleanup, "prepare_cleanup", return_value=plan), \
patch.object(fc_cleanup, "_run_root", return_value=Path("/run")), \
patch.object(fc_cleanup, "_terminate_orphan") as terminate, \
patch("bot_bottle.backend.cleanup_control.shutil.rmtree") as rmtree, \
patch.object(fc_cleanup.shutil, "rmtree") as rmtree, \
patch.object(fc_cleanup, "info"):
fc_cleanup.cleanup(plan)
terminate.assert_called_once_with(101, Path("/run"))
rmtree.assert_called_once_with(Path("/run/dev-x"))
rmtree.assert_called_once_with("/run/dev-x", ignore_errors=True)
def test_cleanup_skips_resources_no_longer_in_refreshed_plan(self):
preview = FirecrackerBottleCleanupPlan(
@@ -181,7 +149,7 @@ class TestCleanupRemoval(unittest.TestCase):
fc_cleanup, "prepare_cleanup",
return_value=FirecrackerBottleCleanupPlan(),
), patch.object(fc_cleanup, "_terminate_orphan") as terminate, \
patch("bot_bottle.backend.cleanup_control.shutil.rmtree") as rmtree:
patch.object(fc_cleanup.shutil, "rmtree") as rmtree:
fc_cleanup.cleanup(preview)
terminate.assert_not_called()
rmtree.assert_not_called()
-15
View File
@@ -56,21 +56,6 @@ class TestNetpoolProbes(unittest.TestCase):
with patch.object(netpool.subprocess, "run", side_effect=FileNotFoundError):
self.assertFalse(netpool._run_ok(["nft"]))
def test_run_ok_false_on_unexecutable_binary(self):
# A name on PATH that this user can't execute raises PermissionError,
# not FileNotFoundError — CPython reports that EACCES in preference to
# the ENOENT from the other PATH entries. Catching only the latter made
# `doctor` die with a traceback on a fresh macOS account.
with patch.object(netpool.subprocess, "run",
side_effect=PermissionError(13, "Permission denied", "ip")):
self.assertFalse(netpool._run_ok(["ip", "link", "show", "bbfc0"]))
def test_overlapping_routes_empty_when_ip_unusable(self):
for exc in (FileNotFoundError, PermissionError(13, "Permission denied", "ip")):
with self.subTest(exc=type(exc).__name__), \
patch.object(netpool.subprocess, "run", side_effect=exc):
self.assertEqual([], netpool.overlapping_routes())
def test_tap_and_nft_probes(self):
with patch.object(netpool, "_run_ok", return_value=True) as ok:
self.assertTrue(netpool.tap_present("bbfc0"))
+1 -101
View File
@@ -7,7 +7,6 @@ the smoke-test no-op — the logic that must hold without a VM.
from __future__ import annotations
import subprocess
import tempfile
import unittest
from pathlib import Path
@@ -92,106 +91,6 @@ class TestBuildAgentRootfsDir(unittest.TestCase):
self.assertNotEqual(first, second)
class TestBuildContext(unittest.TestCase):
"""The COPY-source parsing + context shipping that lets a Dockerfile pin an
input by COPYing a committed file (codex checksum list, claude/pi npm
lockfiles) through the otherwise-empty VM-side build context."""
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.root = Path(self._tmp.name)
self.addCleanup(self._tmp.cleanup)
codex = self.root / "bot_bottle" / "contrib" / "codex"
codex.mkdir(parents=True)
self.sums = codex / "codex-package_SHA256SUMS"
self.sums.write_text("aaaa codex-package-x86_64-unknown-linux-musl.tar.gz\n")
self.dockerfile = self.root / "Dockerfile"
self.dockerfile.write_text(
"FROM node:22-slim\n"
"COPY --chown=node:node "
"bot_bottle/contrib/codex/codex-package_SHA256SUMS /tmp/x\n"
)
def _patch_root(self):
return patch.object(
image_builder.resources, "build_root", return_value=self.root)
def test_copy_sources_parses_flags_and_continuations(self):
df = self.root / "Multi"
df.write_text(
"FROM x\n"
"COPY a/one.json \\\n a/two.json /dest/\n"
"COPY --from=builder /built /built\n" # excluded: build stage
"COPY --chown=n:n b/three /dest\n"
)
self.assertEqual(
image_builder._context_copy_sources(df),
["a/one.json", "a/two.json", "b/three"],
)
def test_context_files_resolves_existing_under_root(self):
with self._patch_root():
files = image_builder._context_files(self.dockerfile)
self.assertEqual(
[rel for rel, _ in files],
["bot_bottle/contrib/codex/codex-package_SHA256SUMS"],
)
def test_context_files_recurses_into_directory_sources(self):
nested = self.root / "assets" / "nested"
nested.mkdir(parents=True)
(self.root / "assets" / "one").write_text("one")
(nested / "two").write_text("two")
df = self.root / "Directory"
df.write_text("FROM x\nCOPY assets /opt/assets\n")
with self._patch_root():
files = image_builder._context_files(df)
self.assertEqual(
[rel for rel, _ in files],
["assets/nested/two", "assets/one"],
)
def test_context_files_drops_missing_and_traversal(self):
df = self.root / "Bad"
df.write_text("FROM x\nCOPY ../escape /d\nCOPY does/not/exist /d\n")
with self._patch_root():
self.assertEqual(image_builder._context_files(df), [])
def test_rootfs_digest_tracks_context_file_content(self):
with self._patch_root(), \
patch.object(image_builder.util, "cache_dir", return_value=self.root):
first = image_builder._rootfs_digest(self.dockerfile)
self.sums.write_text("bbbb codex-package-x86_64-unknown-linux-musl.tar.gz\n")
second = image_builder._rootfs_digest(self.dockerfile)
self.assertNotEqual(first, second)
def test_send_build_context_noop_without_copy(self):
df = self.root / "None"
df.write_text("FROM x\n")
with self._patch_root(), \
patch.object(image_builder.subprocess, "Popen") as popen:
image_builder._send_build_context(Path("/k"), "10.0.0.1", df, "/tmp/c")
popen.assert_not_called()
def test_send_build_context_streams_copied_files_to_vm(self):
completed = subprocess.CompletedProcess([], 0, stdout=b"", stderr=b"")
with self._patch_root(), \
patch.object(image_builder.subprocess, "Popen") as popen, \
patch.object(image_builder.subprocess, "run",
return_value=completed) as run:
popen.return_value.stdout = None
popen.return_value.returncode = 0
popen.return_value.wait.return_value = 0
image_builder._send_build_context(
Path("/k"), "10.0.0.1", self.dockerfile, "/tmp/c")
tar_argv = popen.call_args.args[0]
self.assertEqual(tar_argv[:3], ["tar", "-C", str(self.root)])
self.assertIn("--", tar_argv)
self.assertIn(
"bot_bottle/contrib/codex/codex-package_SHA256SUMS", tar_argv)
self.assertIn("tar -C /tmp/c/ctx -xf -", run.call_args.args[0][-1])
class TestSmokeTest(unittest.TestCase):
def test_buildah_receives_centralized_image_build_args(self):
with patch.object(
@@ -215,6 +114,7 @@ class TestSmokeTest(unittest.TestCase):
ssh.assert_not_called()
def test_failed_smoke_dies(self):
import subprocess
result = subprocess.CompletedProcess([], 1, stdout="broken", stderr="")
with patch.object(image_builder, "_ssh", return_value=result), \
self.assertRaises(SystemExit):
-14
View File
@@ -5,7 +5,6 @@ import subprocess
import tempfile
import threading
import unittest
import urllib.error
import urllib.request
from pathlib import Path
from unittest import mock
@@ -487,19 +486,6 @@ class TestMalformedStatusHeader(unittest.TestCase):
)
self.assertEqual(500, status)
def test_backend_timeout_returns_503(self):
with mock.patch(
"bot_bottle.gateway.git_gate.http_backend.subprocess.run",
side_effect=subprocess.TimeoutExpired(["git", "http-backend"], 1),
):
req = urllib.request.Request(
f"http://127.0.0.1:{self._port}/repo.git/info/refs",
method="GET",
)
with self.assertRaises(urllib.error.HTTPError) as raised:
urllib.request.urlopen(req, timeout=3)
self.assertEqual(503, raised.exception.code)
class TestContentLengthBounds(unittest.TestCase):
"""PRD 0041: malformed or oversized Content-Length is rejected before
-12
View File
@@ -284,18 +284,6 @@ class TestEnsureArtifact(_CacheMixin):
ia.ensure_artifact_gz(version, role=_ROLE)
self.assertEqual(first_calls, len(net.calls))
def test_download_uses_network_deadline(self) -> None:
response = mock.MagicMock()
response.__enter__.return_value = io.BytesIO(b"payload")
with tempfile.TemporaryDirectory() as d, mock.patch.object(
ia.urllib.request, "urlopen", return_value=response,
) as urlopen:
ia._download("https://registry/artifact", Path(d) / "artifact")
self.assertEqual(
ia.ARTIFACT_HTTP_TIMEOUT_SECONDS,
urlopen.call_args.kwargs["timeout"],
)
def test_checksum_mismatch_fails_closed(self) -> None:
version = "beefbeefbeefbeef"
gz = _gz(b"payload")
+41 -83
View File
@@ -9,7 +9,7 @@ create the config tree, install the package, and verify with `doctor`.
from __future__ import annotations
import os
import re
import sysconfig
import unittest
from pathlib import Path
@@ -17,22 +17,6 @@ REPO_ROOT = Path(__file__).resolve().parents[2]
INSTALL_SH = REPO_ROOT / "install.sh"
def code_only(text: str) -> str:
"""Script text with string literals and comments removed.
Both are places the script *talks about* commands rather than running
them remediation advice quite reasonably says "sudo apt install …"
so assertions about what the script actually executes must not see them.
Strings are stripped before comments because a '#' inside a quoted string
is not a comment, and several literals here span multiple lines.
"""
without_strings = re.sub(r"\"(?:[^\"\\]|\\.)*\"|'[^']*'", "", text)
return "\n".join(
ln for ln in without_strings.splitlines()
if ln.strip() and not ln.lstrip().startswith("#")
)
class TestInstallScript(unittest.TestCase):
@classmethod
def setUpClass(cls):
@@ -48,45 +32,20 @@ class TestInstallScript(unittest.TestCase):
self.assertIn("set -eu", self.text)
def test_never_uses_sudo(self):
# The installer must never *invoke* sudo. It may print it: the "no
# usable python" error suggests 'sudo apt install python3.12'.
self.assertNotIn("sudo", code_only(self.text))
# Only executable lines matter; the header comment may mention sudo.
code = [
ln for ln in self.text.splitlines()
if ln.strip() and not ln.lstrip().startswith("#")
]
self.assertNotIn("sudo", "\n".join(code))
def test_creates_config_tree(self):
self.assertIn(".bot-bottle/agents", self.text)
self.assertIn(".bot-bottle/bottles", self.text)
def test_installs_via_pipx_with_venv_fallback(self):
def test_installs_via_pipx_with_pip_fallback(self):
self.assertIn("pipx install", self.text)
self.assertIn("-m venv", self.text)
def test_no_pip_user_fallback(self):
# `pip install --user` is not a fallback, it's a dead end: PEP 668
# blocks it on Homebrew, python.org and Debian/Ubuntu interpreters,
# which is every Python a Mac realistically offers. A private venv is
# exempt from PEP 668 and needs no bootstrap, since venv is stdlib.
# code_only, because the comment explaining the absence says the words.
code = code_only(self.text)
self.assertNotIn("pip install --user", code)
self.assertNotIn("--break-system-packages", code)
def test_venv_lives_under_the_config_dir(self):
# Keeps the whole install footprint inside ~/.bot-bottle (plus the
# entry-point symlink), which is what makes deleting a throwaway
# account a complete reset in scripts/macos-install-test.sh.
self.assertIn(".bot-bottle/venv", self.text)
self.assertIn("BOT_BOTTLE_VENV", self.text)
def test_venv_failure_is_actionable(self):
# Debian/Ubuntu ship venv separately; failing there must say so rather
# than dumping ensurepip's error.
self.assertIn("python3-venv", self.text)
def test_entry_point_is_exposed_outside_the_venv(self):
# A venv's bin dir is never on PATH, so the console script has to be
# linked somewhere conventional or `bot-bottle` is unreachable.
self.assertIn(".local/bin", self.text)
self.assertIn("ln -sf", self.text)
self.assertIn("pip install --user", self.text)
def test_runs_doctor_after_install(self):
self.assertIn("doctor", self.text)
@@ -101,43 +60,42 @@ class TestInstallScript(unittest.TestCase):
self.assertIn("command -v git", self.text)
self.assertIn("git+*|*.git", self.text)
def test_installs_into_the_venv_with_its_own_pip(self):
# The venv's pip, not the base interpreter's — the base one may not
# exist, and using it would install outside the venv.
self.assertIn("${VENV}/bin/python\" -m pip install", self.text)
def test_checks_pip_usable_before_fallback(self):
self.assertIn("python3 -m pip --version", self.text)
def test_pipx_is_preferred_when_present(self):
# The venv is a fallback, not a takeover: someone who already manages
# their Python apps with pipx keeps doing so.
self.assertIn("command -v pipx", self.text)
def test_detects_externally_managed_python(self):
# PEP 668: 'pip install --user' is blocked on externally-managed
# interpreters; the script must detect this and point at pipx.
self.assertIn("EXTERNALLY-MANAGED", self.text)
self.assertIn("pipx", self.text)
def test_asks_pipx_where_its_bin_dir_is(self):
# PIPX_BIN_DIR is configurable, so the post-install "is it on PATH?"
# check must ask rather than assume ~/.local/bin.
self.assertIn("PIPX_BIN_DIR", self.text)
def test_resolves_user_scripts_dir_not_hardcoded(self):
# The pip --user scripts dir differs by platform; the script must ask
# the interpreter (sysconfig + the preferred *user* scheme) rather than
# hardcoding Linux's ~/.local/bin (which is wrong on macOS python.org).
self.assertIn("get_preferred_scheme", self.text)
self.assertIn("sysconfig", self.text)
# No hardcoded Linux path in executable lines (a comment may mention it).
code = "\n".join(
ln for ln in self.text.splitlines()
if ln.strip() and not ln.lstrip().startswith("#")
)
self.assertNotIn(".local/bin", code)
def test_searches_beyond_path_for_an_interpreter(self):
# `python3` on PATH is the *oldest* interpreter on a stock Mac: a fresh
# account's PATH is /etc/paths, so python3 is the 3.9.6 CLT stub while
# the usable build sits somewhere only a shell profile puts on PATH.
# Giving up at that point dead-ends every new macOS user.
for candidate in ("python3.11", "/opt/homebrew/bin", "Python.framework"):
self.assertIn(candidate, self.text)
def test_macos_user_scheme_is_not_dot_local_bin(self):
# The case the fix exists for: a python.org macOS interpreter uses the
# osx_framework_user scheme, whose scripts land under
# ~/Library/Python/<X.Y>/bin — NOT ~/.local/bin. Drive the same
# sysconfig lookup install.sh uses, with a mac-like userbase, to prove
# it resolves a non-~/.local/bin directory.
self.assertIn("osx_framework_user", sysconfig.get_scheme_names())
scripts = sysconfig.get_path(
"scripts", "osx_framework_user",
vars={"userbase": "/Users/dev/Library/Python/3.11"},
)
self.assertEqual("/Users/dev/Library/Python/3.11/bin", scripts)
self.assertNotIn("/.local/bin", scripts)
def test_interpreter_is_overridable(self):
self.assertIn("BOT_BOTTLE_PYTHON", self.text)
def test_pipx_is_pinned_to_the_vetted_interpreter(self):
# Without --python, pipx builds the venv with whichever interpreter
# pipx itself was installed with, which need not be the one that
# passed the version check.
self.assertIn("pipx install --python", self.text)
def test_version_failure_is_actionable(self):
# The failure a new macOS user actually hits must say what to do about
# it, not just state the requirement.
self.assertIn("brew install python@", self.text)
self.assertIn("BOT_BOTTLE_PYTHON=/path/to/python3", self.text)
if __name__ == "__main__":
unittest.main()
-98
View File
@@ -1,98 +0,0 @@
"""Unit: how the CLI tells users to re-run it under sudo.
`sudo bot-bottle ` is wrong for the users who followed the documented
install: sudo's secure_path excludes ~/.local/bin, where both pipx and
install.sh put the entry point. These lock in the absolute-path form.
"""
from __future__ import annotations
import contextlib
import io
import os
import tempfile
import unittest
from pathlib import Path
from unittest import mock
from bot_bottle import invocation
class TestSelfPath(unittest.TestCase):
def test_absolute_argv0_is_used_as_is(self):
with mock.patch.object(invocation.sys, "argv", ["/opt/venv/bin/bot-bottle"]):
self.assertEqual("/opt/venv/bin/bot-bottle", invocation.self_path())
def test_bare_name_is_resolved_through_path(self):
# The case that matters: invoked as `bot-bottle`, installed in a
# directory sudo would drop.
with tempfile.TemporaryDirectory() as d:
entry = Path(d, "bot-bottle")
entry.write_text("#!/bin/sh\n")
entry.chmod(0o755)
with mock.patch.object(invocation.sys, "argv", ["bot-bottle"]), \
mock.patch.dict(os.environ, {"PATH": d}):
self.assertEqual(str(entry), invocation.self_path())
def test_relative_path_is_made_absolute(self):
with tempfile.TemporaryDirectory() as d:
entry = Path(d, "bot-bottle")
entry.write_text("#!/bin/sh\n")
entry.chmod(0o755)
cwd = os.getcwd()
try:
os.chdir(d)
with mock.patch.object(invocation.sys, "argv", ["./bot-bottle"]):
self.assertTrue(os.path.isabs(invocation.self_path()))
finally:
os.chdir(cwd)
def test_unresolvable_entry_point_falls_back_to_the_name(self):
# `python -m`-style invocation, or an argv[0] that no longer exists.
# A slightly wrong hint beats a traceback raised while reporting some
# unrelated problem.
with mock.patch.object(invocation.sys, "argv", ["/nonexistent/gone"]), \
mock.patch.object(invocation.shutil, "which", return_value=None):
self.assertEqual("/nonexistent/gone", invocation.self_path())
with mock.patch.object(invocation.sys, "argv", [""]), \
mock.patch.object(invocation.shutil, "which", return_value=None):
self.assertEqual("bot-bottle", invocation.self_path())
class TestSudoCommand(unittest.TestCase):
def test_names_an_absolute_path_not_the_bare_command(self):
with mock.patch.object(invocation, "self_path",
return_value="/home/u/.local/bin/bot-bottle"):
cmd = invocation.sudo_command("backend", "setup", "--backend=firecracker")
self.assertEqual(
"sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker",
cmd,
)
# The regression this exists to prevent.
self.assertNotIn("sudo bot-bottle", cmd)
class TestFirecrackerSetupUsesIt(unittest.TestCase):
def test_root_reinvocation_hint_names_an_absolute_path(self):
# The message that prompted all this. Drive the real code path rather
# than scanning the source, which would also match the comment
# explaining why the bare form is wrong.
from bot_bottle.backend.firecracker import setup as fc_setup
err, out = io.StringIO(), io.StringIO()
with mock.patch.object(fc_setup.os, "geteuid", return_value=501), \
mock.patch.object(fc_setup.invocation, "self_path",
return_value="/home/u/.local/bin/bot-bottle"), \
contextlib.redirect_stderr(err), contextlib.redirect_stdout(out):
fc_setup._setup_systemd()
printed = err.getvalue()
self.assertIn(
"sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker",
printed,
)
self.assertNotIn("sudo bot-bottle", printed)
if __name__ == "__main__":
unittest.main()
+1 -17
View File
@@ -6,7 +6,6 @@ import unittest
from unittest.mock import patch
from bot_bottle.backend import EnumerationError
from bot_bottle.backend.cleanup_control import CleanupError
from bot_bottle.backend.macos_container import cleanup, enumerate as enum_mod
from bot_bottle.backend.macos_container.bottle_cleanup_plan import (
MacosContainerBottleCleanupPlan,
@@ -32,10 +31,7 @@ class TestMacosContainerCleanup(unittest.TestCase):
containers=("bot-bottle-a",),
networks=("bot-bottle-net-a",),
)
completed = cleanup.subprocess.CompletedProcess(
args=[], returncode=0, stdout="", stderr="",
)
with patch.object(cleanup.subprocess, "run", return_value=completed) as run:
with patch.object(cleanup.subprocess, "run") as run:
cleanup.cleanup(plan)
self.assertEqual(
["container", "delete", "--force", "bot-bottle-a"],
@@ -46,18 +42,6 @@ class TestMacosContainerCleanup(unittest.TestCase):
run.call_args_list[1].args[0],
)
def test_cleanup_attempts_all_resources_then_raises(self):
plan = MacosContainerBottleCleanupPlan(
containers=("bot-bottle-a",), networks=("bot-bottle-net-a",),
)
failed = cleanup.subprocess.CompletedProcess(
args=[], returncode=1, stdout="", stderr="unavailable",
)
with patch.object(cleanup.subprocess, "run", return_value=failed) as run, \
self.assertRaisesRegex(CleanupError, "unavailable"):
cleanup.cleanup(plan)
self.assertEqual(2, run.call_count)
def test_container_enumeration_failure_aborts(self):
completed = cleanup.subprocess.CompletedProcess(
args=[], returncode=1, stdout="", stderr="service unavailable",
+1 -29
View File
@@ -158,11 +158,8 @@ resolver #2
f"registry:5000/agent:sha256-{'a' * 64}",
pinned,
)
# The tag source is the name:tag ref, not the image ID: Apple
# Container's `image tag` rejects a bare 64-byte hex string as a
# reference. The second image_id call is what keeps this fail-closed.
run.assert_called_once_with(
["container", "image", "tag", "registry:5000/agent:latest", pinned],
["container", "image", "tag", image, pinned],
capture_output=True,
text=True,
check=False,
@@ -172,31 +169,6 @@ resolver #2
[call.args for call in inspect.call_args_list],
)
def test_pinned_local_image_ref_accepts_bare_hex_image_id(self):
# `container image inspect` reports "id" without the sha256: prefix.
image = "b" * 64
completed = util.subprocess.CompletedProcess(
args=[], returncode=0, stdout="", stderr="",
)
with patch.object(util, "image_id", return_value=image), \
patch.object(util.subprocess, "run", return_value=completed) as run:
pinned = util.pinned_local_image_ref("agent:latest")
self.assertEqual(f"agent:sha256-{'b' * 64}", pinned)
self.assertEqual(
["container", "image", "tag", "agent:latest", pinned],
run.call_args.args[0],
)
def test_pinned_local_image_ref_dies_when_tag_moved_mid_flight(self):
completed = util.subprocess.CompletedProcess(
args=[], returncode=0, stdout="", stderr="",
)
with patch.object(
util, "image_id", side_effect=["c" * 64, "d" * 64],
), patch.object(util.subprocess, "run", return_value=completed), \
self.assertRaises(SystemExit):
util.pinned_local_image_ref("agent:latest")
def test_commit_container_execs_tar_and_builds_image(self):
# stderr is bytes because subprocess.run uses stderr=PIPE without text=True
completed = util.subprocess.CompletedProcess(
@@ -224,17 +224,6 @@ class TestAgentSecrets(unittest.TestCase):
reopened = RegistryStore(self.db)
self.assertEqual({"K": "v"}, reopened.get_agent_secrets("bottle-1"))
def test_v6_migration_clears_legacy_secret_rows(self) -> None:
self.store.store_agent_secrets("bottle-1", {"K": "legacy"})
with closing(sqlite3.connect(self.db)) as conn:
conn.execute(
"UPDATE schema_versions SET version = 5 "
"WHERE module = 'orchestrator_registry'"
)
conn.commit()
self.store.migrate()
self.assertEqual({}, self.store.get_agent_secrets("bottle-1"))
class TestReapAbsent(unittest.TestCase):
"""`reap_absent` — the self-heal for rows whose bottle is gone.
+12 -15
View File
@@ -3,6 +3,8 @@
from __future__ import annotations
import base64
import hashlib
import hmac
import unittest
from bot_bottle.orchestrator.store.secret_store import (
@@ -81,21 +83,16 @@ class TestDecryptErrors(unittest.TestCase):
with self.assertRaisesRegex(ValueError, "authentication failed"):
decrypt_value(self.secret, tampered)
def test_rejects_legacy_ciphertext(self) -> None:
legacy = base64.urlsafe_b64encode(
b"0123456789abcdeflegacy-token",
).rstrip(b"=").decode()
with self.assertRaisesRegex(ValueError, "unsupported ciphertext format"):
decrypt_value(self.secret, legacy)
def test_rejects_authenticated_blob_with_changed_version(self) -> None:
raw = bytearray(base64.urlsafe_b64decode(
encrypt_value(self.secret, "secret-token") + "=="
))
raw[0] ^= 1
downgraded = base64.urlsafe_b64encode(raw).rstrip(b"=").decode()
with self.assertRaisesRegex(ValueError, "unsupported ciphertext format"):
decrypt_value(self.secret, downgraded)
def test_reads_legacy_ciphertext_for_migration(self) -> None:
key = base64.urlsafe_b64decode(self.secret + "==")
nonce = b"0123456789abcdef"
plaintext = b"legacy-token"
stream = hmac.new(
key, nonce + (0).to_bytes(4, "big"), hashlib.sha256,
).digest()
ciphertext = bytes(p ^ k for p, k in zip(plaintext, stream))
legacy = base64.urlsafe_b64encode(nonce + ciphertext).rstrip(b"=").decode()
self.assertEqual("legacy-token", decrypt_value(self.secret, legacy))
def test_truncated_blob_raises_value_error(self) -> None:
with self.assertRaises(ValueError):
+9
View File
@@ -70,6 +70,15 @@ def dispatch(
response = asyncio.run(request())
payload = response.json()
if response.status_code == 422:
detail = payload.get("detail", []) if isinstance(payload, dict) else []
field = ""
if isinstance(detail, list) and detail and isinstance(detail[0], dict):
location = detail[0].get("loc", ())
if isinstance(location, (list, tuple)) and len(location) > 1:
field = str(location[1])
suffix = f": {field}" if field else ""
return 400, {"error": f"invalid request body{suffix}"}
if isinstance(payload, dict) and "detail" in payload and "error" not in payload:
payload = {"error": payload["detail"]}
return response.status_code, payload
+3 -35
View File
@@ -55,11 +55,7 @@ class TestProvisionGitGate(unittest.TestCase):
def test_copies_creds_and_runs_namespaced_init(self) -> None:
calls: list[list[str]] = []
with patch(_RUN, side_effect=_recorder(calls)):
provision_git_gate(
DockerGatewayTransport("gw"),
"bottle1",
_plan(_up("foo", known_hosts="/host/kh")),
)
provision_git_gate(DockerGatewayTransport("gw"), "bottle1", _plan(_up("foo", known_hosts="/host/kh")))
cps = [c for c in calls if c[:2] == ["docker", "cp"]]
self.assertIn(["docker", "cp", "/host/keys/id", "gw:/git-gate/creds/bottle1/foo-key"], cps)
@@ -85,39 +81,11 @@ class TestProvisionGitGate(unittest.TestCase):
["docker", "exec", "gw", "chmod", "+x", "/etc/git-gate/access-hook"], calls,
)
def test_applies_private_modes_inside_gateway(self) -> None:
calls: list[list[str]] = []
with patch(_RUN, side_effect=_recorder(calls)):
provision_git_gate(
DockerGatewayTransport("gw"),
"bottle1",
_plan(_up("foo", known_hosts="/host/kh")),
)
self.assertIn(
[
"docker", "exec", "gw", "chmod", "700",
"/git-gate/creds/bottle1",
],
calls,
)
self.assertIn(
[
"docker", "exec", "gw", "chmod", "600",
"/git-gate/creds/bottle1/foo-key",
"/git-gate/creds/bottle1/foo-known_hosts",
],
calls,
)
def test_omits_known_hosts_copy_when_absent(self) -> None:
calls: list[list[str]] = []
with patch(_RUN, side_effect=_recorder(calls)):
# No known-hosts file: only the identity key is copied.
provision_git_gate(DockerGatewayTransport("gw"), "b1", _plan(_up("foo")))
creds_cps = [
c for c in calls
if c[:2] == ["docker", "cp"] and "/git-gate/creds/" in c[3]
]
provision_git_gate(DockerGatewayTransport("gw"), "b1", _plan(_up("foo"))) # no known_hosts
creds_cps = [c for c in calls if c[:2] == ["docker", "cp"] and "/git-gate/creds/" in c[3]]
self.assertEqual(1, len(creds_cps)) # only the key, not known_hosts
self.assertTrue(creds_cps[0][3].endswith("/foo-key"))
-10
View File
@@ -49,16 +49,6 @@ class TestPut(unittest.TestCase):
self.assertNotIsInstance(req.data, (bytes, bytearray))
self.assertEqual(str(len(payload)), req.get_header("Content-length"))
def test_put_uses_network_deadline(self) -> None:
with mock.patch.object(
pub.urllib.request, "urlopen", return_value=_Resp(),
) as urlopen:
pub._put("https://reg/pkg", b"payload", token="")
self.assertEqual(
pub._REGISTRY_HTTP_TIMEOUT_SECONDS,
urlopen.call_args.kwargs["timeout"],
)
def test_small_bytes_body_still_works(self) -> None:
captured: list[urllib.request.Request] = []
+6 -8
View File
@@ -136,13 +136,12 @@ class TestStoreGuardBranches(unittest.TestCase):
db.unlink()
self.assertEqual([], store.list_all_pending_proposals())
def test_queue_store_chmod_oserror_fails_closed(self):
def test_queue_store_chmod_oserror_is_swallowed(self):
with tempfile.TemporaryDirectory() as d:
db = Path(d) / "q.db"
store = QueueStore("key", db_path=db)
with patch("os.fchmod", side_effect=OSError("ro")):
with self.assertRaisesRegex(OSError, "ro"):
store.migrate()
with patch("pathlib.Path.chmod", side_effect=OSError("ro")):
store.migrate() # must not raise
def test_audit_store_missing_db_read_returns_empty(self):
with tempfile.TemporaryDirectory() as d:
@@ -152,13 +151,12 @@ class TestStoreGuardBranches(unittest.TestCase):
db.unlink()
self.assertEqual([], store.read_audit_entries("egress", "slug"))
def test_audit_store_chmod_oserror_fails_closed(self):
def test_audit_store_chmod_oserror_is_swallowed(self):
with tempfile.TemporaryDirectory() as d:
db = Path(d) / "a.db"
store = AuditStore(db_path=db)
with patch("os.fchmod", side_effect=OSError("ro")):
with self.assertRaisesRegex(OSError, "ro"):
store.migrate()
with patch("pathlib.Path.chmod", side_effect=OSError("ro")):
store.migrate() # must not raise
if __name__ == "__main__":