Compare commits
57 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c527841d55 | |||
| 5940b75bb7 | |||
| 5e01c28016 | |||
| 2f8539c2c7 | |||
| ad100b8a84 | |||
| c7375051fd | |||
| d9e685e860 | |||
| b4b73a8acc | |||
| b1ebc6f1b8 | |||
| 8b5b5730ae | |||
| 44479f328e | |||
| 2de223a33b | |||
| af1690ab22 | |||
| 09debcf4f0 | |||
| fa11ad9a4a | |||
| ad6471af12 | |||
| 137df6f853 | |||
| 4252ca3562 | |||
| 701f5bf5e3 | |||
| d589c08d9d | |||
| 559dc03bb5 | |||
| 9172bf3a42 | |||
| 0adbf25977 | |||
| d1aec706e3 | |||
| a589604aa0 | |||
| 4c01e31e96 | |||
| 6f885af4b4 | |||
| 127ba49372 | |||
| 0d696674e3 | |||
| 626f07efa6 | |||
| d117460192 | |||
| e72ec71047 | |||
| 7aff69fbe0 | |||
| 1d91db3e31 | |||
| 686ca0d74b | |||
| 6d44a1be0a | |||
| 32e85de16f | |||
| a1d2c4a500 | |||
| a6fe31a424 | |||
| 41b2b24b36 | |||
| 37045ca147 | |||
| 9b54cfa854 | |||
| c193b04338 | |||
| c7ab3e0957 | |||
| 034f774529 | |||
| 5b359fe8d2 | |||
| 015ff52eda | |||
| 4302678f3e | |||
| 3a6fbad057 | |||
| a800a417d9 | |||
| 293218035d | |||
| 727eafe0f9 | |||
| 1ec114b6d7 | |||
| aa44feea02 | |||
| f2e2572a40 | |||
| 7069fa225d | |||
| aa224c4381 |
@@ -1,6 +1,10 @@
|
||||
[run]
|
||||
branch = True
|
||||
source = .
|
||||
# Store paths relative to the project root so .coverage.* files produced on
|
||||
# different runners (ubuntu-latest vs self-hosted KVM) can be combined by the
|
||||
# coverage job without a [paths] remapping section.
|
||||
relative_files = True
|
||||
|
||||
[report]
|
||||
# Coverage policy: see docs/decisions/0004-coverage-policy.md.
|
||||
|
||||
@@ -24,8 +24,19 @@ jobs:
|
||||
|
||||
- name: Run pylint
|
||||
run: |
|
||||
# Run pylint on all Python files in the repo
|
||||
find . -name '*.py' -not -path './.venv/*' -not -path './.git/*' | xargs pylint --fail-under=8.0
|
||||
# Pylint's normal exit code is nonzero for any emitted finding,
|
||||
# regardless of --fail-under. Preserve the full report but enforce
|
||||
# the aggregate score this workflow promises.
|
||||
set +e
|
||||
find . -name '*.py' -not -path './.venv/*' -not -path './.git/*' \
|
||||
| xargs pylint --fail-under=8.0 \
|
||||
| tee /tmp/pylint-output.txt
|
||||
set -e
|
||||
SCORE=$(sed -n \
|
||||
's/^Your code has been rated at \([-0-9.]*\)\/10.*/\1/p' \
|
||||
/tmp/pylint-output.txt | tail -1)
|
||||
test -n "$SCORE"
|
||||
awk -v score="$SCORE" 'BEGIN { exit !(score >= 8.0) }'
|
||||
|
||||
- name: Run pyright
|
||||
run: |
|
||||
|
||||
+193
-30
@@ -4,16 +4,17 @@
|
||||
# dependencies are required to execute it. Tests are split by directory:
|
||||
#
|
||||
# tests/unit/ — pure unit tests; always run
|
||||
# tests/integration/ — need a reachable Docker daemon; skip cleanly
|
||||
# (via tests/_docker.py:skip_unless_docker) when
|
||||
# Docker isn't available on the runner
|
||||
# tests/integration/ — need a reachable backend; skip cleanly when
|
||||
# the backend isn't available on the runner
|
||||
# tests/canaries/ — upstream regression canaries; run on a separate
|
||||
# schedule (see canaries.yml), not here
|
||||
#
|
||||
# This workflow assumes the Gitea Actions runner exposes the host Docker
|
||||
# socket to the job container so `docker` commands inside the job can
|
||||
# reach the daemon. If that's not yet configured on the runner the
|
||||
# integration tests will skip rather than fail.
|
||||
# Each test job runs once under coverage and uploads a small .coverage.*
|
||||
# artifact. The `coverage` job combines them — no test reruns, no KVM
|
||||
# dependency on that job. For main-branch pushes only, the tested rootfs
|
||||
# and matching dropbear are uploaded so `publish-infra` can publish the
|
||||
# byte-identical artifact that was tested. PRs avoid the ~194 MB rootfs
|
||||
# transfer entirely.
|
||||
|
||||
name: test
|
||||
|
||||
@@ -23,9 +24,22 @@ on:
|
||||
- main
|
||||
paths:
|
||||
- '**.py'
|
||||
- '.gitea/workflows/**.yml'
|
||||
- 'scripts/**'
|
||||
- 'README.md'
|
||||
# Dockerfiles and pyproject.toml are baked into the infra rootfs; a
|
||||
# change here alters what the integration/coverage jobs build locally.
|
||||
- 'Dockerfile*'
|
||||
- 'pyproject.toml'
|
||||
pull_request:
|
||||
paths:
|
||||
- '**.py'
|
||||
- '.gitea/workflows/**.yml'
|
||||
- 'scripts/**'
|
||||
- 'README.md'
|
||||
- 'Dockerfile*'
|
||||
- 'pyproject.toml'
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
unit:
|
||||
@@ -42,13 +56,19 @@ jobs:
|
||||
- name: Install dev requirements
|
||||
run: python3 -m pip install --break-system-packages -r requirements-dev.txt
|
||||
|
||||
- name: Run unit tests
|
||||
run: python3 -m coverage run -m unittest discover -t . -s tests/unit -v
|
||||
- name: Run unit tests with coverage
|
||||
run: python3 -m coverage run --data-file=.coverage.unit -m unittest discover -t . -s tests/unit -v
|
||||
|
||||
- name: Report unit coverage
|
||||
run: python3 -m coverage report -m
|
||||
run: python3 -m coverage report --data-file=.coverage.unit -m
|
||||
|
||||
integration:
|
||||
- name: Upload unit coverage artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: coverage-unit
|
||||
path: ${{ github.workspace }}/.coverage.unit
|
||||
|
||||
integration-docker:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
@@ -56,6 +76,9 @@ jobs:
|
||||
|
||||
# No actions/setup-python (see the note in the `unit` job); the
|
||||
# container's system Python 3.12 runs the stdlib test suite directly.
|
||||
- name: Install coverage
|
||||
run: python3 -m pip install --break-system-packages coverage
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
python3 --version
|
||||
@@ -65,33 +88,173 @@ jobs:
|
||||
echo "docker not on PATH — integration tests will skip"
|
||||
fi
|
||||
|
||||
- name: Run integration tests
|
||||
run: python3 -m unittest discover -t . -s tests/integration -v
|
||||
- name: Run integration tests (docker) with coverage
|
||||
env:
|
||||
BOT_BOTTLE_BACKEND: docker
|
||||
run: python3 -m coverage run --data-file=.coverage.docker -m unittest discover -t . -s tests/integration -v
|
||||
|
||||
# Combined unit+integration coverage report (informational). See
|
||||
# docs/decisions/0004-coverage-policy.md.
|
||||
- name: Upload docker coverage artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: coverage-docker
|
||||
path: ${{ github.workspace }}/.coverage.docker
|
||||
|
||||
# Integration tests against the Firecracker backend. Runs on a self-hosted
|
||||
# KVM runner (label `kvm`) where /dev/kvm and the TAP/nft pool are available.
|
||||
#
|
||||
# The hard diff-coverage gate (changed lines >= 90%) is DEFERRED: the
|
||||
# Firecracker backend's VM/SSH orchestration is covered by the integration
|
||||
# suite, which needs /dev/kvm + the provisioned TAP/nft pool — a
|
||||
# container-based runner skips it and those lines read uncovered, so the
|
||||
# gate can't pass here. Re-enabling it on a self-hosted KVM runner is
|
||||
# tracked separately (see PRD 0069 / #348 and the ci-runner branch).
|
||||
# Restricted to same-repo PRs, push to main, and workflow_dispatch — fork
|
||||
# PRs don't execute untrusted code on the privileged runner.
|
||||
#
|
||||
# Runner prerequisites (provision once; see README "Firecracker on Linux"):
|
||||
# `firecracker` on PATH, `/dev/kvm` accessible, cached kernel +
|
||||
# static dropbear at /var/cache/bot-bottle-fc/dropbear, and the pool as a
|
||||
# persistent systemd unit.
|
||||
#
|
||||
# The infra candidate is built here directly (no artifact download) to
|
||||
# eliminate the ~70 s ubuntu-latest upload + ~83 s combined download that
|
||||
# the old build-infra → integration-firecracker + coverage chain incurred.
|
||||
# For main-branch pushes the tested rootfs and matching dropbear are
|
||||
# uploaded so publish-infra can publish the byte-identical artifact; PRs
|
||||
# skip those uploads entirely.
|
||||
integration-firecracker:
|
||||
runs-on: [self-hosted, kvm]
|
||||
if: >-
|
||||
github.event_name == 'push' ||
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'pull_request' &&
|
||||
github.event.pull_request.head.repo.full_name == github.repository)
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Preflight — Firecracker host is ready
|
||||
run: |
|
||||
command -v firecracker >/dev/null || {
|
||||
echo "firecracker not on PATH — provision the runner (README: Firecracker on Linux)"; exit 1; }
|
||||
test -e /dev/kvm || { echo "/dev/kvm missing — KVM not available on this runner"; exit 1; }
|
||||
# `backend status` exits non-zero unless the TAP pool is up + no
|
||||
# range overlap; it prints the exact `backend setup` fix.
|
||||
python3 cli.py backend status --backend=firecracker
|
||||
|
||||
- name: Build infra candidate from this checkout
|
||||
env:
|
||||
BOT_BOTTLE_FC_DROPBEAR: /var/cache/bot-bottle-fc/dropbear
|
||||
run: python3 -m bot_bottle.backend.firecracker.publish_infra --output infra-candidate
|
||||
|
||||
- name: Replace the persistent infra VM with the candidate
|
||||
run: python3 -c 'from bot_bottle.backend.firecracker import infra_vm; infra_vm.stop()'
|
||||
|
||||
# No dev-requirements install: `coverage` is already provided by the
|
||||
# self-hosted runner's Nix python env, and that env has no `pip`
|
||||
# module to install into anyway.
|
||||
- name: Run integration tests (firecracker) with coverage
|
||||
env:
|
||||
BOT_BOTTLE_BACKEND: firecracker
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_DIR: ${{ github.workspace }}/infra-candidate
|
||||
run: python3 -m coverage run --data-file=.coverage.firecracker -m unittest discover -t . -s tests/integration -v
|
||||
|
||||
- name: Upload firecracker coverage artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: coverage-firecracker
|
||||
path: ${{ github.workspace }}/.coverage.firecracker
|
||||
|
||||
# Only upload the large rootfs artifact on main-branch pushes;
|
||||
# PRs avoid the ~194 MB transfer. publish-infra only runs on main
|
||||
# and downloads these to publish the byte-identical tested rootfs.
|
||||
- name: Upload tested rootfs (main branch only)
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate/
|
||||
|
||||
- name: Upload dropbear for publish verification (main branch only)
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: /var/cache/bot-bottle-fc/dropbear
|
||||
|
||||
# Combined coverage gate: aggregates .coverage.* artifacts uploaded by each
|
||||
# test job, then runs the diff-coverage gate (new/changed lines >= 90%).
|
||||
#
|
||||
# Runs on ubuntu-latest — no KVM needed, no test reruns. Coverage files use
|
||||
# relative_files = True (.coveragerc) so they combine cleanly across runners.
|
||||
#
|
||||
# Restricted to the same events as integration-firecracker: it depends on
|
||||
# that job's coverage artifact and skips for fork PRs alongside it.
|
||||
coverage:
|
||||
needs: [unit, integration-docker, integration-firecracker]
|
||||
timeout-minutes: 15
|
||||
runs-on: ubuntu-latest
|
||||
if: >-
|
||||
github.event_name == 'push' ||
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'pull_request' &&
|
||||
github.event.pull_request.head.repo.full_name == github.repository)
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
# No actions/setup-python: the runner image already ships Python 3.12,
|
||||
# and older act_runner engines mishandle setup-python's PATH (coverage
|
||||
# lands in one interpreter, `python3` resolves to another). Install
|
||||
# straight into the ephemeral job container's system Python —
|
||||
# --break-system-packages is safe because the container is disposable.
|
||||
- name: Install dev requirements
|
||||
run: python3 -m pip install --break-system-packages -r requirements-dev.txt
|
||||
- name: Install coverage
|
||||
run: python3 -m pip install --break-system-packages coverage
|
||||
|
||||
- name: Combined coverage report (unit + integration)
|
||||
run: PYTHON=python3 bash scripts/coverage.sh critical
|
||||
- name: Download unit coverage artifact
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: coverage-unit
|
||||
path: ${{ github.workspace }}
|
||||
|
||||
- name: Download docker coverage artifact
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: coverage-docker
|
||||
path: ${{ github.workspace }}
|
||||
|
||||
- name: Download firecracker coverage artifact
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: coverage-firecracker
|
||||
path: ${{ github.workspace }}
|
||||
|
||||
- name: Combined coverage (unit + integration, incl. firecracker)
|
||||
run: PYTHON=python3 bash scripts/coverage.sh aggregate critical
|
||||
|
||||
- name: Diff-coverage gate (changed lines >= 90%)
|
||||
run: |
|
||||
git fetch --no-tags origin main:refs/remotes/origin/main
|
||||
python3 scripts/diff_coverage.py --base origin/main --min 90
|
||||
|
||||
publish-infra:
|
||||
needs: [unit, integration-docker, integration-firecracker, coverage]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
steps:
|
||||
- name: Checkout the tested revision
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Download the tested rootfs
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate
|
||||
|
||||
# publish_infra re-derives the version from the checkout to confirm the
|
||||
# bundle matches before uploading, and the version hashes the dropbear
|
||||
# bytes. Download the SAME dropbear integration-firecracker used, or
|
||||
# the recheck computes a "<missing>"-dropbear version and rejects the
|
||||
# candidate.
|
||||
- name: Download the staged dropbear (matches build's version)
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: firecracker-inputs
|
||||
|
||||
- name: Publish the tested candidate
|
||||
env:
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_TOKEN: ${{ secrets.BOT_BOTTLE_INFRA_ARTIFACT_TOKEN }}
|
||||
BOT_BOTTLE_FC_DROPBEAR: ${{ github.workspace }}/firecracker-inputs/dropbear
|
||||
run: python3 -m bot_bottle.backend.firecracker.publish_infra --publish-dir infra-candidate
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
name: tracker-policy-issues
|
||||
|
||||
on:
|
||||
issues:
|
||||
types: [opened, unlabeled]
|
||||
|
||||
jobs:
|
||||
label-issue:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
issues: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Ensure the issue has a label
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: python3 scripts/tracker_policy.py label-issue
|
||||
@@ -0,0 +1,18 @@
|
||||
name: tracker-policy-pr
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, edited, reopened, synchronize, labeled, unlabeled]
|
||||
|
||||
jobs:
|
||||
check-pr:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
issues: read
|
||||
pull-requests: read
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Require an unlabeled PR linked to an issue
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: python3 scripts/tracker_policy.py check-pr
|
||||
+3
-2
@@ -98,6 +98,9 @@ RUN pip install --no-cache-dir /src/
|
||||
|
||||
# mitmdump -s requires a file path, not a module. Write a one-line shim that
|
||||
# re-exports `addons` from the installed package; mitmdump finds it there.
|
||||
# WORKDIR here also creates /app so the shim + COPYs below can write into it
|
||||
# (nothing created /app before this point).
|
||||
WORKDIR /app
|
||||
RUN printf 'from bot_bottle.egress_addon import addons\n' > /app/egress_addon.py
|
||||
COPY bot_bottle/egress_entrypoint.sh /app/egress-entrypoint.sh
|
||||
RUN chmod +x /app/egress-entrypoint.sh
|
||||
@@ -117,8 +120,6 @@ RUN mkdir -p \
|
||||
# subset the bottle uses.
|
||||
EXPOSE 8888 9099 9418 9420 9100
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# PID 1 is the supervisor. It owns signal handling and exit-code
|
||||
# propagation; no `exec` chain in the entrypoint itself.
|
||||
ENTRYPOINT ["python3", "-m", "bot_bottle.gateway_init"]
|
||||
|
||||
@@ -90,6 +90,8 @@ BOT_BOTTLE_BACKEND=firecracker ./cli.py start <agent>
|
||||
|
||||
> **NixOS:** enable `virtualisation.docker`, ensure the KVM module is loaded (`boot.kernelModules = [ "kvm-intel" ];` or `kvm-amd`), and add your user to the `kvm` and `docker` groups. For the network pool, consume the flake module — `imports = [ inputs.bot-bottle.nixosModules.firecracker-netpool ]; services.bot-bottle-firecracker = { enable = true; owner = "you"; };` — then `nixos-rebuild switch` (imperative nft/TAP rules don't survive a rebuild; channel users can `imports = [ <bot-bottle>/nix/firecracker-netpool.nix ]`). `firecracker` isn't in nixpkgs by default as a user binary — install the release binary (pin the version) and put it on `PATH`.
|
||||
|
||||
> **CI:** the coverage gate (`.gitea/workflows/test.yml` → `coverage` job) runs on a self-hosted runner labelled `kvm`, because the Firecracker backend's VM/SSH orchestration is exercised only by the integration suite, which needs `/dev/kvm` + the provisioned pool (a container runner would skip it and read as uncovered). Provision that runner exactly like a normal Firecracker host — `firecracker` on `PATH`, `/dev/kvm`, the cached guest kernel + static dropbear, and the pool installed as the persistent systemd unit — then register it with the `kvm` label. A Docker-capable hosted job builds the candidate once; KVM tests boot those exact bytes, and a successful main run publishes them. The unit/lint jobs still run on `ubuntu-latest`.
|
||||
|
||||
```sh
|
||||
./cli.py start <agent> # builds the image on first run, drops you into claude
|
||||
```
|
||||
@@ -171,6 +173,15 @@ When an outbound DLP detector matches a token, the route's `dlp.outbound_on_matc
|
||||
|
||||
More examples in `examples/`. Full design lives under `docs/prds/`; the trust-boundary rationale is in `docs/prds/0011-per-file-md-manifest.md`.
|
||||
|
||||
## Tracker policy
|
||||
|
||||
Issues are the canonical work items and own all tracker labels; every issue
|
||||
must have at least one. Pull requests stay unlabeled and deliberately reference
|
||||
an issue with `Closes #…`, `Part of #…`, or another form defined in
|
||||
[`ADR 0005`](docs/decisions/0005-issues-own-tracker-metadata.md). Gitea Actions
|
||||
enforces the convention for new work from 2026-07-18 onward. Earlier closed
|
||||
PRs are grandfathered rather than given artificial retrospective issues.
|
||||
|
||||
## Trademarks
|
||||
|
||||
bot-bottle is an independent project and is not affiliated with, endorsed by, or sponsored by Anthropic, PBC. "Claude" and "Claude Code" are trademarks of Anthropic, PBC; the project name uses "claude" descriptively to indicate that the tool runs Claude Code inside a sandbox.
|
||||
|
||||
@@ -45,10 +45,6 @@ PROVIDER_TEMPLATES = frozenset({PROVIDER_CLAUDE, PROVIDER_CODEX, PROVIDER_PI})
|
||||
# forward_host_credentials is enabled. Pipelock must pass these through
|
||||
# (no TLS MITM) or its header DLP blocks the injected JWT.
|
||||
CODEX_HOST_CREDENTIAL_HOSTS = ("api.openai.com", "chatgpt.com")
|
||||
|
||||
# Host that egress injects the host Claude bearer on when Claude
|
||||
# forward_host_credentials is enabled.
|
||||
CLAUDE_HOST_CREDENTIAL_HOSTS = ("api.anthropic.com",)
|
||||
PromptMode = Literal[
|
||||
"append_file",
|
||||
"read_prompt_file",
|
||||
|
||||
@@ -42,7 +42,7 @@ class AuditStore(DbStore):
|
||||
super().__init__(db_path or host_db_path(), migrations)
|
||||
|
||||
def write_audit_entry(self, entry: AuditEntry) -> Path:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO supervise_audit_entries (
|
||||
@@ -66,7 +66,7 @@ class AuditStore(DbStore):
|
||||
def read_audit_entries(self, component: str, slug: str) -> list[AuditEntry]:
|
||||
if not self.db_path.is_file():
|
||||
return []
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT * FROM supervise_audit_entries
|
||||
|
||||
@@ -45,7 +45,8 @@ from typing import TYPE_CHECKING, Any, Generic, Sequence, TypeVar
|
||||
from ..agent_provider import AgentProvisionPlan, get_provider, build_agent_provision_plan
|
||||
from ..egress import EgressPlan
|
||||
from ..git_gate import GitGatePlan
|
||||
from ..log import die, info
|
||||
from ..log import die, info, warn
|
||||
from ..util import read_tty_line
|
||||
from ..manifest import Manifest, ManifestIndex
|
||||
from ..supervise import SupervisePlan
|
||||
from ..util import expand_tilde
|
||||
@@ -648,19 +649,26 @@ def __getattr__(name: str) -> Any:
|
||||
|
||||
def get_bottle_backend(
|
||||
name: str | None = None,
|
||||
*,
|
||||
prompt: bool = True,
|
||||
) -> BottleBackend[Any, Any]:
|
||||
"""Resolve the bottle backend.
|
||||
|
||||
`name` precedence:
|
||||
1. explicit arg (CLI `--backend=<name>` passes through here)
|
||||
1. explicit arg (e.g. resume passes the recorded backend name)
|
||||
2. BOT_BOTTLE_BACKEND env var
|
||||
3. `macos-container` on compatible macOS hosts
|
||||
4. `firecracker` on KVM-capable Linux hosts
|
||||
5. default `docker`
|
||||
3. auto-selection: VM backend first, docker fallback with prompt
|
||||
|
||||
`prompt` controls whether auto-selection may block on an interactive
|
||||
[i/d/q] prompt when falling back to docker. Pass `prompt=False` in
|
||||
non-interactive contexts (headless launches, CI) so the call dies
|
||||
with an actionable message instead of hanging.
|
||||
|
||||
Dies with a pointer at the known backends if the chosen name
|
||||
isn't implemented."""
|
||||
resolved = name or os.environ.get("BOT_BOTTLE_BACKEND") or _default_backend_name()
|
||||
resolved = name or os.environ.get("BOT_BOTTLE_BACKEND")
|
||||
if resolved is None:
|
||||
resolved = _auto_select_backend(prompt=prompt)
|
||||
backends = _get_backends()
|
||||
if resolved not in backends:
|
||||
known = ", ".join(sorted(backends))
|
||||
@@ -668,7 +676,35 @@ def get_bottle_backend(
|
||||
return backends[resolved]
|
||||
|
||||
|
||||
def _default_backend_name() -> str:
|
||||
def _platform_vm_suggestion() -> str:
|
||||
"""Platform-appropriate VM backend name for install suggestions."""
|
||||
return "macos-container" if sys.platform == "darwin" else "firecracker"
|
||||
|
||||
|
||||
def _print_vm_install_instructions() -> None:
|
||||
"""Print platform-appropriate VM backend install instructions to stderr."""
|
||||
vm = _platform_vm_suggestion()
|
||||
if vm == "macos-container":
|
||||
info("Install Apple Container: https://github.com/apple/container/releases")
|
||||
info("Then start the service: container system start")
|
||||
else:
|
||||
info("Install Firecracker: https://github.com/firecracker-microvm/firecracker/releases")
|
||||
info("Configure the host: ./cli.py backend setup")
|
||||
|
||||
|
||||
def _auto_select_backend(prompt: bool = True) -> str:
|
||||
"""Tier-1 / tier-2 backend auto-selection.
|
||||
|
||||
Tier 1: VM backend — macos-container on macOS when Apple Container is
|
||||
installed; firecracker on KVM-capable Linux even before the binary is
|
||||
present (its preflight prints an install pointer).
|
||||
|
||||
Tier 2: docker, with a security warning and an interactive prompt.
|
||||
When `prompt=False` (headless / CI), dies with an actionable message
|
||||
instead of blocking on a TTY read. When docker is also absent, prints
|
||||
VM install instructions and exits.
|
||||
"""
|
||||
# --- Tier 1: VM backend -----------------------------------------
|
||||
if has_backend("macos-container"):
|
||||
return "macos-container"
|
||||
# A KVM-capable Linux host defaults to firecracker even when the
|
||||
@@ -678,7 +714,37 @@ def _default_backend_name() -> str:
|
||||
from .firecracker import FirecrackerBottleBackend
|
||||
if FirecrackerBottleBackend.is_host_capable():
|
||||
return "firecracker"
|
||||
return "docker"
|
||||
|
||||
# --- Tier 2: docker fallback ------------------------------------
|
||||
if not has_backend("docker"):
|
||||
info("No backend available on this host.")
|
||||
_print_vm_install_instructions()
|
||||
die("no backend available; install a VM backend and re-run")
|
||||
|
||||
vm = _platform_vm_suggestion()
|
||||
warn(
|
||||
"docker is less secure than VM backends — "
|
||||
"containers share the host kernel."
|
||||
)
|
||||
if not prompt:
|
||||
die(
|
||||
f"no VM backend available; set BOT_BOTTLE_BACKEND=docker to proceed "
|
||||
f"with docker, or install the {vm!r} backend."
|
||||
)
|
||||
sys.stderr.write(
|
||||
f"bot-bottle: For better isolation, install the {vm!r} backend.\n"
|
||||
f" [i] show {vm} install instructions and exit\n"
|
||||
" [d] use docker anyway\n"
|
||||
" [q] quit\n"
|
||||
"bot-bottle: choice [i/d/q]: "
|
||||
)
|
||||
sys.stderr.flush()
|
||||
reply = read_tty_line().strip().lower()
|
||||
if reply == "d":
|
||||
return "docker"
|
||||
if reply == "i":
|
||||
_print_vm_install_instructions()
|
||||
die("not proceeding with docker; install a VM backend or set BOT_BOTTLE_BACKEND=docker")
|
||||
|
||||
|
||||
def known_backend_names() -> tuple[str, ...]:
|
||||
|
||||
@@ -27,10 +27,14 @@ def _docker_on_path() -> bool:
|
||||
def _daemon_reachable() -> bool:
|
||||
if not _docker_on_path():
|
||||
return False
|
||||
return subprocess.run(
|
||||
["docker", "info"],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False,
|
||||
).returncode == 0
|
||||
try:
|
||||
return subprocess.run(
|
||||
["docker", "info"],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
check=False, timeout=5,
|
||||
).returncode == 0
|
||||
except subprocess.TimeoutExpired:
|
||||
return False
|
||||
|
||||
|
||||
def _print_install_pointer() -> None:
|
||||
|
||||
@@ -38,25 +38,39 @@ _BUILD_TIMEOUT_SECONDS = 900.0
|
||||
|
||||
|
||||
def _dockerfile_hash(dockerfile: Path) -> str:
|
||||
"""Cache key: the Dockerfile's content. The shipped agent Dockerfiles
|
||||
COPY nothing from the build context (see .dockerignore), so their content
|
||||
fully determines the image; a Dockerfile that adds COPY will want the
|
||||
"""The Dockerfile's content hash. The shipped agent Dockerfiles COPY
|
||||
nothing from the build context (see .dockerignore), so their content fully
|
||||
determines the built image; a Dockerfile that adds COPY will want the
|
||||
context folded in here too."""
|
||||
return hashlib.sha256(dockerfile.read_bytes()).hexdigest()[:16]
|
||||
|
||||
|
||||
def _rootfs_digest(dockerfile: Path) -> str:
|
||||
"""Cache key for the built AND boot-injected agent rootfs. Two inputs
|
||||
determine the on-disk rootfs: the Dockerfile (the image) and the guest init
|
||||
injected into it (`util._GUEST_INIT`). Folding the init in means a fix to
|
||||
it — e.g. making /tmp world-writable — busts the cache instead of silently
|
||||
reusing a stale rootfs built with the old init."""
|
||||
h = hashlib.sha256()
|
||||
h.update(_dockerfile_hash(dockerfile).encode())
|
||||
h.update(b"\0")
|
||||
h.update(util._GUEST_INIT.encode())
|
||||
return h.hexdigest()[:16]
|
||||
|
||||
|
||||
def build_agent_rootfs_dir(
|
||||
dockerfile: Path, *, image_tag: str, smoke_test: tuple[str, ...] = (),
|
||||
) -> Path:
|
||||
"""Build `dockerfile` in the infra VM (buildah, no host docker), export its
|
||||
rootfs, inject the guest boot bits, and return the cached base dir — the
|
||||
same shape `util.build_rootfs_ext4` consumes. Cached by Dockerfile content,
|
||||
so a repeat launch skips the rebuild.
|
||||
same shape `util.build_rootfs_ext4` consumes. Cached by Dockerfile content
|
||||
+ injected guest init, so a repeat launch skips the rebuild but an init or
|
||||
Dockerfile change rebuilds.
|
||||
|
||||
`smoke_test` (the provider's declared argv, e.g. `("claude","--version")`)
|
||||
is run in the freshly built image before export, catching an npm
|
||||
silent-failure image at build time rather than at first agent use."""
|
||||
digest = _dockerfile_hash(dockerfile)
|
||||
digest = _rootfs_digest(dockerfile)
|
||||
base = util.cache_dir() / "rootfs" / f"agent-{digest}"
|
||||
if (base / ".bb-ready").is_file():
|
||||
info(f"using cached agent rootfs {base.name}")
|
||||
|
||||
@@ -85,6 +85,11 @@ def infra_artifact_version(init_script: str, *, repo_root: Path = _REPO_ROOT) ->
|
||||
h.update(name.encode())
|
||||
h.update(b"\0")
|
||||
h.update((repo_root / name).read_bytes())
|
||||
h.update(b"pyproject.toml\0")
|
||||
h.update((repo_root / "pyproject.toml").read_bytes())
|
||||
h.update(b"dropbear\0")
|
||||
dropbear = util.dropbear_path()
|
||||
h.update(dropbear.read_bytes() if dropbear.is_file() else b"<missing>")
|
||||
h.update(b"init\0")
|
||||
h.update(init_script.encode())
|
||||
return h.hexdigest()[:16]
|
||||
@@ -111,6 +116,7 @@ def artifact_url(version: str, filename: str) -> str:
|
||||
|
||||
_GZ_NAME = "rootfs.ext4.gz"
|
||||
_SHA_NAME = "rootfs.ext4.gz.sha256"
|
||||
_CANDIDATE_DIR_ENV = "BOT_BOTTLE_INFRA_ARTIFACT_DIR"
|
||||
|
||||
|
||||
def _cache_root(version: str) -> Path:
|
||||
@@ -160,6 +166,33 @@ def ensure_artifact_gz(version: str) -> Path:
|
||||
"""The verified, cached `rootfs.ext4.gz` for `version` — downloading it (and
|
||||
its `.sha256`) once, then reusing it. Fail-closed on a checksum mismatch:
|
||||
the partial is removed and we die rather than boot an unverified rootfs."""
|
||||
candidate_dir = os.environ.get(_CANDIDATE_DIR_ENV, "").strip()
|
||||
if candidate_dir:
|
||||
root = Path(candidate_dir)
|
||||
version_file = root / "version.txt"
|
||||
# Guard the read so a missing version.txt is a clean error, not a raw
|
||||
# FileNotFoundError.
|
||||
if not version_file.is_file():
|
||||
die(f"infra candidate bundle is incomplete: {root}")
|
||||
declared = version_file.read_text(encoding="utf-8").strip()
|
||||
if declared != version:
|
||||
die(
|
||||
f"infra candidate version mismatch: expected {version}, "
|
||||
f"bundle contains {declared or '<empty>'}"
|
||||
)
|
||||
gz = root / _GZ_NAME
|
||||
sha = root / _SHA_NAME
|
||||
if not gz.is_file() or not sha.is_file():
|
||||
die(f"infra candidate bundle is incomplete: {root}")
|
||||
expected = sha.read_text().split()[0].strip().lower()
|
||||
actual = _sha256_file(gz)
|
||||
if actual != expected:
|
||||
die(
|
||||
f"infra candidate checksum mismatch for {version}:\n"
|
||||
f" expected {expected}\n actual {actual}"
|
||||
)
|
||||
return gz
|
||||
|
||||
root = _cache_root(version)
|
||||
root.mkdir(parents=True, exist_ok=True)
|
||||
gz = root / _GZ_NAME
|
||||
|
||||
@@ -161,20 +161,23 @@ def ensure_running() -> InfraVm:
|
||||
slot = netpool.orch_slot()
|
||||
url = f"http://{slot.guest_ip}:{CONTROL_PLANE_PORT}"
|
||||
key = _infra_dir() / "id_ed25519"
|
||||
if key.exists() and _health_ok(url):
|
||||
want = _expected_version()
|
||||
if _adoptable(key, url, want):
|
||||
info(f"adopting running infra VM at {url}")
|
||||
return InfraVm(guest_ip=slot.guest_ip, private_key=key)
|
||||
|
||||
with _singleton_lock():
|
||||
# Re-check under the lock: another launcher may have booted it while
|
||||
# we waited for the lock (double-checked, so we adopt not re-boot).
|
||||
if key.exists() and _health_ok(url):
|
||||
if _adoptable(key, url, want):
|
||||
info(f"adopting running infra VM at {url}")
|
||||
return InfraVm(guest_ip=slot.guest_ip, private_key=key)
|
||||
stop() # clear a stale/hung VM holding the link before booting fresh
|
||||
# Clear a stale/hung/OUTDATED VM holding the link before booting fresh.
|
||||
stop()
|
||||
ensure_built()
|
||||
infra = boot()
|
||||
wait_for_health(infra)
|
||||
_record_booted_version(want)
|
||||
return infra
|
||||
|
||||
|
||||
@@ -193,9 +196,15 @@ def _singleton_lock() -> Generator[None, None, None]:
|
||||
|
||||
|
||||
def stop() -> None:
|
||||
"""Stop the infra VM singleton (idempotent — absent is success)."""
|
||||
"""Stop the infra VM singleton (idempotent — absent is success). Reaps the
|
||||
recorded VMM AND any orphaned firecracker still bound to the infra config —
|
||||
the PID file drifts after crashes / out-of-band kills, and a survivor would
|
||||
hold the orchestrator TAP so the next boot dies with "tap … Resource busy".
|
||||
Drops the version marker so a stopped VM is never treated as adoptable."""
|
||||
_kill_pidfile()
|
||||
_kill_infra_firecrackers()
|
||||
_pid_file().unlink(missing_ok=True)
|
||||
_version_file().unlink(missing_ok=True)
|
||||
|
||||
|
||||
def boot() -> InfraVm:
|
||||
@@ -238,6 +247,36 @@ def _pid_file() -> Path:
|
||||
return _infra_dir() / "vm.pid"
|
||||
|
||||
|
||||
def _version_file() -> Path:
|
||||
"""Records the infra-artifact version the *running* VM booted from, so a
|
||||
later launcher can tell whether the singleton it found is the current code.
|
||||
Without it, a healthy VM built from an older image gets adopted forever and
|
||||
the new code never boots — every infra change would need an out-of-band
|
||||
kill to dislodge the stale VM (and races whatever launched next)."""
|
||||
return _infra_dir() / "booted-version"
|
||||
|
||||
|
||||
def _expected_version() -> str:
|
||||
return infra_artifact.infra_artifact_version(_infra_init())
|
||||
|
||||
|
||||
def _adoptable(key: Path, url: str, want: str) -> bool:
|
||||
"""Adopt a running infra VM only if it booted from the CURRENT version and
|
||||
its control plane is healthy. A missing/mismatched marker means a prior
|
||||
launcher booted an older infra image — reboot rather than reuse stale code."""
|
||||
if not key.exists():
|
||||
return False
|
||||
try:
|
||||
booted = _version_file().read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
return False
|
||||
return booted == want and _health_ok(url)
|
||||
|
||||
|
||||
def _record_booted_version(version: str) -> None:
|
||||
_version_file().write_text(version + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
# The registry "volume": a host-side ext4 file attached to the infra VM as a
|
||||
# second virtio-block device (guest /dev/vdb), mounted at the control plane's
|
||||
# DB dir. It outlives the ephemeral rootfs, so the bottle registry survives an
|
||||
@@ -309,6 +348,28 @@ def _kill_pidfile() -> None:
|
||||
pass
|
||||
|
||||
|
||||
def _kill_infra_firecrackers(proc_root: Path = Path("/proc")) -> None:
|
||||
"""SIGKILL any firecracker VMM whose `--config-file` is this host's infra
|
||||
config, independent of the PID file — reaps orphans it lost track of so the
|
||||
orchestrator TAP is free to rebind. Scoped to the infra config path, so the
|
||||
interactive pool's agent/infra VMs (other config paths) are untouched."""
|
||||
cfg = str(_infra_dir() / "config.json")
|
||||
for entry in proc_root.iterdir():
|
||||
if not entry.name.isdigit():
|
||||
continue
|
||||
try:
|
||||
if (entry / "comm").read_text().strip() != "firecracker":
|
||||
continue
|
||||
args = (entry / "cmdline").read_bytes().split(b"\0")
|
||||
except OSError:
|
||||
continue # process vanished / not ours
|
||||
if any(a.decode("utf-8", "replace") == cfg for a in args):
|
||||
try:
|
||||
os.kill(int(entry.name), signal.SIGKILL)
|
||||
except (OSError, ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def _health_ok(url: str) -> bool:
|
||||
try:
|
||||
with urllib.request.urlopen(f"{url}/health", timeout=1.0) as resp:
|
||||
@@ -433,7 +494,7 @@ BOT_BOTTLE_ROOT=/var/lib/bot-bottle python3 -m bot_bottle.orchestrator \\
|
||||
BOT_BOTTLE_GATEWAY_DAEMONS=egress,git-http,supervise \\
|
||||
BOT_BOTTLE_ORCHESTRATOR_URL=http://127.0.0.1:{CONTROL_PLANE_PORT} \\
|
||||
SUPERVISE_DB_PATH=/var/lib/bot-bottle/db/bot-bottle.db \\
|
||||
python3 /app/gateway_init.py &
|
||||
python3 -m bot_bottle.gateway_init &
|
||||
|
||||
# Reap as PID 1; children are backgrounded, so `wait` blocks.
|
||||
while : ; do wait ; done
|
||||
|
||||
@@ -11,7 +11,8 @@ The `<version>` is `infra_artifact.infra_artifact_version(...)`, the content
|
||||
hash of the rootfs inputs, so a launch host at the same code checkout resolves
|
||||
the exact artifact this produced.
|
||||
|
||||
python3 -m bot_bottle.backend.firecracker.publish_infra [--dry-run] [--force]
|
||||
python3 -m bot_bottle.backend.firecracker.publish_infra --output DIR
|
||||
python3 -m bot_bottle.backend.firecracker.publish_infra --publish-dir DIR
|
||||
|
||||
Auth: a token with `write:package` on the target owner, from
|
||||
`BOT_BOTTLE_INFRA_ARTIFACT_TOKEN`.
|
||||
@@ -24,7 +25,6 @@ import gzip
|
||||
import hashlib
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
@@ -131,36 +131,79 @@ def build_artifact(out_dir: Path) -> tuple[str, Path, Path]:
|
||||
return version, gz, sha
|
||||
|
||||
|
||||
def _publish_bundle(root: Path, token: str) -> str:
|
||||
version_file = root / "version.txt"
|
||||
# Guard the read so a missing version.txt is a clean error, not a raw
|
||||
# FileNotFoundError.
|
||||
if not version_file.is_file():
|
||||
raise SystemExit(f"incomplete artifact bundle: {root}")
|
||||
version = version_file.read_text(encoding="utf-8").strip()
|
||||
expected = infra_artifact.infra_artifact_version(infra_vm._infra_init())
|
||||
if version != expected:
|
||||
raise SystemExit(
|
||||
f"artifact bundle version {version!r} does not match checkout {expected!r}"
|
||||
)
|
||||
gz = root / "rootfs.ext4.gz"
|
||||
sha = root / "rootfs.ext4.gz.sha256"
|
||||
if not gz.is_file() or not sha.is_file():
|
||||
raise SystemExit(f"incomplete artifact bundle: {root}")
|
||||
expected_sha = sha.read_text().split()[0].strip().lower()
|
||||
if _sha256(gz) != expected_sha:
|
||||
raise SystemExit("artifact bundle checksum mismatch")
|
||||
|
||||
gz_url = infra_artifact.artifact_url(version, gz.name)
|
||||
sha_url = infra_artifact.artifact_url(version, sha.name)
|
||||
about_url = infra_artifact.artifact_url(version, _ABOUT_NAME)
|
||||
|
||||
# Publishing is idempotent. If this exact complete artifact is already
|
||||
# present, a test-only main commit is a no-op. Otherwise clear any partial
|
||||
# upload left by an interrupted prior attempt and upload the complete set.
|
||||
try:
|
||||
with urllib.request.urlopen(infra_artifact._open(sha_url)) as resp:
|
||||
remote_sha = resp.read().decode("utf-8").split()[0].strip().lower()
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code != 404:
|
||||
raise SystemExit(f"checking existing artifact failed (HTTP {e.code})")
|
||||
remote_sha = ""
|
||||
except urllib.error.URLError as e:
|
||||
raise SystemExit(f"registry unreachable: {sha_url} ({e.reason})")
|
||||
if remote_sha == expected_sha:
|
||||
print(f"infra rootfs {version} already published")
|
||||
return version
|
||||
|
||||
for url in (gz_url, sha_url, about_url):
|
||||
_delete(url, token)
|
||||
_put(gz_url, gz, token)
|
||||
_put(sha_url, sha.read_bytes(), token)
|
||||
_put(about_url, _ABOUT_TEXT.encode(), token)
|
||||
return version
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="publish_infra", description="Build + publish the infra rootfs artifact.")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="build the artifact but do not upload")
|
||||
parser.add_argument("--force", action="store_true",
|
||||
help="overwrite an already-published artifact of this version")
|
||||
mode = parser.add_mutually_exclusive_group(required=True)
|
||||
mode.add_argument("--output", type=Path,
|
||||
help="build a candidate bundle in DIR without publishing")
|
||||
mode.add_argument("--publish-dir", type=Path,
|
||||
help="publish an already-built and tested candidate bundle")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
_, _, token = infra_artifact._config()
|
||||
if not args.dry_run and not token:
|
||||
if args.publish_dir is not None and not token:
|
||||
raise SystemExit(
|
||||
"no publish token: set BOT_BOTTLE_INFRA_ARTIFACT_TOKEN to a token "
|
||||
"with write:package")
|
||||
|
||||
with tempfile.TemporaryDirectory(prefix="bb-publish-infra.") as tmp:
|
||||
version, gz, sha = build_artifact(Path(tmp))
|
||||
gz_url = infra_artifact.artifact_url(version, gz.name)
|
||||
sha_url = infra_artifact.artifact_url(version, sha.name)
|
||||
about_url = infra_artifact.artifact_url(version, _ABOUT_NAME)
|
||||
if args.dry_run:
|
||||
print(f"dry-run: would upload -> {gz_url}")
|
||||
return 0
|
||||
if args.force:
|
||||
_delete(gz_url, token)
|
||||
_delete(sha_url, token)
|
||||
_delete(about_url, token)
|
||||
_put(gz_url, gz, token) # streamed from disk (hundreds of MB)
|
||||
_put(sha_url, sha.read_bytes(), token) # tiny, in-memory is fine
|
||||
_put(about_url, _ABOUT_TEXT.encode(), token) # package description
|
||||
if args.output is not None:
|
||||
args.output.mkdir(parents=True, exist_ok=True)
|
||||
version, _gz, _sha = build_artifact(args.output)
|
||||
(args.output / "version.txt").write_text(version + "\n", encoding="utf-8")
|
||||
print(f"built infra rootfs candidate {version}")
|
||||
return 0
|
||||
|
||||
assert args.publish_dir is not None
|
||||
version = _publish_bundle(args.publish_dir, token)
|
||||
print(f"published infra rootfs {version}")
|
||||
return 0
|
||||
|
||||
|
||||
@@ -88,7 +88,7 @@ def require_firecracker() -> None:
|
||||
booting a VM without it."""
|
||||
if not is_linux():
|
||||
die("firecracker backend is only supported on Linux (KVM). "
|
||||
"On macOS use --backend=macos-container.")
|
||||
"On macOS use the macos-container backend.")
|
||||
if shutil.which("firecracker") is None:
|
||||
info("Firecracker is required but was not found on PATH.")
|
||||
info("Install: https://github.com/firecracker-microvm/firecracker/releases")
|
||||
@@ -368,6 +368,11 @@ mount -t devtmpfs dev /dev 2>/dev/null
|
||||
mkdir -p /dev/pts && mount -t devpts devpts /dev/pts 2>/dev/null
|
||||
mount -o remount,rw / 2>/dev/null
|
||||
|
||||
# /tmp must be world-writable + sticky. The rootless rootfs build can land
|
||||
# it 0755/root-owned, leaving the agent (uid 1000 node) unable to create
|
||||
# scratch dirs there — git worktrees, build temp, `git init /tmp/...`, etc.
|
||||
mkdir -p /tmp && chmod 1777 /tmp
|
||||
|
||||
# Install the per-bottle SSH pubkey from the kernel cmdline.
|
||||
KEY=$(sed -n 's/.*bb_pubkey=\([^ ]*\).*/\1/p' /proc/cmdline | base64 -d 2>/dev/null)
|
||||
if [ -n "$KEY" ]; then
|
||||
|
||||
@@ -68,9 +68,14 @@ class MacosContainerBottle(Bottle):
|
||||
# reaches the agent (PRD 0070): registration mints it *after* the
|
||||
# container exists — its source IP is the registration key and Apple
|
||||
# Container assigns that by DHCP — so it cannot be in the run-time env
|
||||
# the way docker's compose spec does it. `container exec --env` wins
|
||||
# over the run-time value, so the token-bearing proxy URL set here
|
||||
# supersedes the token-less one baked in at launch.
|
||||
# the way docker's compose spec does it.
|
||||
#
|
||||
# `container exec --env` does NOT override a run-time value — it
|
||||
# appends, leaving duplicate entries in the agent's `environ` whose
|
||||
# resolution is runtime-specific (Node last-wins, Rust first-wins). So
|
||||
# nothing here may rely on superseding: the proxy vars are supplied
|
||||
# *only* at exec time and are deliberately absent from the run-time
|
||||
# env. See `launch._agent_env_entries`.
|
||||
self._exec_env = dict(exec_env or {})
|
||||
self._closed = False
|
||||
|
||||
|
||||
@@ -8,7 +8,11 @@ from ...bottle_state import read_metadata
|
||||
from .. import ActiveAgent
|
||||
from .infra import INFRA_NAME
|
||||
|
||||
_PREFIX = "bot-bottle-"
|
||||
# The name every agent container carries: `bot-bottle-<slug>`. Exported
|
||||
# because callers that act on a running bottle (gateway-host rewrites,
|
||||
# registry reconciliation) have to map an enumerated slug back to a
|
||||
# container name.
|
||||
CONTAINER_NAME_PREFIX = "bot-bottle-"
|
||||
# The shared per-host infra container carries the same prefix as agent
|
||||
# containers but is infrastructure, not a bottle — one control plane + gateway
|
||||
# serves every agent, so listing it as an agent would invent one per host.
|
||||
@@ -26,9 +30,9 @@ def enumerate_active() -> list[ActiveAgent]:
|
||||
return []
|
||||
out: list[ActiveAgent] = []
|
||||
for name in sorted(line.strip() for line in result.stdout.splitlines()):
|
||||
if not name.startswith(_PREFIX) or name in _INFRA_NAMES:
|
||||
if not name.startswith(CONTAINER_NAME_PREFIX) or name in _INFRA_NAMES:
|
||||
continue
|
||||
slug = name[len(_PREFIX):]
|
||||
slug = name[len(CONTAINER_NAME_PREFIX):]
|
||||
metadata = read_metadata(slug)
|
||||
out.append(ActiveAgent(
|
||||
backend_name="macos-container",
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
"""Stable gateway name for macOS agents, via each bottle's `/etc/hosts`.
|
||||
|
||||
The shared gateway's address is assigned by vmnet's DHCP and changes whenever
|
||||
the infra container is recreated — a source-hash bump, an image upgrade, a
|
||||
crash. Every agent-facing URL (egress proxy, git-http, supervise) embeds that
|
||||
address, and the proxy URL reaches the agent as **process environment** at
|
||||
`container exec` time. A running process's `environ` cannot be rewritten from
|
||||
outside, so a moved gateway used to strand every running bottle permanently:
|
||||
not degraded, unreachable, until the bottle was relaunched and its session
|
||||
thrown away.
|
||||
|
||||
So the agent never learns the address. It is given a stable *name*
|
||||
(`GATEWAY_HOSTNAME`) in every URL, resolved through its own `/etc/hosts`.
|
||||
Unlike `environ`, that is a file — it can be rewritten inside a container that
|
||||
is already running, so a gateway that comes back at a new address is picked up
|
||||
by live bottles instead of orphaning them.
|
||||
|
||||
Apple Container 1.0 offers no container-name DNS on a user network (the only
|
||||
nameserver an agent sees is vmnet's, which does not know container names) and
|
||||
`container run` has no `--add-host`, so the entry is written by exec after the
|
||||
container starts.
|
||||
|
||||
Writing it needs root, and the agent runs as `node`: the agent therefore
|
||||
cannot repoint its own gateway name, while the host (which drives `container
|
||||
exec --user root`) can. That asymmetry is deliberate — keep it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ...log import warn
|
||||
from . import util as container_mod
|
||||
from .enumerate import CONTAINER_NAME_PREFIX, enumerate_active
|
||||
|
||||
# The name every agent-facing gateway URL uses. Must not collide with a real
|
||||
# DNS name the agent might resolve; it is bottle-local by construction.
|
||||
GATEWAY_HOSTNAME = "bot-bottle-gateway"
|
||||
|
||||
# Marker so the rewrite is idempotent and only ever touches our own line —
|
||||
# the rest of /etc/hosts (localhost, the container's own name) is preserved.
|
||||
_MARKER = "# bot-bottle gateway"
|
||||
|
||||
|
||||
def _rewrite_script(gateway_ip: str) -> str:
|
||||
"""A shell one-liner that replaces our managed line in `/etc/hosts`.
|
||||
|
||||
Rewrites in place via a temp file + `cat` rather than `mv`, so the file
|
||||
keeps its original inode, ownership, and mode — a bind-mounted or
|
||||
pre-created `/etc/hosts` must not be replaced by a root-owned 0644 copy
|
||||
that the runtime then refuses to update.
|
||||
"""
|
||||
return (
|
||||
"set -e; "
|
||||
f"grep -v '{_MARKER}' /etc/hosts > /tmp/.bb-hosts || true; "
|
||||
f"printf '%s %s %s\\n' '{gateway_ip}' '{GATEWAY_HOSTNAME}' "
|
||||
f"'{_MARKER}' >> /tmp/.bb-hosts; "
|
||||
"cat /tmp/.bb-hosts > /etc/hosts; "
|
||||
"rm -f /tmp/.bb-hosts"
|
||||
)
|
||||
|
||||
|
||||
def set_gateway_host(container_name: str, gateway_ip: str) -> None:
|
||||
"""Point `GATEWAY_HOSTNAME` at `gateway_ip` inside one running container.
|
||||
|
||||
Must run before the agent is exec'd: the agent's proxy URL names the
|
||||
gateway, so the entry has to exist for its first connection. Idempotent —
|
||||
re-running with the same address is a no-op in effect.
|
||||
"""
|
||||
container_mod.exec_container_as_root(
|
||||
container_name, ["sh", "-c", _rewrite_script(gateway_ip)],
|
||||
)
|
||||
|
||||
|
||||
def refresh_gateway_host(gateway_ip: str) -> list[str]:
|
||||
"""Re-point every running bottle at the current gateway address.
|
||||
|
||||
Called once the shared gateway is known to be up, so a bottle stranded by
|
||||
an earlier gateway restart re-attaches instead of needing a relaunch.
|
||||
Returns the containers updated.
|
||||
|
||||
Best-effort per bottle: one container that refuses the write (already
|
||||
exiting, say) must not stop the others from being repaired, and must not
|
||||
fail the launch that triggered the sweep.
|
||||
"""
|
||||
updated: list[str] = []
|
||||
for agent in enumerate_active():
|
||||
name = f"{CONTAINER_NAME_PREFIX}{agent.slug}"
|
||||
try:
|
||||
set_gateway_host(name, gateway_ip)
|
||||
updated.append(name)
|
||||
# One bad bottle must not stop the sweep, so this is deliberately broad.
|
||||
except Exception as e: # noqa: BLE001 # pylint: disable=broad-exception-caught
|
||||
warn(f"could not re-point {name} at the gateway: {e}")
|
||||
return updated
|
||||
|
||||
|
||||
__all__ = ["GATEWAY_HOSTNAME", "set_gateway_host", "refresh_gateway_host"]
|
||||
@@ -101,7 +101,7 @@ def _init_script(port: int) -> str:
|
||||
# Gateway data plane, multi-tenant against the local control plane.
|
||||
f"( cd /app && BOT_BOTTLE_GATEWAY_DAEMONS={_GATEWAY_DAEMONS} "
|
||||
f"BOT_BOTTLE_ORCHESTRATOR_URL=http://127.0.0.1:{port} "
|
||||
f"SUPERVISE_DB_PATH={_DB_PATH_IN_CONTAINER} python3 /app/gateway_init.py ) &\n"
|
||||
f"SUPERVISE_DB_PATH={_DB_PATH_IN_CONTAINER} python3 -m bot_bottle.gateway_init ) &\n"
|
||||
"while : ; do wait ; done\n"
|
||||
)
|
||||
|
||||
|
||||
@@ -59,6 +59,11 @@ from ..docker.egress import EGRESS_PORT
|
||||
from ..util import AGENT_CA_BUNDLE, AGENT_CA_PATH
|
||||
from . import util as container_mod
|
||||
from .bottle import MacosContainerBottle
|
||||
from .gateway_hosts import (
|
||||
GATEWAY_HOSTNAME,
|
||||
refresh_gateway_host,
|
||||
set_gateway_host,
|
||||
)
|
||||
from .bottle_plan import MacosContainerBottlePlan
|
||||
from .consolidated_launch import (
|
||||
GatewayEndpoint,
|
||||
@@ -100,6 +105,11 @@ def launch(
|
||||
# Step 1: the per-host singletons. Must precede the agent run — its
|
||||
# proxy env needs the gateway's address at `container run` time.
|
||||
endpoint = ensure_gateway()
|
||||
# The gateway's address may have changed since these bottles launched
|
||||
# (any infra recreate re-runs DHCP). They name the gateway rather than
|
||||
# address it, so re-pointing /etc/hosts re-attaches them in place
|
||||
# instead of leaving them stranded until relaunch.
|
||||
refresh_gateway_host(endpoint.gateway_ip)
|
||||
|
||||
# Step 2: mint this bottle's deploy keys, then point it at the SHARED
|
||||
# gateway's CA + git-http/supervise ports.
|
||||
@@ -117,6 +127,9 @@ def launch(
|
||||
# attribution key; `--cap-drop CAP_NET_RAW` at run is what makes it
|
||||
# unforgeable. Poll: `container run --detach` can return before vmnet's
|
||||
# DHCP has assigned the address.
|
||||
# Resolve the gateway name before anything execs: every agent-facing
|
||||
# URL uses it, so the entry must exist for the first connection.
|
||||
set_gateway_host(plan.container_name, endpoint.gateway_ip)
|
||||
source_ip = container_mod.wait_container_ipv4_on_network(
|
||||
plan.container_name, endpoint.network,
|
||||
)
|
||||
@@ -231,13 +244,20 @@ def _stamp_agent_urls(
|
||||
) -> MacosContainerBottlePlan:
|
||||
"""Point the agent's git-gate insteadOf rewrites + supervise MCP at the
|
||||
shared gateway's ports. Both bypass the egress proxy (NO_PROXY covers the
|
||||
gateway address)."""
|
||||
gateway name).
|
||||
|
||||
Addressed by `GATEWAY_HOSTNAME`, never by IP: these URLs are baked into
|
||||
the agent's gitconfig and MCP config at provision time, so an address here
|
||||
would strand the bottle the moment the gateway moved. The name is resolved
|
||||
per connection through `/etc/hosts`, which stays rewritable while the
|
||||
bottle runs."""
|
||||
del endpoint # addressed by name; the address reaches the bottle via /etc/hosts
|
||||
git_gate_url = (
|
||||
f"http://{endpoint.gateway_ip}:{_GIT_HTTP_PORT}"
|
||||
f"http://{GATEWAY_HOSTNAME}:{_GIT_HTTP_PORT}"
|
||||
if plan.git_gate_plan.upstreams else ""
|
||||
)
|
||||
supervise_url = (
|
||||
f"http://{endpoint.gateway_ip}:{SUPERVISE_PORT}/"
|
||||
f"http://{GATEWAY_HOSTNAME}:{SUPERVISE_PORT}/"
|
||||
if plan.supervise_plan is not None else ""
|
||||
)
|
||||
return dataclasses.replace(
|
||||
@@ -247,30 +267,43 @@ def _stamp_agent_urls(
|
||||
)
|
||||
|
||||
|
||||
def _proxy_url(gateway_ip: str, identity_token: str = "") -> str:
|
||||
def _proxy_url(identity_token: str = "") -> str:
|
||||
"""The agent's egress proxy URL. The identity token rides as proxy
|
||||
credentials — the gateway reads Proxy-Authorization, resolves the
|
||||
(source_ip, token) pair against the control plane, and strips it before
|
||||
upstream. Without a valid pair `/resolve` denies the request (#366)."""
|
||||
upstream. Without a valid pair `/resolve` denies the request (#366).
|
||||
|
||||
Names the gateway rather than addressing it: this URL reaches the agent as
|
||||
process environment, which cannot be rewritten once the agent is running,
|
||||
so an address baked here is unfixable if the gateway moves."""
|
||||
cred = f"bottle:{identity_token}@" if identity_token else ""
|
||||
return f"http://{cred}{gateway_ip}:{EGRESS_PORT}"
|
||||
return f"http://{cred}{GATEWAY_HOSTNAME}:{EGRESS_PORT}"
|
||||
|
||||
|
||||
def _no_proxy(gateway_ip: str) -> str:
|
||||
def _no_proxy() -> str:
|
||||
# git-http + supervise live on the gateway and must NOT go through the
|
||||
# egress proxy — the agent reaches them directly by its address.
|
||||
return f"localhost,127.0.0.1,{gateway_ip}"
|
||||
# egress proxy — the agent reaches them directly by name. Deliberately
|
||||
# address-free: NO_PROXY is baked into the run-time env and is therefore
|
||||
# just as unfixable as the proxy URL if the gateway moves.
|
||||
return f"localhost,127.0.0.1,{GATEWAY_HOSTNAME}"
|
||||
|
||||
|
||||
def _identity_proxy_env(
|
||||
endpoint: GatewayEndpoint, identity_token: str,
|
||||
) -> dict[str, str]:
|
||||
"""The token-bearing proxy env applied at `container exec`. It supersedes
|
||||
the token-less run-time value (exec `--env` wins), which is the only way to
|
||||
get the token in: it does not exist until after the container runs."""
|
||||
"""The token-bearing proxy env applied at `container exec` — the only way
|
||||
to get the token in, since it does not exist until after the container
|
||||
runs (registration keys on the DHCP-assigned address).
|
||||
|
||||
This is the *sole* source of `*_PROXY` for the agent. It deliberately does
|
||||
not rely on overriding a run-time value: `container exec --env` appends
|
||||
rather than replaces, so a run-time `HTTPS_PROXY` would survive alongside
|
||||
this one and first-wins runtimes would read the wrong entry. See
|
||||
`_agent_env_entries`."""
|
||||
if not identity_token:
|
||||
return {}
|
||||
url = _proxy_url(endpoint.gateway_ip, identity_token)
|
||||
del endpoint # the gateway is named, not addressed
|
||||
url = _proxy_url(identity_token)
|
||||
return {
|
||||
"HTTPS_PROXY": url, "HTTP_PROXY": url,
|
||||
"https_proxy": url, "http_proxy": url,
|
||||
@@ -317,16 +350,23 @@ def _agent_run_argv(
|
||||
def _agent_env_entries(
|
||||
plan: MacosContainerBottlePlan, endpoint: GatewayEndpoint,
|
||||
) -> tuple[str, ...]:
|
||||
# Token-less at run time — the token does not exist yet (see
|
||||
# `_identity_proxy_env`). Anything egressing before the exec-time override
|
||||
# is denied by `/resolve`, which is the safe direction.
|
||||
proxy_url = _proxy_url(endpoint.gateway_ip)
|
||||
no_proxy = _no_proxy(endpoint.gateway_ip)
|
||||
# No `*_PROXY` here on purpose. The token-bearing URL is applied at
|
||||
# `container exec` (`_identity_proxy_env`), and Apple's `container exec
|
||||
# --env` **appends** to the run-time environment rather than replacing it:
|
||||
# setting a token-less value here leaves two `HTTPS_PROXY` entries in the
|
||||
# agent's `environ`, token-less first. Which one a runtime reads is then
|
||||
# pure luck — Node takes the last (and worked), Rust's `std::env::var`
|
||||
# takes the first, so Codex proxied without its identity token and
|
||||
# `/resolve` fail-closed on every request.
|
||||
#
|
||||
# A token-less proxy URL has no legitimate consumer anyway: the init
|
||||
# process is `sleep` and everything that egresses arrives via exec. Its
|
||||
# only value was a tidy 403 for unattributed callers, which is not worth
|
||||
# silently dropping attribution for. Without it a process that egresses
|
||||
# before the exec-time env still fails closed — the agent network is
|
||||
# host-only, so there is no route off it except the gateway.
|
||||
no_proxy = _no_proxy()
|
||||
env = [
|
||||
f"HTTPS_PROXY={proxy_url}",
|
||||
f"HTTP_PROXY={proxy_url}",
|
||||
f"https_proxy={proxy_url}",
|
||||
f"http_proxy={proxy_url}",
|
||||
f"NO_PROXY={no_proxy}",
|
||||
f"no_proxy={no_proxy}",
|
||||
f"NODE_EXTRA_CA_CERTS={AGENT_CA_PATH}",
|
||||
|
||||
@@ -360,6 +360,21 @@ def exec_container(name: str, argv: list[str]) -> None:
|
||||
)
|
||||
|
||||
|
||||
def exec_container_as_root(name: str, argv: list[str]) -> None:
|
||||
"""`exec_container`, but as uid 0 inside the container.
|
||||
|
||||
For host-driven maintenance the agent itself must not be able to perform —
|
||||
rewriting `/etc/hosts` to point the gateway name at an address. The agent
|
||||
runs as `node`, so it cannot repoint its own gateway; the host can.
|
||||
"""
|
||||
result = _run_container_op([_CONTAINER, "exec", "--user", "root", name, *argv])
|
||||
if result.returncode != 0:
|
||||
die(
|
||||
f"container exec (root) in {name} failed: "
|
||||
f"{(result.stderr or '').strip() or '<no stderr>'}"
|
||||
)
|
||||
|
||||
|
||||
def _run_container_op(cmd: list[str]) -> subprocess.CompletedProcess[str]:
|
||||
result = subprocess.run(
|
||||
cmd,
|
||||
|
||||
@@ -38,6 +38,13 @@ COMMANDS = {
|
||||
"supervise": cmd_supervise,
|
||||
}
|
||||
|
||||
# Commands that manage host prerequisites (or are otherwise store-free) and
|
||||
# must run before — or without — a migrated DB. `backend` provisions/probes
|
||||
# the host (TAP pool, /dev/kvm, firecracker) and never opens the store, so
|
||||
# gating it on the schema breaks preflight on a fresh CI runner where stdin
|
||||
# isn't a TTY and the migration prompt can't be answered.
|
||||
NO_MIGRATION_COMMANDS = frozenset({"backend"})
|
||||
|
||||
|
||||
def usage() -> None:
|
||||
sys.stderr.write(f"usage: {PROG} <command> [args...]\n\n")
|
||||
@@ -80,7 +87,7 @@ def main(argv: list[str] | None = None) -> int:
|
||||
usage()
|
||||
die(f"unknown command: {command}")
|
||||
mgr = StoreManager.instance()
|
||||
if not mgr.is_migrated():
|
||||
if command not in NO_MIGRATION_COMMANDS and not mgr.is_migrated():
|
||||
sys.stderr.write("bot-bottle: database schema is out of date\n")
|
||||
sys.stderr.write("Migrate now? [y/N] ")
|
||||
sys.stderr.flush()
|
||||
|
||||
@@ -3,18 +3,10 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from ..util import read_tty_line as read_tty_line
|
||||
|
||||
PROG = "cli.py"
|
||||
USER_CWD = os.getcwd()
|
||||
REPO_DIR = str(Path(__file__).resolve().parent.parent.parent)
|
||||
|
||||
|
||||
def read_tty_line() -> str:
|
||||
"""Mirror `IFS= read -r REPLY </dev/tty`. Falls back to stdin."""
|
||||
try:
|
||||
with open("/dev/tty", "r", encoding="utf-8") as tty:
|
||||
return tty.readline().rstrip("\n")
|
||||
except OSError:
|
||||
return sys.stdin.readline().rstrip("\n")
|
||||
|
||||
@@ -21,16 +21,20 @@ from __future__ import annotations
|
||||
|
||||
import sys
|
||||
|
||||
from ..backend import get_bottle_backend, known_backend_names
|
||||
from ..backend import get_bottle_backend, has_backend, known_backend_names
|
||||
from ..log import info
|
||||
from ._common import read_tty_line
|
||||
|
||||
|
||||
def cmd_cleanup(_argv: list[str]) -> int:
|
||||
# Order: stable backend iteration so the y/N output is
|
||||
# deterministic across runs.
|
||||
# deterministic across runs. Skip backends whose runtime
|
||||
# isn't available on this host so e.g. macos-container
|
||||
# doesn't error on Linux.
|
||||
plans = [
|
||||
(name, get_bottle_backend(name)) for name in known_backend_names()
|
||||
(name, get_bottle_backend(name))
|
||||
for name in known_backend_names()
|
||||
if has_backend(name)
|
||||
]
|
||||
prepared = [(name, b, b.prepare_cleanup()) for name, b in plans]
|
||||
|
||||
|
||||
+7
-18
@@ -27,7 +27,6 @@ from ..backend import (
|
||||
BottleSpec,
|
||||
enumerate_active_agents,
|
||||
get_bottle_backend,
|
||||
known_backend_names,
|
||||
)
|
||||
from ..backend.docker import util as docker_mod
|
||||
from ..backend.docker.bottle_plan import DockerBottlePlan
|
||||
@@ -57,15 +56,6 @@ def cmd_start(argv: list[str]) -> int:
|
||||
"into a cached layer."
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--backend",
|
||||
choices=known_backend_names(),
|
||||
default=None,
|
||||
help=(
|
||||
"backend to launch the bottle on (default: $BOT_BOTTLE_BACKEND "
|
||||
"or host auto-selection). Overrides the env var when set."
|
||||
),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--headless",
|
||||
action="store_true",
|
||||
@@ -115,11 +105,10 @@ def cmd_start(argv: list[str]) -> int:
|
||||
os.environ["BOT_BOTTLE_NO_CACHE"] = "1"
|
||||
|
||||
manifest = ManifestIndex.resolve(USER_CWD)
|
||||
backend_name: str | None = args.backend
|
||||
|
||||
if args.headless:
|
||||
return _start_headless(
|
||||
manifest, args, dry_run=dry_run, backend_name=backend_name
|
||||
manifest, args, dry_run=dry_run
|
||||
)
|
||||
|
||||
agent_name: str | None = args.name
|
||||
@@ -170,7 +159,6 @@ def cmd_start(argv: list[str]) -> int:
|
||||
return _launch_bottle(
|
||||
spec,
|
||||
dry_run=dry_run,
|
||||
backend_name=backend_name,
|
||||
)
|
||||
|
||||
|
||||
@@ -182,7 +170,6 @@ def _start_headless(
|
||||
args: argparse.Namespace,
|
||||
*,
|
||||
dry_run: bool,
|
||||
backend_name: str | None,
|
||||
) -> int:
|
||||
"""Non-interactive launch path for orchestrators / CI / webhooks.
|
||||
|
||||
@@ -230,7 +217,6 @@ def _start_headless(
|
||||
return _launch_bottle(
|
||||
spec,
|
||||
dry_run=dry_run,
|
||||
backend_name=backend_name,
|
||||
assume_yes=True,
|
||||
headless_prompt_text=prompt,
|
||||
)
|
||||
@@ -268,15 +254,18 @@ def prepare_with_preflight(
|
||||
injected callable, prompt y/N via the injected callable.
|
||||
|
||||
`backend_name` selects which backend prepares the plan
|
||||
(`None` → `$BOT_BOTTLE_BACKEND` → host auto-selection). The CLI
|
||||
passes whatever `--backend` resolved to.
|
||||
(`None` → `$BOT_BOTTLE_BACKEND` → host auto-selection).
|
||||
|
||||
When `spec.headless` is True the docker-fallback prompt is suppressed:
|
||||
auto-selection dies with an actionable message rather than blocking
|
||||
on a TTY read (which would hang CI, webhook dispatch, and orchestrators).
|
||||
|
||||
Returns `(plan, identity)`. `plan` is None on dry-run or
|
||||
operator-N, but `identity` is set as soon as `backend.prepare`
|
||||
returns so callers can reap the prepare-time state dir via
|
||||
`settle_state(identity)` in their finally — exactly the existing
|
||||
semantics."""
|
||||
backend = get_bottle_backend(backend_name)
|
||||
backend = get_bottle_backend(backend_name, prompt=not spec.headless)
|
||||
plan = backend.prepare(spec, stage_dir=stage_dir)
|
||||
identity = _identity_from_plan(plan)
|
||||
|
||||
|
||||
@@ -23,9 +23,8 @@ from ...agent_provider import (
|
||||
provider_startup_args,
|
||||
)
|
||||
from ...backend.docker import util as docker_mod
|
||||
from ...egress import CLAUDE_HOST_CREDENTIAL_TOKEN_REF, EgressRoute
|
||||
from ...egress import EgressRoute
|
||||
from ...log import die, info, warn
|
||||
from .claude_auth import claude_host_access_token
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -119,6 +118,7 @@ class ClaudeAgentProvider(AgentProvider):
|
||||
color: str = "",
|
||||
provider_settings: dict[str, object] | None = None,
|
||||
) -> AgentProvisionPlan:
|
||||
del forward_host_credentials, host_env
|
||||
resolved_guest_env = dict(guest_env or {})
|
||||
startup_args = provider_startup_args(provider_settings)
|
||||
guest_home = self.guest_home
|
||||
@@ -180,24 +180,13 @@ class ClaudeAgentProvider(AgentProvider):
|
||||
claude_settings,
|
||||
f"{guest_home}/.claude/settings.json",
|
||||
))
|
||||
provisioned_env: dict[str, str] = {}
|
||||
if forward_host_credentials:
|
||||
_host_env = host_env or dict(os.environ)
|
||||
provisioned_env[CLAUDE_HOST_CREDENTIAL_TOKEN_REF] = (
|
||||
claude_host_access_token(_host_env)
|
||||
)
|
||||
|
||||
cred_token_ref = (
|
||||
CLAUDE_HOST_CREDENTIAL_TOKEN_REF if forward_host_credentials
|
||||
else auth_token
|
||||
)
|
||||
egress_routes = (EgressRoute(
|
||||
host="api.anthropic.com",
|
||||
auth_scheme="Bearer" if (auth_token or forward_host_credentials) else "",
|
||||
token_ref=cred_token_ref,
|
||||
auth_scheme="Bearer" if auth_token else "",
|
||||
token_ref=auth_token,
|
||||
),)
|
||||
hidden_env_names: frozenset[str] = frozenset()
|
||||
if auth_token or forward_host_credentials:
|
||||
if auth_token:
|
||||
env_vars["CLAUDE_CODE_OAUTH_TOKEN"] = "egress-placeholder"
|
||||
hidden_env_names = frozenset({"CLAUDE_CODE_OAUTH_TOKEN"})
|
||||
|
||||
@@ -219,7 +208,6 @@ class ClaudeAgentProvider(AgentProvider):
|
||||
files=tuple(files),
|
||||
egress_routes=egress_routes,
|
||||
hidden_env_names=hidden_env_names,
|
||||
provisioned_env=provisioned_env,
|
||||
)
|
||||
|
||||
def provision_skills(self, plan: "BottlePlan", bottle: "Bottle") -> None:
|
||||
|
||||
@@ -1,114 +0,0 @@
|
||||
"""Host Claude auth helpers.
|
||||
|
||||
Reads the host's Claude Code credentials and returns only the access
|
||||
token needed by egress. Does not expose refresh tokens or raw payloads.
|
||||
|
||||
Credential storage by platform:
|
||||
Linux — ~/.claude/.credentials.json
|
||||
macOS — macOS Keychain, service "Claude Code-credentials"
|
||||
(file path is tried first; Keychain is the fallback)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from ...log import die
|
||||
|
||||
|
||||
_KEYCHAIN_SERVICE = "Claude Code-credentials"
|
||||
|
||||
|
||||
def claude_auth_path(host_env: dict[str, str] | None = None) -> Path:
|
||||
env = os.environ if host_env is None else host_env
|
||||
home = env.get("HOME")
|
||||
if home:
|
||||
return Path(home) / ".claude" / ".credentials.json"
|
||||
return Path.home() / ".claude" / ".credentials.json"
|
||||
|
||||
|
||||
def _read_keychain() -> dict[str, object] | None:
|
||||
"""Try the macOS Keychain. Returns parsed JSON dict or None."""
|
||||
if sys.platform != "darwin":
|
||||
return None
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["security", "find-generic-password", "-s", _KEYCHAIN_SERVICE, "-w"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=10,
|
||||
)
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
if result.returncode != 0 or not result.stdout.strip():
|
||||
return None
|
||||
try:
|
||||
raw = json.loads(result.stdout.strip())
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
return raw if isinstance(raw, dict) else None
|
||||
|
||||
|
||||
def claude_host_access_token(
|
||||
host_env: dict[str, str] | None = None,
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
) -> str:
|
||||
path = claude_auth_path(host_env)
|
||||
raw: dict[str, object] | None = None
|
||||
|
||||
if path.is_file():
|
||||
try:
|
||||
raw = json.loads(path.read_text())
|
||||
except (OSError, json.JSONDecodeError) as e:
|
||||
die(f"claude host credentials: could not read valid JSON at {path}: {e}")
|
||||
if not isinstance(raw, dict):
|
||||
die(f"claude host credentials: {path} must contain a JSON object")
|
||||
else:
|
||||
raw = _read_keychain()
|
||||
if raw is None:
|
||||
die(
|
||||
f"claude host credentials: auth file missing at {path} and "
|
||||
f"macOS Keychain lookup for '{_KEYCHAIN_SERVICE}' failed. "
|
||||
"Run `claude login` on the host or disable "
|
||||
"agent_provider.forward_host_credentials."
|
||||
)
|
||||
|
||||
oauth = raw.get("claudeAiOauth")
|
||||
if not isinstance(oauth, dict):
|
||||
die(
|
||||
"claude host credentials: claudeAiOauth is missing from credentials. "
|
||||
"Run `claude login` on the host or disable "
|
||||
"agent_provider.forward_host_credentials."
|
||||
)
|
||||
|
||||
access_token = oauth.get("accessToken")
|
||||
if not isinstance(access_token, str) or not access_token:
|
||||
die(
|
||||
"claude host credentials: claudeAiOauth.accessToken is missing or empty. "
|
||||
"Run `claude login` on the host and restart the bottle."
|
||||
)
|
||||
|
||||
# expiresAt is in milliseconds
|
||||
expires_at = oauth.get("expiresAt")
|
||||
if isinstance(expires_at, (int, float)):
|
||||
check_now = now or datetime.now(timezone.utc)
|
||||
exp_dt = datetime.fromtimestamp(float(expires_at) / 1000.0, timezone.utc)
|
||||
if exp_dt <= check_now:
|
||||
die(
|
||||
"claude host credentials: host Claude access token is expired. "
|
||||
"Run `claude login` on the host and restart the bottle."
|
||||
)
|
||||
|
||||
return access_token
|
||||
|
||||
|
||||
__all__ = [
|
||||
"claude_auth_path",
|
||||
"claude_host_access_token",
|
||||
]
|
||||
+12
-2
@@ -3,6 +3,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlite3
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
@@ -28,12 +29,21 @@ class DbStore:
|
||||
conn.row_factory = sqlite3.Row
|
||||
return conn
|
||||
|
||||
@contextmanager
|
||||
def _connection(self):
|
||||
conn = self._connect()
|
||||
try:
|
||||
with conn:
|
||||
yield conn
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
def is_migrated(self) -> bool:
|
||||
"""Return True if the DB is fully up-to-date, False if migration is needed."""
|
||||
if not self.db_path.exists():
|
||||
return False
|
||||
try:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT version FROM schema_versions WHERE module = ?",
|
||||
(self._migrations.schema_key,),
|
||||
@@ -45,7 +55,7 @@ class DbStore:
|
||||
|
||||
def migrate(self) -> None:
|
||||
"""Apply any pending migrations and set permissions on the DB file."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
self._migrations.apply(conn)
|
||||
self._chmod()
|
||||
|
||||
|
||||
@@ -30,7 +30,6 @@ if TYPE_CHECKING:
|
||||
from .manifest import ManifestBottle
|
||||
|
||||
CODEX_HOST_CREDENTIAL_TOKEN_REF = "BOT_BOTTLE_CODEX_HOST_ACCESS_TOKEN"
|
||||
CLAUDE_HOST_CREDENTIAL_TOKEN_REF = "BOT_BOTTLE_CLAUDE_HOST_ACCESS_TOKEN"
|
||||
|
||||
EGRESS_HOSTNAME = "egress"
|
||||
|
||||
@@ -401,7 +400,6 @@ class Egress(ABC):
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"CLAUDE_HOST_CREDENTIAL_TOKEN_REF",
|
||||
"CODEX_HOST_CREDENTIAL_TOKEN_REF",
|
||||
"EGRESS_HOSTNAME",
|
||||
"EGRESS_ROUTES_FILENAME",
|
||||
|
||||
@@ -419,18 +419,24 @@ PY
|
||||
while IFS=' ' read -r old new ref; do
|
||||
[ -z "$ref" ] && continue
|
||||
[ "$new" = "$zero" ] && continue
|
||||
if [ "$old" = "$zero" ]; then
|
||||
# New ref: scan only the commits this push introduces — those
|
||||
# reachable from $new but not from any ref the gate already has.
|
||||
# Everything already on the gate arrived via upstream mirror-fetch
|
||||
# or a previously gitleaks-scanned push, so it's already-upstream
|
||||
# or already-scanned; re-scanning it (the old `$new` full-ancestry
|
||||
# range) only resurfaces historical findings and blocks every new
|
||||
# branch. See PRD 0028 / issue #106.
|
||||
log_opts="$new --not --all"
|
||||
else
|
||||
log_opts="$old..$new"
|
||||
fi
|
||||
# Scan only the commits this push introduces — those reachable from
|
||||
# $new but not from any ref the gate already has. Everything already
|
||||
# on the gate arrived via upstream mirror-fetch or a previously
|
||||
# gitleaks-scanned push, so it's already-upstream or already-scanned;
|
||||
# re-scanning it only resurfaces historical fixture findings.
|
||||
#
|
||||
# Applies to both new refs and updates. The old existing-branch range
|
||||
# `$old..$new` walks commits reachable from the new tip but not the
|
||||
# *old branch tip*: on a rebase/force-push onto a freshly-advanced
|
||||
# main that pulls in all of main's new history (incl. the deliberate
|
||||
# sandbox-escape gitleaks fixtures), blocking the push. `--not --all`
|
||||
# excludes anything already on the gate regardless of ancestry, so it
|
||||
# is also correct for non-fast-forward pushes (a rebase can skip
|
||||
# commits off the direct path). Security-equivalent per PRD 0028's
|
||||
# analysis: the bare repo's refs come only from trusted upstream
|
||||
# mirror-fetch or gitleaks-gated pushes.
|
||||
# See PRD 0028 (open question) / issues #106, #346.
|
||||
log_opts="$new --not --all"
|
||||
echo "git-gate: gitleaks scanning $ref ($log_opts)" >&2
|
||||
if ! gitleaks git --log-opts="$log_opts" --no-banner --redact 1>&2; then
|
||||
echo "git-gate: gitleaks rejected push to $ref" >&2
|
||||
|
||||
@@ -25,9 +25,8 @@ class ManifestAgentProvider:
|
||||
header, and sets a placeholder CLAUDE_CODE_OAUTH_TOKEN in the agent
|
||||
so the Claude Code CLI starts.
|
||||
|
||||
`forward_host_credentials` forwards the host provider auth token into
|
||||
the egress sidecar (Codex and Claude). For Codex this reads
|
||||
`~/.codex/auth.json`; for Claude it reads `~/.claude/.credentials.json`.
|
||||
`forward_host_credentials` forwards the host Codex auth token into
|
||||
the egress daemon (Codex only).
|
||||
"""
|
||||
|
||||
template: str = "claude"
|
||||
@@ -93,15 +92,10 @@ class ManifestAgentProvider:
|
||||
f"is only supported for built-in templates "
|
||||
f"({', '.join(sorted(PROVIDER_TEMPLATES))})"
|
||||
)
|
||||
if forward_host_credentials and template not in {"codex", "claude"}:
|
||||
if forward_host_credentials and template != "codex":
|
||||
raise ManifestError(
|
||||
f"bottle '{bottle_name}' agent_provider.forward_host_credentials "
|
||||
"is only supported for templates 'codex' and 'claude'"
|
||||
)
|
||||
if forward_host_credentials and auth_token:
|
||||
raise ManifestError(
|
||||
f"bottle '{bottle_name}' agent_provider.forward_host_credentials "
|
||||
"and auth_token both set; use one or the other"
|
||||
"is currently only supported for template 'codex'"
|
||||
)
|
||||
settings = _parse_provider_settings(bottle_name, template, d.get("settings"))
|
||||
return cls(
|
||||
|
||||
@@ -167,7 +167,7 @@ class RegistryStore(DbStore):
|
||||
metadata=metadata,
|
||||
policy=policy,
|
||||
)
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"DELETE FROM orchestrator_bottles "
|
||||
"WHERE source_ip = ? AND state = 'active' AND bottle_id != ?",
|
||||
@@ -193,7 +193,7 @@ class RegistryStore(DbStore):
|
||||
def set_policy(self, bottle_id: str, policy: str) -> bool:
|
||||
"""Update a bottle's policy in place (live reload). Returns True if
|
||||
the bottle exists."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
cur = conn.execute(
|
||||
"UPDATE orchestrator_bottles SET policy = ? WHERE bottle_id = ?",
|
||||
(policy, bottle_id),
|
||||
@@ -203,7 +203,7 @@ class RegistryStore(DbStore):
|
||||
|
||||
def deregister(self, bottle_id: str) -> bool:
|
||||
"""Remove a bottle. Returns True if a row was deleted."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
cur = conn.execute(
|
||||
"DELETE FROM orchestrator_bottles WHERE bottle_id = ?", (bottle_id,)
|
||||
)
|
||||
@@ -211,7 +211,7 @@ class RegistryStore(DbStore):
|
||||
|
||||
def get(self, bottle_id: str) -> BottleRecord | None:
|
||||
"""Return the bottle by id, or None if absent."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT * FROM orchestrator_bottles WHERE bottle_id = ?", (bottle_id,)
|
||||
).fetchone()
|
||||
@@ -219,7 +219,7 @@ class RegistryStore(DbStore):
|
||||
|
||||
def all(self) -> list[BottleRecord]:
|
||||
"""Every registered bottle, oldest first."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT * FROM orchestrator_bottles ORDER BY created_at"
|
||||
).fetchall()
|
||||
@@ -232,7 +232,7 @@ class RegistryStore(DbStore):
|
||||
source IP is unspoofable (Firecracker `/31` + nft) and the control
|
||||
plane is reachable only by the trusted gateway; pair with the
|
||||
identity token (`attribute`) elsewhere."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT * FROM orchestrator_bottles "
|
||||
"WHERE source_ip = ? AND state = 'active'",
|
||||
|
||||
@@ -66,7 +66,7 @@ class QueueStore(DbStore):
|
||||
super().__init__(resolved, migrations)
|
||||
|
||||
def write_proposal(self, proposal: Proposal) -> Path:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT OR REPLACE INTO supervise_proposals (
|
||||
@@ -89,7 +89,7 @@ class QueueStore(DbStore):
|
||||
return self.db_path
|
||||
|
||||
def read_proposal(self, proposal_id: str) -> Proposal:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
row = conn.execute(
|
||||
"""
|
||||
SELECT * FROM supervise_proposals
|
||||
@@ -104,7 +104,7 @@ class QueueStore(DbStore):
|
||||
def list_pending_proposals(self) -> list[Proposal]:
|
||||
if not self.db_path.is_file():
|
||||
return []
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT p.* FROM supervise_proposals p
|
||||
@@ -125,7 +125,7 @@ class QueueStore(DbStore):
|
||||
def list_all_pending_proposals(self) -> list[Proposal]:
|
||||
if not self.db_path.is_file():
|
||||
return []
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT p.* FROM supervise_proposals p
|
||||
@@ -142,7 +142,7 @@ class QueueStore(DbStore):
|
||||
return [self._row_to_proposal(row) for row in rows]
|
||||
|
||||
def write_response(self, response: Response) -> Path:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT OR REPLACE INTO supervise_responses (
|
||||
@@ -161,7 +161,7 @@ class QueueStore(DbStore):
|
||||
return self.db_path
|
||||
|
||||
def read_response(self, proposal_id: str) -> Response:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
row = conn.execute(
|
||||
"""
|
||||
SELECT * FROM supervise_responses
|
||||
@@ -176,7 +176,7 @@ class QueueStore(DbStore):
|
||||
def archive_proposal(self, proposal_id: str) -> None:
|
||||
if not self.db_path.is_file():
|
||||
return
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE supervise_proposals SET archived = 1
|
||||
|
||||
@@ -47,6 +47,7 @@ from .supervise_types import (
|
||||
STATUS_MODIFIED,
|
||||
STATUS_REJECTED,
|
||||
TOOLS,
|
||||
TOOL_CHECK_PROPOSAL,
|
||||
TOOL_EGRESS_ALLOW,
|
||||
TOOL_EGRESS_BLOCK,
|
||||
TOOL_EGRESS_TOKEN_ALLOW,
|
||||
@@ -263,6 +264,7 @@ __all__ = [
|
||||
"TOOLS",
|
||||
"EGRESS_FORWARD_PROXY",
|
||||
"EGRESS_INTROSPECT_URL",
|
||||
"TOOL_CHECK_PROPOSAL",
|
||||
"TOOL_EGRESS_ALLOW",
|
||||
"TOOL_EGRESS_BLOCK",
|
||||
"TOOL_GITLEAKS_ALLOW",
|
||||
|
||||
@@ -2,14 +2,24 @@
|
||||
|
||||
Per-bottle MCP server exposing tools the agent calls to propose egress
|
||||
config changes when stuck. The tools are `egress-allow`,
|
||||
`egress-block`, and `list-egress-routes`.
|
||||
`egress-block`, `list-egress-routes`, and `check-proposal`.
|
||||
|
||||
Each queued tool call:
|
||||
Each queued proposal tool call:
|
||||
|
||||
1. Validates the proposed file syntactically.
|
||||
2. Writes a Proposal to the host SQLite database.
|
||||
3. Blocks polling for a matching Response row.
|
||||
4. Returns the operator's `{status, notes}` to the agent.
|
||||
3. Blocks polling for a matching Response row, up to a short grace
|
||||
window (`SUPERVISE_RESPONSE_TIMEOUT_SECONDS`, default 30s).
|
||||
4. On a decision within the window, returns the operator's
|
||||
`{status, notes}`. On timeout, returns `status: pending` **with the
|
||||
proposal id** and leaves the proposal queued — the flow is
|
||||
non-blocking past the grace window (PRD prd-new / issue #412).
|
||||
|
||||
`check-proposal` is the non-blocking companion: given a `proposal_id`
|
||||
returned by a `pending` response, it reports the current decision
|
||||
(`pending` | `approved` | `modified` | `rejected`) without re-proposing,
|
||||
so an approval made out-of-band (e.g. a web review console) can be resumed
|
||||
without holding an HTTP request open.
|
||||
|
||||
One shared server fronts every bottle (PRD 0070) and attributes each
|
||||
proposal to the calling bottle by source IP, resolved from the orchestrator
|
||||
@@ -22,7 +32,9 @@ Speaks MCP over HTTP+JSON-RPC. Methods handled:
|
||||
* `initialize` — handshake; returns server info + caps.
|
||||
* `notifications/initialized` — ack-only.
|
||||
* `tools/list` — returns the tool definitions.
|
||||
* `tools/call` — validates, queues, blocks, returns.
|
||||
* `tools/call` — validates, queues, waits out the grace
|
||||
window, returns (pending past it); or, for
|
||||
`check-proposal`, a non-blocking status poll.
|
||||
|
||||
Everything else returns JSON-RPC error -32601 (method not found).
|
||||
|
||||
@@ -232,6 +244,31 @@ TOOL_DEFINITIONS: list[dict[str, object]] = [
|
||||
),
|
||||
"inputSchema": _proposal_input_schema(),
|
||||
},
|
||||
{
|
||||
"name": _sv.TOOL_CHECK_PROPOSAL,
|
||||
"description": (
|
||||
"Poll a previously queued proposal for the operator's decision "
|
||||
"WITHOUT blocking or re-proposing. Pass the `proposal_id` you "
|
||||
"got back when an `egress-allow`/`egress-block` call returned "
|
||||
"`status: pending`. Returns the current status: `pending` (no "
|
||||
"decision yet — poll again later), `approved`, `modified`, "
|
||||
"`rejected`, or `unknown` (no such queued proposal — wrong id, "
|
||||
"or it was already resolved and read)."
|
||||
),
|
||||
"inputSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"proposal_id": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"The proposal id from a `pending` response."
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["proposal_id"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@@ -353,7 +390,7 @@ def handle_tools_call(
|
||||
deadline=deadline,
|
||||
)
|
||||
except TimeoutError:
|
||||
text = format_pending_response_text(config.response_timeout_seconds)
|
||||
text = format_pending_response_text(proposal.id, config.response_timeout_seconds)
|
||||
return {
|
||||
"content": [{"type": "text", "text": text}],
|
||||
"isError": False,
|
||||
@@ -370,6 +407,54 @@ def handle_tools_call(
|
||||
}
|
||||
|
||||
|
||||
def handle_check_proposal(
|
||||
params: dict[str, object],
|
||||
config: ServerConfig,
|
||||
) -> dict[str, object]:
|
||||
"""Non-blocking poll of a queued proposal's decision, by id.
|
||||
|
||||
Never creates a Proposal (so `check-proposal` isn't in `TOOLS`); it only
|
||||
reads the queue. Resolution order mirrors the synchronous path's terminal
|
||||
step — a decided proposal is archived here exactly as `handle_tools_call`
|
||||
archives it after `wait_for_response`, so `pending` proposals stay visible
|
||||
to the operator until they're both decided *and* polled."""
|
||||
args_raw = params.get("arguments", {})
|
||||
if not isinstance(args_raw, dict):
|
||||
raise _RpcClientError(ERR_INVALID_PARAMS, "tools/call 'arguments' must be an object")
|
||||
proposal_id = args_raw.get("proposal_id")
|
||||
if not isinstance(proposal_id, str) or not proposal_id.strip():
|
||||
raise _RpcClientError(
|
||||
ERR_INVALID_PARAMS,
|
||||
"check-proposal: 'proposal_id' is required and must be a non-empty string",
|
||||
)
|
||||
proposal_id = proposal_id.strip()
|
||||
|
||||
try:
|
||||
response = _sv.read_response(config.bottle_slug, proposal_id)
|
||||
except FileNotFoundError:
|
||||
# No decision yet — distinguish "still queued" from "unknown id".
|
||||
try:
|
||||
_sv.read_proposal(config.bottle_slug, proposal_id)
|
||||
except FileNotFoundError:
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_unknown_proposal_text(proposal_id)}],
|
||||
"isError": True,
|
||||
}
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_still_pending_text(proposal_id)}],
|
||||
"isError": False,
|
||||
}
|
||||
|
||||
try:
|
||||
_sv.archive_proposal(config.bottle_slug, proposal_id)
|
||||
except OSError as e:
|
||||
raise _RpcInternalError(f"failed to archive proposal: {e}") from e
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_response_text(response)}],
|
||||
"isError": response.status == _sv.STATUS_REJECTED,
|
||||
}
|
||||
|
||||
|
||||
def format_response_text(response: "_sv.Response") -> str:
|
||||
"""Pretty-print a Response for the tool's text content. The agent
|
||||
reads the text and decides whether to retry / give up / surface."""
|
||||
@@ -382,12 +467,35 @@ def format_response_text(response: "_sv.Response") -> str:
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def format_pending_response_text(timeout_seconds: float) -> str:
|
||||
def format_pending_response_text(proposal_id: str, timeout_seconds: float) -> str:
|
||||
"""Grace-window timeout: the proposal stays queued, and the agent is
|
||||
told the id so it can `check-proposal` instead of re-proposing."""
|
||||
return "\n".join([
|
||||
"status: pending",
|
||||
f"proposal_id: {proposal_id}",
|
||||
(
|
||||
"notes: operator response timed out after "
|
||||
f"{timeout_seconds:g}s; proposal remains queued"
|
||||
f"notes: no operator decision within {timeout_seconds:g}s; the "
|
||||
"proposal remains queued. Poll it (do not re-propose) by calling "
|
||||
f"`check-proposal` with proposal_id={proposal_id!r}."
|
||||
),
|
||||
])
|
||||
|
||||
|
||||
def format_still_pending_text(proposal_id: str) -> str:
|
||||
return "\n".join([
|
||||
"status: pending",
|
||||
f"proposal_id: {proposal_id}",
|
||||
"notes: still queued; no operator decision yet. Call `check-proposal` again later.",
|
||||
])
|
||||
|
||||
|
||||
def format_unknown_proposal_text(proposal_id: str) -> str:
|
||||
return "\n".join([
|
||||
"status: unknown",
|
||||
f"proposal_id: {proposal_id}",
|
||||
(
|
||||
"notes: no queued proposal with this id for this bottle — the id "
|
||||
"may be wrong, or the proposal was already resolved and read."
|
||||
),
|
||||
])
|
||||
|
||||
@@ -482,6 +590,11 @@ class MCPHandler(http.server.BaseHTTPRequestHandler):
|
||||
# — silently dropping base routes like api.anthropic.com on approval.
|
||||
if req.params.get("name") == _sv.TOOL_LIST_EGRESS_ROUTES:
|
||||
return self._resolved_routes_payload()
|
||||
# `check-proposal` is a non-blocking read of the calling bottle's
|
||||
# own queue — attributed by source IP like a proposal, but it
|
||||
# never queues or blocks.
|
||||
if req.params.get("name") == _sv.TOOL_CHECK_PROPOSAL:
|
||||
return handle_check_proposal(req.params, self._attributed_config(config))
|
||||
# Attribute the proposal to the source-IP-resolved bottle, so the one
|
||||
# shared server queues each bottle's proposal under its own slug.
|
||||
return handle_tools_call(req.params, self._attributed_config(config))
|
||||
|
||||
@@ -20,6 +20,10 @@ TOOL_EGRESS_ALLOW = "egress-allow"
|
||||
TOOL_GITLEAKS_ALLOW = "gitleaks-allow"
|
||||
TOOL_EGRESS_TOKEN_ALLOW = "egress-token-allow"
|
||||
TOOL_LIST_EGRESS_ROUTES = "list-egress-routes"
|
||||
# Read-only agent tool: poll a queued proposal for the operator's decision
|
||||
# without blocking or re-proposing. It never becomes a `Proposal.tool` (no
|
||||
# queue record is created for it), so it is intentionally NOT in `TOOLS`.
|
||||
TOOL_CHECK_PROPOSAL = "check-proposal"
|
||||
TOOLS: tuple[str, ...] = (
|
||||
TOOL_EGRESS_ALLOW,
|
||||
TOOL_EGRESS_BLOCK,
|
||||
@@ -156,6 +160,7 @@ __all__ = [
|
||||
"TOOLS",
|
||||
"TOOL_EGRESS_ALLOW",
|
||||
"TOOL_EGRESS_BLOCK",
|
||||
"TOOL_CHECK_PROPOSAL",
|
||||
"TOOL_EGRESS_TOKEN_ALLOW",
|
||||
"TOOL_GITLEAKS_ALLOW",
|
||||
"TOOL_LIST_EGRESS_ROUTES",
|
||||
|
||||
@@ -7,6 +7,7 @@ from __future__ import annotations
|
||||
|
||||
import ipaddress
|
||||
import os
|
||||
import sys
|
||||
|
||||
|
||||
def is_ip_literal(value: str) -> bool:
|
||||
@@ -17,6 +18,15 @@ def is_ip_literal(value: str) -> bool:
|
||||
return True
|
||||
|
||||
|
||||
def read_tty_line() -> str:
|
||||
"""Mirror `IFS= read -r REPLY </dev/tty`. Falls back to stdin."""
|
||||
try:
|
||||
with open("/dev/tty", "r", encoding="utf-8") as tty:
|
||||
return tty.readline().rstrip("\n")
|
||||
except OSError:
|
||||
return sys.stdin.readline().rstrip("\n")
|
||||
|
||||
|
||||
def expand_tilde(path: str) -> str:
|
||||
"""Expand a leading '~' to $HOME. Leaves paths without a leading
|
||||
tilde unchanged. Falls back to the empty string if $HOME is unset
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
# ADR 0005: Keep tracker metadata on issues
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-07-18
|
||||
- **Deciders:** didericis
|
||||
|
||||
## Context
|
||||
|
||||
Gitea exposes labels on both issues and pull requests. Applying the same labels
|
||||
to both copies planning metadata, creates a synchronization obligation, and
|
||||
makes disagreements between the two records possible. At the same time,
|
||||
unlabelled objects look accidental unless the repository states which object
|
||||
owns the metadata.
|
||||
|
||||
The repository already uses issues as work items and PRs as implementations of
|
||||
those work items. At this decision's cutoff, all open PRs reference issues, but
|
||||
121 of 219 historically merged PRs do not. Manufacturing retrospective issues
|
||||
for that history would create records that never participated in planning and
|
||||
would make the issue history less truthful.
|
||||
|
||||
## Decision
|
||||
|
||||
Issues are the canonical tracker records and own labels. Every issue has at
|
||||
least one label. An issue opened or left without labels receives
|
||||
`Status/Needs Triage` automatically until it is classified.
|
||||
|
||||
Pull requests carry no labels. Every new PR deliberately references at least
|
||||
one existing issue in its title or description with one of these forms:
|
||||
|
||||
- `Closes #123`, `Fixes #123`, or `Resolves #123` when merging completes it.
|
||||
- `Part of #123`, `Related to #123`, `Refs #123`, or `References #123` when it
|
||||
contributes without completing it.
|
||||
|
||||
Gitea Actions enforces both PR rules as a status check and repairs the empty
|
||||
issue-label state. Branch protection makes the PR policy check required.
|
||||
|
||||
The policy applies from 2026-07-18 onward. Existing issues may be labelled as
|
||||
they are encountered, but closed PRs are grandfathered: no retrospective
|
||||
issues or PR labels are created solely to make history conform.
|
||||
|
||||
## Consequences
|
||||
|
||||
- Classification, priority, and workflow metadata have one source of truth.
|
||||
- A PR's issue link is the navigation path to its planning metadata.
|
||||
- Multi-PR issues do not require copied or synchronized labels.
|
||||
- `Status/Needs Triage` is an intentional fallback, not a final
|
||||
classification.
|
||||
- Direct issue creation remains convenient; automation repairs a missing label
|
||||
immediately after creation because Gitea has no native required-label rule.
|
||||
- The required check must be configured in branch protection after this
|
||||
workflow lands.
|
||||
|
||||
## Links
|
||||
|
||||
- Issue #405.
|
||||
- `.gitea/workflows/tracker-policy.yml`.
|
||||
- `scripts/tracker_policy.py`.
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0023: smolmachines bottle backend
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0032: Decompose smolmachines launch and harden bringup sequencing
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis-claude
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0038: smolmachines Env Contract and Secret-Safe Injection
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis-codex
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0039: smolmachines Capability-Block Remediation
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis-codex
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0042: smolmachines Cross-Backend Parity Tests
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis-codex
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0057: Promote smolmachines to default backend; convert Docker to example-only
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0068: smolmachines backend on Linux
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** Claude
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
# PRD prd-new: CI artifact-based coverage and local Firecracker candidate flow
|
||||
|
||||
- **Status:** Active
|
||||
- **Author:** Claude
|
||||
- **Created:** 2026-07-21
|
||||
- **Issue:** #446
|
||||
|
||||
## Summary
|
||||
|
||||
Restructure the CI test pipeline to run each test suite exactly once, upload
|
||||
small `.coverage.*` artifacts, and combine them in a lightweight aggregation
|
||||
job. Move the infra build onto the KVM runner so the ~194 MB rootfs never
|
||||
crosses the network for PRs. On main-branch pushes, publish the byte-identical
|
||||
rootfs that was tested.
|
||||
|
||||
## Motivation
|
||||
|
||||
The prior pipeline had two redundant costs:
|
||||
|
||||
1. **Duplicate artifact transfers.** `build-infra` (ubuntu-latest) built and
|
||||
uploaded the ~194 MB rootfs; `integration-firecracker` downloaded it; the
|
||||
`coverage` job downloaded it a second time. Combined download overhead: ~83
|
||||
seconds per run, plus the ~70-second upload.
|
||||
|
||||
2. **Duplicate test execution.** `integration-firecracker` ran the Firecracker
|
||||
integration suite; `coverage` ran the entire unit + integration suite again
|
||||
on the same KVM runner to collect coverage data. Every line of Firecracker
|
||||
code was tested twice per CI run.
|
||||
|
||||
## Goals
|
||||
|
||||
- Each test suite (unit, integration-docker, integration-firecracker) executes
|
||||
exactly once per workflow run.
|
||||
- PRs incur no large artifact transfers — the rootfs stays on the KVM runner.
|
||||
- Main-branch pushes publish a byte-for-byte identical rootfs to the one that
|
||||
passed the integration tests.
|
||||
- Concurrent workflow runs cannot cross-publish candidates (naturally enforced
|
||||
by Gitea Actions' per-run artifact scoping).
|
||||
- Failed or cancelled runs block publication (enforced by the `needs:` chain on
|
||||
`publish-infra`).
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Changing test semantics or the coverage policy (ADR 0004).
|
||||
- Removing the KVM runner guard on `integration-firecracker` and `coverage`.
|
||||
- Changing how `publish_infra.py` builds or uploads the rootfs.
|
||||
|
||||
## Design
|
||||
|
||||
### Job graph
|
||||
|
||||
```
|
||||
unit ──────────────────────────────────┐
|
||||
integration-docker ────────────────────┤──► coverage ──► publish-infra (main only)
|
||||
integration-firecracker (KVM) ─────────┘
|
||||
```
|
||||
|
||||
### `unit`
|
||||
|
||||
Unchanged except: `coverage run` writes `--data-file=.coverage.unit`; the file
|
||||
is uploaded as the `coverage-unit` artifact.
|
||||
|
||||
### `integration-docker`
|
||||
|
||||
Adds a `coverage` install step. `coverage run` writes `--data-file=.coverage.docker`;
|
||||
the file is uploaded as `coverage-docker`.
|
||||
|
||||
### `integration-firecracker` (KVM runner)
|
||||
|
||||
Replaces the old `stage-firecracker-inputs` → `build-infra` → download chain:
|
||||
|
||||
1. Builds the infra candidate locally with
|
||||
`BOT_BOTTLE_FC_DROPBEAR=/var/cache/bot-bottle-fc/dropbear`.
|
||||
2. Boots the candidate and runs integration tests with coverage, writing
|
||||
`.coverage.firecracker`.
|
||||
3. Uploads the small `coverage-firecracker` artifact unconditionally.
|
||||
4. On main-branch pushes only, uploads the rootfs as `infra-candidate` and the
|
||||
dropbear as `firecracker-inputs` so `publish-infra` can verify and publish
|
||||
the byte-identical artifact.
|
||||
|
||||
### `coverage`
|
||||
|
||||
Moves from a KVM runner to `ubuntu-latest`. No tests are re-executed:
|
||||
|
||||
1. Downloads `coverage-unit`, `coverage-docker`, and `coverage-firecracker`.
|
||||
2. Runs `scripts/coverage.sh aggregate critical`, which calls
|
||||
`coverage combine` then `coverage report`.
|
||||
3. Runs the diff-coverage gate (`scripts/diff_coverage.py`).
|
||||
|
||||
Coverage files use `relative_files = True` (`.coveragerc`) so they combine
|
||||
cleanly across runners with different absolute workspace paths.
|
||||
|
||||
### `publish-infra`
|
||||
|
||||
Depends on all four predecessor jobs (unchanged gate). Downloads `infra-candidate`
|
||||
and `firecracker-inputs` that were uploaded by `integration-firecracker` on
|
||||
main — the same byte sequence that passed the integration tests.
|
||||
|
||||
### Eliminated jobs
|
||||
|
||||
- `stage-firecracker-inputs`: existed only to copy the dropbear to ubuntu-latest
|
||||
for `build-infra`. No longer needed.
|
||||
- `build-infra`: the infra candidate is now built on the KVM runner in
|
||||
`integration-firecracker`.
|
||||
|
||||
### Script changes
|
||||
|
||||
`scripts/coverage.sh` gains an `aggregate` mode (`coverage.sh aggregate [critical]`)
|
||||
that combines pre-existing `.coverage.*` files instead of re-running tests.
|
||||
The existing run mode (`coverage.sh [critical]`) is preserved for local dev.
|
||||
@@ -1,146 +0,0 @@
|
||||
# PRD prd-new: Claude forward_host_credentials
|
||||
|
||||
- **Status:** Draft
|
||||
- **Author:** claude
|
||||
- **Created:** 2026-07-01
|
||||
- **Issue:** #325
|
||||
|
||||
## Summary
|
||||
|
||||
Add `agent_provider.forward_host_credentials: true` support for the
|
||||
`claude` template, mirroring the existing Codex flow. When enabled,
|
||||
bot-bottle reads the host's Claude OAuth session key from
|
||||
`~/.claude/.credentials.json` at launch, forwards it only to the egress sidecar,
|
||||
and injects a placeholder `CLAUDE_CODE_OAUTH_TOKEN` into the agent so
|
||||
Claude Code starts without ever seeing the real credential.
|
||||
|
||||
## Problem
|
||||
|
||||
Running a Claude agent in a container today requires the operator to
|
||||
manually extract a long-lived OAuth token (`claude setup-token`), export
|
||||
it as `BOT_BOTTLE_CLAUDE_OAUTH_TOKEN`, and reference it explicitly in
|
||||
the manifest with `agent_provider.auth_token:
|
||||
"BOT_BOTTLE_CLAUDE_OAUTH_TOKEN"`. This is a two-step manual ceremony
|
||||
that is easy to skip or do incorrectly.
|
||||
|
||||
The host already stores a valid Claude session in `~/.claude/.credentials.json`
|
||||
after `claude login`. Codex already automates an
|
||||
equivalent extraction from `~/.codex/auth.json`. There is no reason
|
||||
Claude bottles cannot do the same.
|
||||
|
||||
## Goals / Success Criteria
|
||||
|
||||
- A Claude bottle with `forward_host_credentials: true` in the manifest
|
||||
uses the host's `~/.claude/.credentials.json` session key at launch with no
|
||||
additional operator steps.
|
||||
- The agent container receives only `CLAUDE_CODE_OAUTH_TOKEN=egress-placeholder`
|
||||
— never the real token.
|
||||
- The real session key lives only in the egress sidecar's environment.
|
||||
- Missing, malformed, or expired host Claude auth fails launch with a
|
||||
clear operator-facing message.
|
||||
- Existing `auth_token` behavior is unchanged.
|
||||
- `forward_host_credentials: true` is rejected in the manifest when both
|
||||
`auth_token` and `forward_host_credentials` are set, since they serve
|
||||
the same purpose.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Refreshing Claude OAuth tokens in the sidecar.
|
||||
- Writing a dummy `~/.claude.json` auth state to the agent (unlike the
|
||||
Codex flow, Claude Code reads its credential from `CLAUDE_CODE_OAUTH_TOKEN`
|
||||
in env, not from an auth file — no guest-side auth marker is needed).
|
||||
- Supporting `forward_host_credentials` for providers other than `codex`
|
||||
and `claude`.
|
||||
|
||||
## Design
|
||||
|
||||
### Manifest schema
|
||||
|
||||
```yaml
|
||||
agent_provider:
|
||||
template: claude
|
||||
forward_host_credentials: true
|
||||
```
|
||||
|
||||
Rejects in manifest validation when:
|
||||
- Template is not `codex` or `claude`.
|
||||
- Both `auth_token` and `forward_host_credentials` are set.
|
||||
|
||||
### Host auth extraction (`contrib/claude/claude_auth.py`)
|
||||
|
||||
Claude Code credential storage varies by platform:
|
||||
|
||||
- **Linux**: `~/.claude/.credentials.json`
|
||||
- **macOS**: macOS Keychain, service `"Claude Code-credentials"`
|
||||
(the file path is tried first; Keychain is the fallback when the file
|
||||
is absent)
|
||||
|
||||
`~/.claude.json` contains only UI state and profile metadata — no token.
|
||||
|
||||
The credentials JSON schema (same whether from file or Keychain):
|
||||
|
||||
```json
|
||||
{
|
||||
"claudeAiOauth": {
|
||||
"accessToken": "<access-token>",
|
||||
"refreshToken": "<refresh-token>",
|
||||
"expiresAt": 1748276587173,
|
||||
"scopes": ["user:inference", "user:profile"]
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`expiresAt` is in **milliseconds** (not seconds).
|
||||
|
||||
At prepare/launch time, when `forward_host_credentials: true`:
|
||||
|
||||
1. Try `~/.claude/.credentials.json`; on macOS, if absent, run
|
||||
`security find-generic-password -s "Claude Code-credentials" -w`
|
||||
and parse its stdout as JSON.
|
||||
2. Require a `claudeAiOauth` dict.
|
||||
3. Require a non-empty `claudeAiOauth.accessToken` string.
|
||||
4. If `claudeAiOauth.expiresAt` is present, divide by 1000 and require
|
||||
the result to be in the future.
|
||||
5. Return only the access token to the launch path.
|
||||
|
||||
Errors name the missing or invalid condition and point the operator at
|
||||
`claude login`, without printing token values.
|
||||
|
||||
### Egress route
|
||||
|
||||
When `forward_host_credentials: true`:
|
||||
|
||||
- Provision the session key in `provisioned_env` under
|
||||
`BOT_BOTTLE_CLAUDE_HOST_ACCESS_TOKEN` (new constant in `egress.py`).
|
||||
- Set up the `api.anthropic.com` egress route with `auth_scheme: Bearer`
|
||||
and `token_ref: BOT_BOTTLE_CLAUDE_HOST_ACCESS_TOKEN`.
|
||||
- Set `CLAUDE_CODE_OAUTH_TOKEN=egress-placeholder` in the agent env and
|
||||
add it to `hidden_env_names`.
|
||||
|
||||
No dummy auth file and no `verify` step are needed — Claude Code reads
|
||||
the credential from the env var, not from a file.
|
||||
|
||||
### Constants
|
||||
|
||||
- `CLAUDE_HOST_CREDENTIAL_TOKEN_REF = "BOT_BOTTLE_CLAUDE_HOST_ACCESS_TOKEN"`
|
||||
in `egress.py` (alongside the existing `CODEX_HOST_CREDENTIAL_TOKEN_REF`).
|
||||
- `CLAUDE_HOST_CREDENTIAL_HOSTS = ("api.anthropic.com",)` in
|
||||
`agent_provider.py` (alongside the existing `CODEX_HOST_CREDENTIAL_HOSTS`).
|
||||
|
||||
### Data flow
|
||||
|
||||
```
|
||||
Host ~/.claude/.credentials.json → bot-bottle launch
|
||||
│
|
||||
├──► egress sidecar env (real token only)
|
||||
│
|
||||
└──► agent env: CLAUDE_CODE_OAUTH_TOKEN=egress-placeholder
|
||||
|
||||
Agent → HTTPS to api.anthropic.com (via egress)
|
||||
Egress → injects Authorization: Bearer <real token>
|
||||
Egress → forwards to api.anthropic.com
|
||||
```
|
||||
|
||||
## Open questions
|
||||
|
||||
None — the Codex precedent makes the design clear.
|
||||
@@ -0,0 +1,125 @@
|
||||
# PRD prd-new: Non-blocking supervise (async approval + proposal polling)
|
||||
|
||||
- **Status:** Draft
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-07-18
|
||||
- **Issue:** #412
|
||||
|
||||
## Summary
|
||||
|
||||
The per-bottle supervise MCP server (`bot_bottle/supervise_server.py`)
|
||||
answers `tools/call` **synchronously**: it queues the agent's proposal and
|
||||
blocks the tool call polling for the operator's decision. On timeout it
|
||||
returns `status: pending` and leaves the proposal queued — but it hands the
|
||||
agent **no proposal id** and offers **no way to poll a specific pending
|
||||
proposal**, so the only way to learn the outcome is to re-propose (a
|
||||
duplicate).
|
||||
|
||||
This PRD makes the MCP flow non-blocking and pollable, so an approval can
|
||||
happen out-of-band (a human taking minutes-to-hours in a review console)
|
||||
without holding an HTTP request open or wedging the agent:
|
||||
|
||||
1. Include the `proposal_id` in the `pending` response.
|
||||
2. Add a `check-proposal` MCP tool: a non-blocking status lookup by
|
||||
proposal id.
|
||||
3. Keep the short synchronous grace window for the common "operator is
|
||||
right there" fast path.
|
||||
|
||||
## Problem
|
||||
|
||||
`handle_tools_call` → `_sv.wait_for_response(...)` blocks up to
|
||||
`SUPERVISE_RESPONSE_TIMEOUT_SECONDS` (default 30s). Two problems follow:
|
||||
|
||||
- **Human latency ≠ tool-call latency.** A real review — rendered diff,
|
||||
RBAC routing to an approver, someone tapping approve on their phone — is
|
||||
minutes-to-hours. Holding the MCP request open that long is fragile
|
||||
(proxy/keepalive timeouts, the mitmproxy egress hop, and the agent
|
||||
harness's own tool-call timeout, which a long block can trip and stall
|
||||
the whole turn).
|
||||
- **No resume path.** The pending fallback already exists, but without a
|
||||
proposal id and a poll tool the agent can't reconnect to that specific
|
||||
decision — it re-proposes, duplicating the queue entry.
|
||||
|
||||
This is also the precondition for the planned web-console human-review
|
||||
flow (RBAC, audit retention, mobile) — see issue #412.
|
||||
|
||||
**Safety note:** the MCP tools only *propose* policy changes; enforcement
|
||||
stays at the egress proxy and the git-gate. Returning early on `pending`
|
||||
therefore opens no hole — the agent still cannot egress or push anything
|
||||
unapproved.
|
||||
|
||||
## Goals / success criteria
|
||||
|
||||
- A `pending` MCP response carries the `proposal_id`.
|
||||
- An agent can call `check-proposal(proposal_id)` and get the current
|
||||
state (`pending` | `approved` | `modified` | `rejected`) **without
|
||||
blocking** and **without creating a new proposal**.
|
||||
- The synchronous fast path (operator approves within the grace window) is
|
||||
unchanged: the first `tools/call` still returns the decision directly.
|
||||
- No change to enforcement, attribution (source-IP → bottle), or the
|
||||
operator-side queue/response schema.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- The git-gate `pre-receive` path (it is synchronous by nature and cannot
|
||||
poll — its async variant is reject-fast + re-push; tracked as a
|
||||
follow-up).
|
||||
- Backpressure / in-flight-proposal caps.
|
||||
- MCP server→client notifications (event-driven resume).
|
||||
- Any web-console UI (this PRD is the protocol groundwork it needs).
|
||||
|
||||
## Design
|
||||
|
||||
### `pending` response carries the id
|
||||
|
||||
`handle_tools_call`'s timeout branch formats the pending text with the
|
||||
`proposal.id` and a pointer to `check-proposal`, so the agent knows what to
|
||||
poll.
|
||||
|
||||
### `check-proposal` tool
|
||||
|
||||
A new read-only MCP tool (`TOOL_CHECK_PROPOSAL = "check-proposal"`),
|
||||
attributed to the calling bottle by source IP exactly like the proposal
|
||||
tools. Input: `{ "proposal_id": string }`. Behavior:
|
||||
|
||||
1. `read_response(slug, id)` →
|
||||
- **found**: archive the proposal (same terminal step the synchronous
|
||||
path takes) and return the decision via `format_response_text`;
|
||||
`isError` iff rejected.
|
||||
2. **not found** → `read_proposal(slug, id)` →
|
||||
- **found**: still queued → return `status: pending`.
|
||||
- **not found**: unknown id, or already resolved-and-archived (e.g. a
|
||||
second poll) → return `status: unknown`, `isError: true`.
|
||||
|
||||
Both lookups already raise `FileNotFoundError` when absent
|
||||
(`queue_store.py`), so the handler needs no new store methods. `check-`
|
||||
`proposal` is the only path (besides the synchronous response) that
|
||||
archives, so a proposal that times out to `pending` stays visible to the
|
||||
operator until it is decided and then polled.
|
||||
|
||||
### Grace window
|
||||
|
||||
Left at the existing 30s default (`SUPERVISE_RESPONSE_TIMEOUT_SECONDS`),
|
||||
which doubles as the instant-approve fast path. Tuning it down is an
|
||||
operator setting, not a code change; noted for the console rollout.
|
||||
|
||||
## Implementation chunks
|
||||
|
||||
1. **(this PR)** `TOOL_CHECK_PROPOSAL` constant; `check-proposal` tool
|
||||
definition + `handle_check_proposal`; dispatch wiring; `proposal_id` in
|
||||
the pending text; unit tests. Files: `bot_bottle/supervise_types.py`,
|
||||
`bot_bottle/supervise.py` (re-export), `bot_bottle/supervise_server.py`,
|
||||
`tests/unit/test_supervise_server.py`.
|
||||
2. **(follow-up)** git-gate `pre-receive` reject-fast + re-push.
|
||||
3. **(follow-up)** per-bottle in-flight-proposal backpressure cap.
|
||||
4. **(follow-up)** MCP notifications for event-driven resume; web-console
|
||||
review flow (RBAC, audit retention) on top.
|
||||
|
||||
## Open questions
|
||||
|
||||
- Should a resolved-but-unpolled proposal auto-archive after some TTL, or
|
||||
only on poll? (Leaning: only on poll, so a decision is never lost to a
|
||||
reaper before the agent sees it.)
|
||||
- Does the agent harness need an explicit "you have a pending proposal"
|
||||
nudge, or is returning `pending` from the original call enough? (Deferred
|
||||
to the notifications chunk.)
|
||||
@@ -1,32 +1,81 @@
|
||||
# Landscape: AI-agent sandbox tools
|
||||
|
||||
A broader survey than [`landscape-containerized-claude.md`](landscape-containerized-claude.md),
|
||||
which focused on Claude-Code-specific containerizers. This one covers
|
||||
general AI-agent sandbox / containment projects — some Claude-specific,
|
||||
some agent-agnostic, some hosted SaaS — and contrasts them with
|
||||
bot-bottle's design.
|
||||
Survey of AI-agent sandbox and containment projects — including local
|
||||
coding-agent wrappers, agent-agnostic runtimes, hosted platforms, and
|
||||
governance layers — contrasted with bot-bottle's design. The original
|
||||
Claude-Code-specific containerizer survey was folded into this note on
|
||||
2026-07-20 so there is one landscape and one positioning verdict.
|
||||
|
||||
Research conducted 2026-05-11.
|
||||
Research conducted 2026-05-11. CubeSandbox added 2026-07-18 (see its
|
||||
per-project note and the addendum at the end). Also updated 2026-07-18:
|
||||
bot-bottle no longer uses **pipelock** — outbound DLP is now bot-bottle's
|
||||
own (deliberately simple) egress scanner (a mitmproxy addon with custom
|
||||
detectors, PRD 0017 / 0052), and git-push secret scanning is handled by
|
||||
**gitleaks** in the git-gate. "pipelock" below has been replaced with the
|
||||
current mechanism; it survives only in older PRDs as history.
|
||||
|
||||
Updated again 2026-07-18: six additional tools added (Cleanroom,
|
||||
container-use, Docker sbx, Anthropic srt, Microsoft AGT, Open Agent
|
||||
Passport); an **Agent-tailored policy** row added to the comparison table;
|
||||
a separate Governance layers section added for AGT and OAP. See the
|
||||
second addendum at the end.
|
||||
|
||||
Updated 2026-07-20: the borrowable-ideas status was reconciled with the
|
||||
current implementation. In-flight credential injection and the microVM
|
||||
backends have shipped, while per-use SSH confirmation was superseded by
|
||||
keeping git credentials out of the agent entirely.
|
||||
|
||||
Also updated 2026-07-20: **E2B and Daytona added as first-class entries.**
|
||||
Earlier revisions mentioned E2B only as the API and lifecycle model that
|
||||
CubeSandbox implements, and omitted Daytona entirely. That was a survey gap,
|
||||
not a principled scope exclusion: both are major hosted sandbox platforms and
|
||||
belong in this landscape even though they target platform builders rather than
|
||||
bot-bottle's local single-operator workflow.
|
||||
|
||||
## Summary
|
||||
|
||||
Eight projects surveyed. None duplicate bot-bottle's combination of
|
||||
local Docker, declarative JSON manifest, per-agent egress allowlist via
|
||||
pipelock, and bottle/agent split. Two clusters stand out:
|
||||
The main table compares bot-bottle against fifteen isolation/sandbox tools.
|
||||
Governance/pre-action authorization and credential-only layers are covered
|
||||
separately because they don't provide VM or container isolation. None
|
||||
duplicate bot-bottle's combination of local
|
||||
VM-per-bottle isolation, a declarative per-role manifest, per-agent
|
||||
egress allowlist + outbound-content DLP, bottle/agent split, and the
|
||||
composable `extends:` policy model. Three clusters stand out:
|
||||
|
||||
- **Closest neighbours** — agent-safehouse and litterbox: local,
|
||||
single-user, thin wrappers over an existing OS primitive
|
||||
(`sandbox-exec`, Podman + Landlock).
|
||||
- **Different category** — tilde.run (hosted SaaS), boxlite and
|
||||
microsandbox (microVM libraries for platform builders), endo-familiar
|
||||
- **Different category (isolation)** — tilde.run (hosted SaaS), boxlite
|
||||
and microsandbox (microVM libraries for platform builders), E2B and Daytona
|
||||
(hosted sandbox platforms), CubeSandbox (self-hosted multi-tenant microVM
|
||||
service), endo-familiar
|
||||
(capability-security paradigm, no OS isolation).
|
||||
- **New: governance/pre-action layers** — Microsoft AGT and Open Agent
|
||||
Passport (OAP): framework-embedded tool-call interceptors with
|
||||
per-agent declarative policy. Closest competitors on agent-tailored
|
||||
policy, but operate at the tool-call level rather than providing
|
||||
network/filesystem isolation; they complement rather than substitute.
|
||||
|
||||
The microVM cluster (matchlock, smolmachines, boxlite, microsandbox) is
|
||||
the most relevant for the v2 isolation discussion in
|
||||
The microVM cluster (matchlock, smolmachines, boxlite, microsandbox,
|
||||
CubeSandbox) is the most relevant for the v2 isolation discussion in
|
||||
[`stronger-isolation-alternatives.md`](stronger-isolation-alternatives.md):
|
||||
libkrun and Apple's Virtualization.framework have made local microVMs
|
||||
ergonomic enough that a `"runtime": "microvm"` option on a bottle is now
|
||||
plausible without a heavy stack.
|
||||
ergonomic enough that microVMs are **now bot-bottle's default backend**
|
||||
(Firecracker on KVM Linux, Apple Container on macOS), with Docker kept
|
||||
only as a legacy fallback for CI / hosts without KVM or Apple Container.
|
||||
That discussion has since shipped, not just been theorized.
|
||||
|
||||
**The one that matters most for positioning is CubeSandbox** — it ships
|
||||
bot-bottle's bundle of default-deny egress allowlisting, full audit logs, and
|
||||
in-flight credential custody *combined with* per-sandbox microVM isolation,
|
||||
open-source under Apache 2.0, with Tencent Cloud behind it and 10.4k
|
||||
stars. It's a self-hosted multi-tenant service for platform builders, not
|
||||
a single-user declarative tool, so it doesn't collide head-on — but it
|
||||
narrows the "nobody else bundles egress custody + credential injection"
|
||||
claim that the monetization positioning leans on. Daytona now also offers
|
||||
domain/CIDR firewall policy plus in-flight header credential substitution and
|
||||
response scrubbing, although its higher tiers are not default-deny and its
|
||||
production platform is proprietary. See the addendum.
|
||||
|
||||
## Per-project notes
|
||||
|
||||
@@ -63,7 +112,8 @@ plausible without a heavy stack.
|
||||
|
||||
### agent-safehouse
|
||||
- **Source**: https://agent-safehouse.dev/ ; https://github.com/eugene1g/agent-safehouse
|
||||
- **License**: Apache 2.0 (~1,400 stars)
|
||||
- **HN launch**: [#47301085](https://news.ycombinator.com/item?id=47301085) (March 12 2026) — 823 points
|
||||
- **License**: Apache 2.0 (~1,781 stars at launch)
|
||||
- **Isolation**: macOS `sandbox-exec` (Seatbelt) profiles — kernel-level
|
||||
syscall interception, no container.
|
||||
- **Locality**: Local, macOS only.
|
||||
@@ -73,6 +123,16 @@ plausible without a heavy stack.
|
||||
- **Config**: Shell functions or custom `sandbox-exec` profile files;
|
||||
LLM-assisted profile generation supported.
|
||||
- **Network policy**: Not addressed.
|
||||
- **Notable from HN thread**: Creator acknowledged the project is "just a
|
||||
policy-generator for `sandbox-exec` — no dependencies, no daemons, no
|
||||
subscription; I did put in many hours to identify the minimum required
|
||||
permissions for agents to continue working." Simon Willison noted that
|
||||
evaluating whether a sandboxing tool actually works as intended is hard.
|
||||
Top community sentiment: *"I honestly think that sandboxing is currently
|
||||
THE major challenge that needs to be solved for the tech to fully realise
|
||||
its potential."* The macOS Docker gap (Docker for Mac runs inside a Linux
|
||||
VM, so `sandbox-exec` is the only native primitive for bare-metal macOS
|
||||
processes) was the stated motivation.
|
||||
- **Maturity**: Active through March 2026.
|
||||
|
||||
### matchlock
|
||||
@@ -155,73 +215,474 @@ plausible without a heavy stack.
|
||||
also supported.
|
||||
- **Maturity**: Active through April 2026.
|
||||
|
||||
### E2B *(added 2026-07-20)*
|
||||
|
||||
- **Source**: https://github.com/e2b-dev/e2b ; https://e2b.dev/docs
|
||||
- **License**: Apache 2.0 (~12.4k stars); commercial hosted service with
|
||||
self-hosting/BYOC support.
|
||||
- **Isolation**: Firecracker microVM per sandbox.
|
||||
- **Locality**: Cloud-hosted by default; self-hosting uses Terraform on AWS or
|
||||
GCP (with other targets documented as works in progress).
|
||||
- **Agent integration**: LLM-agnostic Python and JavaScript/TypeScript SDKs;
|
||||
code-interpreter and desktop-sandbox products. Platform primitive rather
|
||||
than a coding-agent wrapper.
|
||||
- **Config**: Programmatic SDK/API plus templates. Network configuration
|
||||
supports internet on/off, outbound allow/deny rules, and a custom egress
|
||||
proxy.
|
||||
- **Network policy**: Configurable per sandbox, but not documented as
|
||||
default-deny and no built-in outbound-content DLP is documented.
|
||||
- **Credentials**: Environment variables passed to the sandbox are explicitly
|
||||
not private at the OS level. No built-in in-flight application-credential
|
||||
injection is documented.
|
||||
- **Persistence**: Full memory + filesystem pause/resume, snapshots, and
|
||||
auto-resume. Continuous runtime is tier-limited, while paused sandboxes are
|
||||
retained indefinitely.
|
||||
- **Maturity**: Established hosted platform and the API compatibility target
|
||||
used by CubeSandbox.
|
||||
|
||||
### Daytona *(added 2026-07-20)*
|
||||
|
||||
- **Source**: https://github.com/daytonaio/daytona ;
|
||||
https://www.daytona.io/docs/
|
||||
- **License**: Current production platform is proprietary. The former AGPL
|
||||
repository remains public but is no longer maintained after Daytona moved
|
||||
production development closed-source in June 2026.
|
||||
- **Isolation**: Hosted container sandboxes by default, with separate Linux
|
||||
and Windows VM sandbox classes for dedicated-OS workloads. Each sandbox has
|
||||
its own filesystem and network stack; VM-only features include memory
|
||||
pause/resume and forking.
|
||||
- **Locality**: Hosted multi-tenant service, with dedicated/custom regions and
|
||||
customer runners available.
|
||||
- **Agent integration**: LLM/framework-agnostic SDKs (Python, TypeScript, Go,
|
||||
Ruby, Java), API, and CLI; official agent-framework guides. Platform
|
||||
primitive rather than a local coding-agent wrapper.
|
||||
- **Config**: Programmatic per-sandbox image/snapshot, resources, lifecycle,
|
||||
firewall, and secrets.
|
||||
- **Network policy**: Per-sandbox IPv4/domain allowlists and block-all mode,
|
||||
subordinate to organization/tier policy. Full internet access is the
|
||||
default on higher tiers, so it is configurable rather than uniformly
|
||||
default-deny.
|
||||
- **Credentials**: First-class secret manager with the same phantom-token
|
||||
pattern as bot-bottle: the sandbox environment gets an opaque placeholder,
|
||||
an HTTPS proxy substitutes the real secret in headers only for allowed
|
||||
hosts, and responses are scrubbed back to the placeholder.
|
||||
- **Persistence**: Persistent filesystem for stopped container sandboxes;
|
||||
memory + filesystem pause/resume for VM sandboxes; snapshots and configurable
|
||||
auto-stop.
|
||||
- **Maturity**: Production commercial platform. Notable April 2026 credential
|
||||
exposure was patched; the June 2026 closed-source transition materially
|
||||
changes its transparency/self-hosting posture.
|
||||
|
||||
### Other hosted runtimes carried forward from the earlier survey
|
||||
|
||||
- **Northflank Sandboxes** — hosted or customer-cloud, microVM-backed
|
||||
containers with SDK-managed lifecycle, optional persistent volumes, and
|
||||
sub-second claimed boot. This is a platform primitive for untrusted code and
|
||||
agents, not a local agent wrapper or role-policy layer.
|
||||
- **Cloudflare Sandbox SDK** — Workers/Durable Objects API over VM-isolated
|
||||
Linux containers for command, file, process, and service execution. It is a
|
||||
hosted TypeScript platform primitive; application authentication,
|
||||
authorization, and credential-proxy patterns remain the integrator's job.
|
||||
|
||||
Both belong to the same “build your agent platform on this runtime” category as
|
||||
E2B and Daytona. They were named but not analyzed in depth by the original
|
||||
Claude-specific note, so they remain outside the main comparison table rather
|
||||
than being presented with false precision.
|
||||
|
||||
### CubeSandbox *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/TencentCloud/CubeSandbox ;
|
||||
HN launch https://news.ycombinator.com/item?id=47863430
|
||||
- **License**: Apache 2.0 (~10.4k stars). By Tencent Cloud; described as
|
||||
"battle-tested, production-ready" infra already running in Tencent
|
||||
Cloud. Rust / Go / C.
|
||||
- **Isolation**: MicroVMs via RustVMM + KVM — "each sandbox gets its own
|
||||
Guest OS kernel, no Docker shared-kernel escapes." Hardware-level
|
||||
isolation, dedicated kernel per instance.
|
||||
- **Locality**: Self-hosted, but **server/cluster-oriented**, not a
|
||||
single-user local CLI. Deploy guides target PVM cloud VMs, bare metal,
|
||||
and dev. A single 96-vCPU host is claimed to run 2,000+ concurrent
|
||||
sandboxes.
|
||||
- **Agent integration**: **Drop-in E2B SDK replacement** (single env-var
|
||||
change) — the headline compatibility claim. OpenClaw assistant
|
||||
integration; general LLM-code execution. Aimed at platform builders,
|
||||
not one developer's laptop.
|
||||
- **Config**: Programmatic via the E2B-compatible SDK. No declarative
|
||||
manifest.
|
||||
- **Network policy**: This is the striking part — **domain allowlists,
|
||||
instant block on unauthorized egress, full audit logs, per-sandbox
|
||||
traffic tokens, policy-routing egress**, enforced by an eBPF-based
|
||||
virtual switch giving kernel-level network isolation. Closest match yet
|
||||
to bot-bottle's own default-deny + per-bottle allowlist egress model.
|
||||
- **Credentials**: **Credential vault** — agents call external APIs / LLMs
|
||||
while "keys never enter the sandbox, model context, or logs." Same
|
||||
in-flight-injection idea as matchlock, but productized as a vault.
|
||||
- **Performance**: <60ms cold start (claimed 2.5–50× faster than
|
||||
alternatives), <5MB memory per instance; millisecond snapshot rollback
|
||||
is upcoming.
|
||||
- **Maturity**: Open-sourced July 2026 off production Tencent Cloud use;
|
||||
most-starred project in this set (~10.4k).
|
||||
|
||||
### Cleanroom *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/buildkite/cleanroom
|
||||
- **License**: Apache 2.0
|
||||
- **Isolation**: MicroVM — Firecracker on Linux, Virtualization.framework
|
||||
on macOS. Digest-pinned OCI images.
|
||||
- **Locality**: Self-hosted server (CI-oriented).
|
||||
- **Agent integration**: Generic process sandbox; CI-first, not a
|
||||
Claude/agent wrapper.
|
||||
- **Config**: `cleanroom.yaml` in the repo being sandboxed defines egress
|
||||
rules, resources, and network policy. Cleanroom resolves this from the
|
||||
commit being run.
|
||||
- **Network policy**: Default-deny + per-repo hostname allowlist (resolved
|
||||
from DNS answers + destination IP:port). Co-hosted services on the same
|
||||
IP:port are not distinguished. OIDC-backed auth for remote servers.
|
||||
- **Credentials**: Host-side only; not injected in-flight but not present
|
||||
in the VM.
|
||||
- **Notable**: Policy lives in the *repo being sandboxed*, not in an
|
||||
agent-role definition — closer to per-repo scoping than per-role.
|
||||
Supports Docker-inside-sandbox (`services.docker.required: true`), OIDC
|
||||
authorization, suspend/resume lifecycle.
|
||||
- **Maturity**: Active Buildkite product.
|
||||
|
||||
### container-use *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/dagger/container-use
|
||||
- **License**: Apache 2.0
|
||||
- **Isolation**: Docker container per agent + git worktree per agent.
|
||||
Containers share the host kernel; stronger than bare host but weaker
|
||||
than microVM.
|
||||
- **Locality**: Local.
|
||||
- **Agent integration**: MCP stdio server — Claude Code, Cursor, Windsurf.
|
||||
`claude mcp add container-use -- container-use stdio`.
|
||||
- **Config**: None for security policy. Environments are provisioned on
|
||||
demand; no allowlist or credential config.
|
||||
- **Network policy**: Not addressed.
|
||||
- **Notable**: Per-agent git branches (`container-use/<env_name>`);
|
||||
parallel agents without filesystem conflict; real-time log visibility
|
||||
and terminal attach for intervention; git-based review workflow.
|
||||
Oriented toward parallel development safety, not security containment.
|
||||
- **Maturity**: Early development, active.
|
||||
|
||||
### Docker sbx *(added 2026-07-18)*
|
||||
- **Source**: Docker proprietary (`sbx` CLI, separate from `docker`).
|
||||
- **License**: Proprietary.
|
||||
- **Isolation**: MicroVM (Docker's own implementation) — each session gets
|
||||
its own kernel, Docker daemon inside the VM, and filesystem.
|
||||
- **Locality**: Local (macOS and Windows; does not require Docker Desktop).
|
||||
- **Agent integration**: Explicit wrapper — Claude Code, Codex, Gemini
|
||||
CLI, Copilot CLI, Kiro. Launches agent inside the VM with
|
||||
`--dangerously-skip-permissions` by default.
|
||||
- **Config**: Open / Balanced / Locked Down network presets at launch. No
|
||||
per-role manifest.
|
||||
- **Network policy**: Default-deny; preset levels control strictness. TUI
|
||||
dashboard shows a live log of every outbound connection (allowed and
|
||||
blocked) with point-and-click allow/block for hosts.
|
||||
- **Credentials**: OS keychain + host-side proxy injection — API keys
|
||||
never enter the VM.
|
||||
- **Notable**: Best DX among microVM tools (one command, works like native
|
||||
yolo Claude but inside a VM); branch mode creates a git worktree in
|
||||
`.sbx/`. Network policy is preset-based, not role-declarative.
|
||||
- **Maturity**: GA 2026.
|
||||
|
||||
### Anthropic srt *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/anthropic-experimental/sandbox-runtime
|
||||
(`@anthropic-ai/sandbox-runtime` on npm, `sandbox-runtime` on PyPI)
|
||||
- **License**: Apache 2.0 (experimental).
|
||||
- **Isolation**: OS-level only — Seatbelt (`sandbox-exec`) on macOS,
|
||||
bubblewrap on Linux, WFP (Windows Filtering Platform) account-fenced on
|
||||
Windows. **No container or VM.** Lowest overhead in the set.
|
||||
- **Locality**: Local.
|
||||
- **Agent integration**: Claude Code's sandboxed bash tool uses this
|
||||
internally. Can wrap any arbitrary process (`srt <command>`). Cloud
|
||||
Claude Code sessions use full microVMs instead.
|
||||
- **Config**: Programmatic per-invocation — allow/deny path lists for
|
||||
filesystem; allow/denylist for network (HTTP proxy + SOCKS5).
|
||||
- **Network policy**: Proxy-based filtering (HTTP + SOCKS5); domain
|
||||
allowlist/denylist enforced at proxy layer. Custom proxy supported
|
||||
(e.g. mitmproxy for inspection + audit). Processes that ignore proxy
|
||||
env vars may bypass filtering on some platforms.
|
||||
- **Notable**: Cross-platform (macOS/Linux/Windows); wraps any process,
|
||||
not just agents; no role/manifest concept. Annotated as a research
|
||||
preview — APIs may change.
|
||||
- **Maturity**: Early research preview.
|
||||
|
||||
## Claude-specific wrappers and developer environments
|
||||
|
||||
These projects were the focus of the original containerized-Claude survey.
|
||||
They remain useful comparisons for local developer experience, but most are
|
||||
templates or wrappers rather than policy-bearing sandbox platforms, so they
|
||||
are grouped here instead of widening the main table further.
|
||||
|
||||
### claudebox
|
||||
|
||||
- **Source**: https://github.com/RchGrav/claudebox
|
||||
- **Isolation**: Docker, with per-project images, authentication state, and
|
||||
configuration.
|
||||
- **Agent integration**: Claude Code wrapper with 15+ preconfigured language
|
||||
and task profiles.
|
||||
- **Network policy**: Per-project firewall allowlists.
|
||||
- **Closest overlap**: local one-command developer workflow and project-scoped
|
||||
network policy.
|
||||
- **Difference**: profiles describe development toolchains, not named agent
|
||||
roles. There is no bottle/agent split, composable role manifest, provider
|
||||
plugin layer, or outbound-content DLP.
|
||||
|
||||
### Spritz / claude-code-sandbox
|
||||
|
||||
- **Source**: https://github.com/textcortex/claude-code-sandbox (archived;
|
||||
points to its successor, Spritz).
|
||||
- **Isolation**: The original project ran Claude Code in local Docker with
|
||||
bypass permissions; Spritz moved toward Kubernetes-native multi-agent
|
||||
infrastructure.
|
||||
- **Difference**: the successor targets cluster orchestration rather than a
|
||||
low-dependency local launcher. It is architecturally closer to hosted or
|
||||
Kubernetes platform runtimes than to bot-bottle's single-operator CLI.
|
||||
|
||||
### Trail of Bits claude-code-devcontainer
|
||||
|
||||
- **Source**: https://github.com/trailofbits/claude-code-devcontainer
|
||||
- **Isolation**: A Docker devcontainer that exposes only project files and is
|
||||
designed to run Claude Code with `bypassPermissions` for security audits and
|
||||
untrusted-code review.
|
||||
- **Difference**: a hardened, reusable environment definition rather than an
|
||||
agent launcher or fleet. It has no named-role manifest, per-role credential
|
||||
custody, supervision plane, or multi-backend abstraction.
|
||||
|
||||
### Smaller wrappers and official templates
|
||||
|
||||
Projects such as `arezi/claude-sandbox`, `nkrefman/claude-sandbox`, and
|
||||
`VishalJ99/claude-docker`, plus Docker/Anthropic devcontainer templates, prove
|
||||
there is steady demand for “Claude in a container.” They are deliberately
|
||||
small launch/build configurations. They compete on setup simplicity, not on
|
||||
role-aware policy, credential custody, persistent supervision, or a fleet
|
||||
model, and are better treated as a product category than as individual rows.
|
||||
|
||||
### SuperHQ
|
||||
|
||||
- **Source**: https://superhq.ai/
|
||||
- **Isolation**: Apple-Silicon desktop application using local microVMs via
|
||||
Virtualization.framework/libkrun-era components.
|
||||
- **Agent integration**: Claude Code, Codex, and Pi in a GUI, with mobile
|
||||
remote access.
|
||||
- **Credentials and review**: host-side auth gateway injects credentials on
|
||||
the wire; a temporary overlay stages writes for diff-and-accept review.
|
||||
- **Closest overlap**: local microVM isolation, multi-provider launching, and
|
||||
credential custody for security-minded individual developers.
|
||||
- **Difference**: GUI desktop product on Apple Silicon rather than a
|
||||
cross-platform declarative CLI/fleet layer. The July 2026 snapshot in the
|
||||
original survey recorded a user request for per-run tool-call and network
|
||||
audit logging; treat that as point-in-time rather than a permanent gap.
|
||||
|
||||
## Credential gateway without isolation
|
||||
|
||||
### OneCLI
|
||||
|
||||
[OneCLI](https://onecli.sh/) is a framework-agnostic identity gateway rather
|
||||
than a sandbox. Its phantom-token design gives the agent a placeholder and
|
||||
substitutes the encrypted real credential at the network layer. It therefore
|
||||
matches bot-bottle closely on secret custody, and is more portable because it
|
||||
can sit in front of agents launched by anything, but it supplies no container
|
||||
or VM boundary, filesystem isolation, role manifest, or egress-content DLP.
|
||||
|
||||
The positioning consequence from the earlier survey still holds: secret
|
||||
custody alone is not unique. bot-bottle's relevant combination is local
|
||||
isolation + default-deny egress + payload DLP + declarative roles + credential
|
||||
custody. OneCLI's managed tier also places custody with a third party, whereas
|
||||
bot-bottle keeps it within operator-controlled infrastructure. See
|
||||
[`agent-credential-proxy-landscape.md`](agent-credential-proxy-landscape.md)
|
||||
for the detailed build-versus-adopt analysis.
|
||||
|
||||
## Governance / pre-action authorization layers
|
||||
|
||||
These two tools don't provide VM or filesystem isolation; they intercept
|
||||
tool calls before execution and evaluate them against a per-agent
|
||||
declarative policy. They are the closest competitors on **agent-tailored
|
||||
policy** and complement isolation sandboxes rather than substituting for
|
||||
them.
|
||||
|
||||
### Microsoft Agent Governance Toolkit (AGT) *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/microsoft/agent-governance-toolkit
|
||||
- **License**: MIT (~3.3k stars, open-sourced April 2, 2026).
|
||||
- **Isolation**: None (OS/VM). Execution rings (0–3, inspired by CPU
|
||||
privilege levels) control what an agent can do at the framework layer.
|
||||
MCP security gateway treats MCP traffic as an untrusted boundary.
|
||||
- **Locality**: Embedded in the agent framework (Python, TypeScript, .NET,
|
||||
Rust, Go; 20+ framework adapters).
|
||||
- **Agent integration**: Framework-agnostic. Plugs into Semantic Kernel,
|
||||
AutoGen, and others as a middleware layer.
|
||||
- **Config**: YAML policy per agent — tools can be `allowed`, `denied`,
|
||||
`sandboxed`, or routed through an `approval` step. Every action passes
|
||||
through a governance gate checking: agent DID, trust score, risk tier,
|
||||
requested tool, action type, and policy rules.
|
||||
- **Network policy**: Not directly — operates at tool-call level.
|
||||
- **Credentials**: Per-agent DID (Ed25519 decentralized identifier); agent
|
||||
does not borrow a human's credentials.
|
||||
- **Notable**: Dynamic trust score (0–1,000, behavioral decay) —
|
||||
privilege follows observed behaviour, not just provisioning. Covers all
|
||||
10 OWASP Agentic Top 10 risks. Kill switch + SLO monitoring. Sub-ms
|
||||
policy enforcement.
|
||||
- **Maturity**: MIT, ~3.3k ⭐, v3.7.0 May 2026.
|
||||
|
||||
### Open Agent Passport (OAP) *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/aporthq/aport-spec ; spec at
|
||||
https://api.aport.io/spec/spec/oap/oap-spec.md/ ; arXiv 2603.20953
|
||||
- **License**: Open specification.
|
||||
- **Isolation**: None. Pre-action hook only — intercepts tool calls
|
||||
synchronously before execution, evaluates against a cloud-registry
|
||||
declarative policy, fails closed.
|
||||
- **Locality**: Local hook + cloud policy registry.
|
||||
- **Agent integration**: Framework-agnostic; hook pattern.
|
||||
- **Config**: Declarative policy rules in a cloud registry (evaluated in
|
||||
order; first failing rule denies). Ed25519-signed, hash-chained audit
|
||||
records per decision.
|
||||
- **Network policy**: Not directly.
|
||||
- **Notable**: 53ms median authorization decision (N=1,000). In an
|
||||
adversarial testbed ($5,000 bounty, 1,151 sessions), social engineering
|
||||
succeeded 74.6% of the time under a permissive policy; under a
|
||||
restrictive OAP policy, 0% success across 879 attempts. Assumes
|
||||
framework runtime is not compromised.
|
||||
- **Maturity**: Specification + reference implementation, 2026.
|
||||
|
||||
## Comparison table
|
||||
|
||||
| Axis | bot-bottle | endo-familiar | litterbox | agent-safehouse | matchlock | tilde.run | boxlite | microsandbox | smolmachines |
|
||||
|---|---|---|---|---|---|---|---|---|---|
|
||||
| Isolation | Docker + internal net + pipelock; gVisor if present | Object-capability (no OS isolation) | Podman + opt. Landlock | macOS `sandbox-exec` | MicroVM (Firecracker / Virt.fw) | Hosted container (unverified) | MicroVM (KVM / Hypervisor.fw) | MicroVM (libkrun) | MicroVM (libkrun / KVM) |
|
||||
| Local vs hosted | Local | Local | Local (Linux) | Local (macOS) | Local | Hosted SaaS | Local | Local | Local |
|
||||
| Open source | Apache 2.0 | Apache 2.0 | Apache 2.0 | Apache 2.0 | MIT | No | Apache 2.0 | Apache 2.0 | Apache 2.0 |
|
||||
| Agent target | Claude Code | Generic (demo) | Generic | Multi-agent wrapper | Generic (+ Claude/OpenAI SDKs) | Claude focus | Generic | Claude + Cursor (MCP/Skills) | Generic (AGENTS.md) |
|
||||
| Network policy | Default-deny via pipelock + per-bottle allowlist + DLP | Capability model only | Limited | Not addressed | Default-deny + allowlist + secret-injecting proxy | Default-deny + logging | Per-VM net (unverified) | Not documented | Off by default + allowlist |
|
||||
| Parallel agents | Yes (one bottle per agent) | n/a | Not addressed | One at a time | Multiple VMs | Yes (dashboard) | SDK-level | SDK-level | Architectural |
|
||||
| Config | JSON manifest (bottles + agents) | Programmatic refs | CLI wizard | Profile files / shell fns | CLI / SDK | DSL + CLI + SDK | SDK | CLI / SDK / MCP | TOML Smolfile |
|
||||
| Maturity | Active May 2026 | Research (2022+) | Early (~66 ⭐) | Active (~1.4k ⭐) | Experimental (~574 ⭐) | Private preview | YC, ~4.7k ⭐ | YC, ~6k ⭐, beta | ~3.1k ⭐ |
|
||||
*Isolation/sandbox tools only. AGT and OAP are governance layers — see their per-project notes above.*
|
||||
|
||||
| Axis | bot-bottle | endo-familiar | litterbox | agent-safehouse | matchlock | tilde.run | boxlite | microsandbox | smolmachines | E2B | Daytona | CubeSandbox | Cleanroom | container-use | Docker sbx | Anthropic srt |
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
| Isolation | MicroVM per bottle default (Firecracker/KVM on Linux, Apple Container on macOS) + own egress DLP scanner; Docker legacy fallback, gVisor there if present | Object-capability (no OS isolation) | Podman + opt. Landlock | macOS `sandbox-exec` | MicroVM (Firecracker / Virt.fw) | Hosted container (unverified) | MicroVM (KVM / Hypervisor.fw) | MicroVM (libkrun) | MicroVM (libkrun / KVM) | Firecracker microVM | Container or Linux/Windows VM class | MicroVM (RustVMM / KVM) | MicroVM (Firecracker / Virt.fw) | Docker container + git worktree | MicroVM (proprietary) | OS-level (Seatbelt / bubblewrap / WFP) — no container |
|
||||
| Local vs hosted | Local | Local | Local (Linux) | Local (macOS) | Local | Hosted SaaS | Local | Local | Local | Hosted; self-host/BYOC available | Hosted; dedicated/custom regions | Self-hosted (server/cluster) | Self-hosted server | Local | Local | Local |
|
||||
| Open source | Apache 2.0 | Apache 2.0 | Apache 2.0 | Apache 2.0 | MIT | No | Apache 2.0 | Apache 2.0 | Apache 2.0 | Apache 2.0 | Production closed-source; legacy AGPL repo unmaintained | Apache 2.0 | Apache 2.0 | Apache 2.0 | Proprietary | Apache 2.0 (experimental) |
|
||||
| Agent target | Claude Code, Codex, Pi, and provider plugins | Generic (demo) | Generic | Multi-agent wrapper | Generic (+ Claude/OpenAI SDKs) | Claude focus | Generic | Claude + Cursor (MCP/Skills) | Generic (AGENTS.md) | LLM-agnostic platform builders | LLM-agnostic platform builders | E2B-compatible (platform builders) | CI / generic process | Claude Code, Cursor, Windsurf (MCP) | Claude Code, Codex, Gemini CLI, Copilot, Kiro | Claude Code (and any process) |
|
||||
| Network policy | Default-deny via own egress scanner + per-bottle allowlist + content DLP + gitleaks on git push | Capability model only | Limited | Not addressed | Default-deny + allowlist + secret-injecting proxy | Default-deny + logging | Per-VM net (unverified) | Not documented | Off by default + allowlist | Per-sandbox allow/deny rules and custom egress proxy; internet configurable | Per-sandbox CIDR/domain allowlist or block-all; tier policy; secret-injecting proxy | Default-deny allowlist + instant egress block + audit logs + per-sandbox tokens (eBPF) + credential vault | Default-deny + per-repo host allowlist (cleanroom.yaml) | Not addressed | Default-deny; Open / Balanced / Locked Down presets; live TUI network panel | Proxy-based allowlist/denylist (HTTP + SOCKS5); custom proxy supported |
|
||||
| Parallel agents | Yes (one bottle per agent) | n/a | Not addressed | One at a time | Multiple VMs | Yes (dashboard) | SDK-level | SDK-level | Architectural | Yes (platform service) | Yes (platform service) | Yes (2,000+/host claimed) | Yes (server model) | Yes (per-agent containers + worktrees) | Yes | Yes |
|
||||
| Long-running posture | Persistent by default (named, supervised) | n/a (demo) | Session (up while in use) | Per-invocation | Ephemeral VM per run | Per-run (versioned) | Ephemeral + snapshot/fork | Ephemeral / on-demand | Named persistent by default | Runtime tier limits + indefinite pause/resume | Persistent filesystem; VM pause/resume; configurable auto-stop | Ephemeral + auto pause/resume | Per-run + suspend/resume | Per-agent container (ephemeral) | Per-session; branch mode creates git worktree in .sbx/ | Per-invocation |
|
||||
| DX: run Claude yolo-style | One command → interactive yolo Claude (`start <agent>`, `--dangerously-skip-permissions` default) | n/a (lib demo) | Wizard + build, then run claude inside (Linux only) | One-command wrapper (`safehouse claude --dangerously-skip-permissions`) | CLI: run a cmd in a VM (not a Claude wrapper) | Hosted (`tilde exec`), not local-native | SDK code required (build the run yourself) | CLI/MCP: sandbox-as-a-tool for the agent, not a wrapper around it | SSH into a named machine, run claude there | SDK/CLI sandbox; wire the agent yourself | SDK/CLI sandbox; wire the agent yourself | Stand up a cluster + drive via E2B SDK | CI-oriented, not a Claude wrapper | MCP server: `claude mcp add container-use -- container-use stdio` | One command: `sbx` wraps claude with `--dangerously-skip-permissions` default | Library/wrapper, not a standalone CLI |
|
||||
| Config | YAML-in-Markdown manifests (bottles + agents) | Programmatic refs | CLI wizard | Profile files / shell fns | CLI / SDK | DSL + CLI + SDK | SDK | CLI / SDK / MCP | TOML Smolfile | SDK/API + templates | SDK/API/CLI + images/snapshots | E2B-compatible SDK | cleanroom.yaml in repo | None (no policy config) | Preset levels at launch | Programmatic per-invocation (allow/deny lists) |
|
||||
| Agent-tailored policy | Yes — bottle/agent split; declarative per-role egress + credentials; composable via `extends:` | Partial — capability model scopes per-agent, but no declarative role manifest | No | Partial — per-agent profile files (Seatbelt); no egress | No | Yes — per-agent DSL RBAC (allow/deny/approve per action/repo/agent) | No | No | No | No — per-sandbox SDK config | No — per-sandbox SDK config | No — per-sandbox SDK config, not role-scoped | Partial — per-repo cleanroom.yaml, not per-role | No | No — network presets only | No |
|
||||
| Maturity | Active July 2026 | Research (2022+) | Early (~66 ⭐) | Active (~1.8k ⭐) | Experimental (~574 ⭐) | Private preview | YC, ~4.7k ⭐ | YC, ~6k ⭐, beta | ~3.1k ⭐ | Established hosted platform, ~12.4k ⭐ | Production commercial; closed-source since June 2026 | Tencent, prod, ~10.4k ⭐ | Active (Buildkite product) | Early development | GA 2026 | Early research preview |
|
||||
|
||||
## What's closest, what's different
|
||||
|
||||
**Closest in design and scope.** agent-safehouse and litterbox sit
|
||||
nearest bot-bottle: local, single-user, thin wrappers over an
|
||||
existing OS primitive, low-dep. The split is the isolation primitive —
|
||||
bot-bottle uses Docker + pipelock egress (plus gVisor where
|
||||
available); agent-safehouse uses `sandbox-exec`; litterbox uses Podman +
|
||||
Landlock. matchlock and smolmachines are spiritually close on the
|
||||
*policy* side (default-deny net, per-host allowlist) but use microVMs
|
||||
instead of containers.
|
||||
bot-bottle now defaults to a VM per bottle (Firecracker microVM on KVM
|
||||
Linux, Apple Container on macOS) with its own DLP-scanning egress proxy,
|
||||
keeping Docker only as a legacy fallback; agent-safehouse uses
|
||||
`sandbox-exec`; litterbox uses Podman + Landlock. matchlock and
|
||||
smolmachines are close on *both* the policy side (default-deny net,
|
||||
per-host allowlist) and — now that bot-bottle has moved off
|
||||
containers-by-default — the microVM isolation primitive. Note: Apple
|
||||
Container 1.0 stable shipped June 9 2026 (frozen CLI and APIs), which
|
||||
makes the macOS backend stable surface area rather than a moving target.
|
||||
|
||||
**New closest on agent-tailored policy.** Two governance tools are the
|
||||
direct competitors on the "coarse-grained sandbox" axis. **tilde.run**
|
||||
has had per-agent DSL RBAC since its launch (though it's hosted SaaS).
|
||||
**Microsoft AGT** is the most serious new entrant: per-agent DID
|
||||
identity, YAML policy that can allow/deny/sandbox/approve individual tool
|
||||
calls per agent, and a dynamic behavioural trust score. It operates at
|
||||
the framework tool-call layer, not the network layer — so it's
|
||||
complementary to bot-bottle's network/filesystem isolation rather than a
|
||||
direct substitute, but on the "does this sandbox know what this agent is
|
||||
for?" question it is the most complete answer in the field. OAP's
|
||||
pre-action hook pattern achieves similar goals with cryptographic audit
|
||||
and a 0% adversarial-attack success rate under a restrictive policy.
|
||||
|
||||
**New closest on DX.** **Docker sbx** is the first tool in this set that
|
||||
matches bot-bottle on the "one command, dangerously-skip-permissions safe
|
||||
by default" DX bar, at microVM isolation strength, with host-side
|
||||
credential injection. It is proprietary, preset-based (not role-
|
||||
declarative), and cloud-agent-specific, but it directly competes on the
|
||||
UX proposition. agent-safehouse was the previous DX peer; Docker sbx
|
||||
materially raises the bar.
|
||||
|
||||
**New closest on repo-scoped policy.** **Cleanroom** (Buildkite) is the
|
||||
first tool to combine microVM isolation with a declarative egress policy
|
||||
file — though the policy lives in the repo being sandboxed
|
||||
(`cleanroom.yaml`), not in an agent-role manifest. That makes it per-
|
||||
repo rather than per-role: the same Cleanroom config applies to any
|
||||
agent running in that repo. The distinction matters for bot-bottle's
|
||||
use case (one developer running multiple agent *roles* with different
|
||||
egress footprints), but for CI/CD use cases Cleanroom is a direct
|
||||
alternative.
|
||||
|
||||
**Solving a different problem.** tilde.run is hosted SaaS for team /
|
||||
production agent pipelines with data-versioned rollback — explicitly
|
||||
opposite to bot-bottle's "infrastructure I control" goal. boxlite and
|
||||
microsandbox are infrastructure libraries aimed at platform builders
|
||||
embedding sandboxes into agent frameworks; they would be a *backend*
|
||||
bot-bottle could call, not a competitor to its manifest layer.
|
||||
endo-familiar is in a different paradigm entirely: capability passing
|
||||
rather than kernel boundaries.
|
||||
opposite to bot-bottle's "infrastructure I control" goal. E2B and Daytona
|
||||
are hosted sandbox platforms, while boxlite, microsandbox, and CubeSandbox
|
||||
are infrastructure libraries/services aimed at platform builders embedding
|
||||
sandboxes into agent frameworks; they
|
||||
would be a *backend* bot-bottle could call, not a competitor to its
|
||||
manifest layer. endo-familiar is in a different paradigm entirely:
|
||||
capability passing rather than kernel boundaries.
|
||||
|
||||
## Borrowable ideas
|
||||
|
||||
What bot-bottle already has that the survey suggested as
|
||||
differentiators:
|
||||
- Default-deny egress with a per-agent allowlist (pipelock).
|
||||
### Already shipped or otherwise addressed
|
||||
|
||||
- Default-deny egress with a per-agent allowlist (own egress scanner).
|
||||
- DLP scanning of outbound traffic.
|
||||
- Bottle / agent split (manifest layer above the isolation primitive).
|
||||
- gVisor auto-detection on Linux.
|
||||
- **In-flight secret injection** (suggested by matchlock) — **shipped.**
|
||||
Real provider and git-host tokens are held outside the agent and injected
|
||||
by the egress gateway on matching routes. The agent receives only proxy
|
||||
URLs and, where a client requires a credential-shaped value, a placeholder;
|
||||
`GITEA_TOKEN` and equivalent real tokens do not appear in the agent's
|
||||
environment.
|
||||
- **MicroVM backend** — **shipped.** MicroVMs are now the default:
|
||||
Firecracker on KVM Linux and Apple Container on macOS. Docker is the legacy
|
||||
fallback.
|
||||
- **Per-use SSH key confirmation** (suggested by litterbox) — **addressed by
|
||||
stronger credential custody instead.** The agent does not hold the upstream
|
||||
git SSH key or an SSH-agent socket: git-gate holds the credential and gates
|
||||
git operations. A confirmation wrapper inside the agent would therefore
|
||||
protect a credential that is no longer there. Operator approval at the gate
|
||||
remains the appropriate control point for any future per-use confirmation.
|
||||
|
||||
Ideas worth considering, without abandoning the Python-stdlib-first / local-Docker
|
||||
stance:
|
||||
### Still worth considering
|
||||
|
||||
1. **Per-use SSH key confirmation** (from litterbox). Even with
|
||||
KnownHostKey pinning and pipelock egress, a wrapper SSH agent that
|
||||
prompts on each key use (e.g. via `osascript` / `notify-send`) would
|
||||
catch an agent doing something off-policy with a key it legitimately
|
||||
holds. Pure-stdlib, no new deps.
|
||||
2. **In-flight secret injection** (from matchlock). Pipelock already
|
||||
does egress allowlisting and DLP; teaching it to *inject* tokens at
|
||||
proxy time so e.g. `GITEA_TOKEN` never appears in the container's
|
||||
env would close the "agent reads its own env and exfiltrates" path.
|
||||
Fits the existing pipelock architecture.
|
||||
3. **MicroVM backend as an opt-in bottle type** — already on the radar
|
||||
in `stronger-isolation-alternatives.md`. microsandbox, smolmachines,
|
||||
and matchlock all show that libkrun + Apple's
|
||||
Virtualization.framework is ergonomic enough that a
|
||||
`"runtime": "microvm"` field on a bottle is plausible without a heavy
|
||||
stack.
|
||||
- **Live network activity in the supervisor TUI** (from Docker sbx): show
|
||||
allowed and blocked connections and let the operator propose policy changes
|
||||
from the existing supervision surface.
|
||||
- **Tamper-evident audit records** (from OAP): sign and hash-chain egress and
|
||||
supervision decisions for compliance-sensitive deployments.
|
||||
- **Behaviour-informed policy downgrade** (from Microsoft AGT): use repeated
|
||||
DLP alerts or supervision holds as a signal to narrow policy or request
|
||||
closer review. This needs a carefully specified trust model before it can be
|
||||
more than a heuristic.
|
||||
|
||||
Not worth borrowing: the SDK-first programmatic API style of boxlite /
|
||||
microsandbox (cuts against the declarative-manifest stance), and the
|
||||
hosted-SaaS dashboard model of tilde.run (cuts against the
|
||||
"infrastructure I control" goal).
|
||||
|
||||
## Publishing and positioning verdict
|
||||
|
||||
Publishing remains worthwhile, but the defensible claim is the combination,
|
||||
not any single primitive. Credential custody is matched by OneCLI, matchlock,
|
||||
Daytona, Docker sbx, and CubeSandbox; local one-command isolation is matched by
|
||||
agent-safehouse and Docker sbx; hosted microVM execution is a crowded platform
|
||||
category.
|
||||
|
||||
bot-bottle remains unusual in combining:
|
||||
|
||||
- local, operator-controlled execution with persistent named bottles;
|
||||
- one declarative role layer across Claude Code, Codex, Pi, and provider
|
||||
plugins;
|
||||
- composable agent/bottle manifests, skills, and system prompts;
|
||||
- Firecracker/Apple Container isolation with a Docker fallback;
|
||||
- default-deny per-role egress, payload DLP, and git-push secret scanning;
|
||||
- credentials injected outside the agent process; and
|
||||
- supervision and audit state suited to long-running parallel agents.
|
||||
|
||||
The practical wedge is “as easy as native yolo, with declarative role policy
|
||||
and self-hosted custody,” including scoped access to private LAN/Tailnet
|
||||
services that cloud-first runtimes cannot provide without additional network
|
||||
plumbing. The main competitive risks are a local wrapper such as claudebox or
|
||||
Docker sbx growing a role-manifest layer, and GUI products such as SuperHQ
|
||||
adding equivalent policy and audit depth.
|
||||
|
||||
## Caveats
|
||||
|
||||
- Star counts and last-commit dates are point-in-time snapshots.
|
||||
@@ -230,3 +691,180 @@ hosted-SaaS dashboard model of tilde.run (cuts against the
|
||||
- The `superradcompany/microsandbox` URL in the original prompt
|
||||
redirects to `microsandbox/microsandbox`; the surveyed project is the
|
||||
same.
|
||||
- CubeSandbox performance/scale numbers (<60ms cold start, <5MB/instance,
|
||||
2,000+ sandboxes per 96-vCPU host) are the project's own launch claims,
|
||||
not independently verified here.
|
||||
|
||||
## Addendum 2026-07-18 — CubeSandbox and the positioning read
|
||||
|
||||
CubeSandbox (Tencent Cloud, Apache 2.0, ~10.4k stars, HN launch
|
||||
[#47863430](https://news.ycombinator.com/item?id=47863430)) is the first
|
||||
open-source, self-hostable project in this survey to combine, in one stack,
|
||||
the main primitives
|
||||
bot-bottle treated as its differentiator:
|
||||
|
||||
- **Egress custody (connection level)** — default-deny domain allowlist
|
||||
(L7 domain/SNI filtering), instant block on unauthorized egress,
|
||||
per-sandbox traffic tokens, full audit logs of destinations (eBPF
|
||||
virtual switch, "CubeVS"). This matches bot-bottle's egress scanner at
|
||||
the *connection level*, productized — see the one thing it does **not**
|
||||
match, below.
|
||||
- **Credential custody** — a vault where keys "never enter the sandbox,
|
||||
model context, or logs." This is the in-flight-injection idea from
|
||||
matchlock, but as a first-class feature, and it's exactly the
|
||||
cross-vendor "egress audit + custody" wedge the monetization
|
||||
positioning treats as the one defensible moat.
|
||||
- **Isolation on par with bot-bottle's current default** — a dedicated
|
||||
guest kernel per sandbox (RustVMM/KVM). bot-bottle now defaults to the
|
||||
same class of boundary (Firecracker microVM / Apple Container), so this
|
||||
is parity, not an edge; CubeSandbox's remaining edge is running that
|
||||
per-kernel isolation multi-tenant at scale on one host.
|
||||
|
||||
The one axis CubeSandbox does **not** cover — and where bot-bottle stays
|
||||
distinctive:
|
||||
|
||||
- **Content DLP on *authorized* channels.** CubeSandbox's egress control
|
||||
is connection-level: it decides *whether* a destination is allowed and
|
||||
logs it, and its vault keeps *injected* credentials out of the sandbox
|
||||
entirely. Neither inspects the *payload* of traffic to an allowed
|
||||
destination. So an agent that exfiltrates over a permitted channel —
|
||||
pasting a repo's contents, an agent-derived secret, or PHI into an
|
||||
allowed API/domain — is not caught by CubeSandbox. bot-bottle's own
|
||||
egress DLP scanner does scan that: response + websocket content against
|
||||
the resolved per-flow config, with per-bottle token redaction (see
|
||||
recent egress commits). The vault
|
||||
approach is arguably *stronger* for the specific case of pre-known
|
||||
injected credentials (they can't leak if they were never present), but
|
||||
it is not a substitute for content inspection of everything else.
|
||||
|
||||
**Long-running posture — a sharper axis than raw isolation.** E2B and
|
||||
CubeSandbox are *ephemeral-per-task* by design; a long-running agent is an
|
||||
architected pattern on top, not the default. E2B: 5-minute default
|
||||
timeout, continuous runtime tier-capped (~1h Hobby / ~24h Pro), duration
|
||||
achieved via **pause/resume** (preserves filesystem + memory + processes;
|
||||
reconnect by sandbox ID via `Sandbox.connect()`; resume resets the timeout
|
||||
to 5 min; auto-pause via `on_timeout: "pause"`). CubeSandbox mirrors this
|
||||
(E2B drop-in) with first-class auto pause/resume and hundred-ms
|
||||
checkpoint/fork — and, self-hosted, sets its own timeout policy with no
|
||||
vendor tier caps. bot-bottle inverts the model: a bottle is **persistent,
|
||||
named, and supervised by default** — long-running *is* the default, not a
|
||||
session-management loop over pause/resume. smolmachines is the other
|
||||
persistent-by-default project in this set. For anyone building agents that
|
||||
run for hours/days, this posture difference matters more than the
|
||||
isolation primitive.
|
||||
|
||||
**DX — the "run Claude yolo-style" bar.** The reason `claude
|
||||
--dangerously-skip-permissions` is so widely used is DX: it's one command
|
||||
and the agent just goes. The bottle thesis is to make a *sandboxed* run
|
||||
that easy — `start <agent>` builds the image on first run and drops you
|
||||
into an interactive Claude session that already has
|
||||
`--dangerously-skip-permissions` on by default
|
||||
(`contrib/claude/agent_provider.py`), with the sandbox as the guardrail
|
||||
instead of per-action prompts. On this axis the field splits cleanly:
|
||||
- **Wrappers around the agent** (as-easy-as-native): bot-bottle and
|
||||
**agent-safehouse** (`safehouse claude --dangerously-skip-permissions`).
|
||||
These *are* the run-Claude experience. agent-safehouse is the real DX
|
||||
peer — but it's macOS-only Seatbelt, single-run, and doesn't address
|
||||
network egress; bot-bottle adds VM-grade isolation, egress DLP, and
|
||||
persistent/parallel bottles across macOS + Linux.
|
||||
- **Libraries / services** (you build the run yourself): boxlite,
|
||||
microsandbox, CubeSandbox, E2B, Daytona. These hand you an SDK or a cluster and
|
||||
expect you to wire the agent in — powerful for platform builders,
|
||||
heavyweight for "just run Claude on my laptop." microsandbox's MCP/Skills
|
||||
angle is *sandbox-as-a-tool the agent calls*, which is the inverse of
|
||||
wrapping the agent.
|
||||
- **In between:** litterbox (wizard + build, Linux only), smolmachines
|
||||
(SSH into a named machine), matchlock (run a command in a VM).
|
||||
|
||||
So DX is a genuine bot-bottle differentiator. agent-safehouse matches the
|
||||
one-command wrapper with weaker isolation and no egress story; Docker sbx now
|
||||
matches it at microVM strength but remains proprietary and preset-based. "As
|
||||
easy as native yolo, with declarative role policy" is the narrower defensible
|
||||
one-liner.
|
||||
|
||||
Why it still doesn't collide head-on:
|
||||
|
||||
1. **Shape.** CubeSandbox is a *multi-tenant service for platform
|
||||
builders* (drop-in E2B replacement, SDK-driven, 2,000 sandboxes on a
|
||||
box). bot-bottle is a *single-operator, declarative-manifest tool for
|
||||
the infrastructure I run*. Different buyer, different ergonomics — no
|
||||
declarative role manifest, no bottle/agent split, no "one command on my
|
||||
laptop."
|
||||
2. **Backend, not competitor.** Like boxlite/microsandbox, CubeSandbox is
|
||||
something bot-bottle could sit *on top of* — a `"runtime": "microvm"`
|
||||
or `"runtime": "cubesandbox"` backend under the manifest layer — while
|
||||
keeping the manifest, the bottle/agent split, and the local,
|
||||
single-operator default.
|
||||
|
||||
Why it matters anyway:
|
||||
|
||||
- The "nobody else bundles connection-level egress allowlist + audit +
|
||||
in-flight credential custody" line is **no longer true for the
|
||||
primitive** — CubeSandbox ships the open-source/self-hosted combination,
|
||||
and Daytona ships a proprietary firewall + credential-substitution variant.
|
||||
But **content DLP on authorized channels is still not matched** (see
|
||||
above), and neither is the *layer above* the primitive (declarative
|
||||
manifest, cross-vendor orchestration, operator UX, the
|
||||
phone-control/dashboard north star). Those two — outbound-payload DLP
|
||||
and the orchestration layer — are where the defensible ground now sits;
|
||||
the connection-level allowlist + vault mechanism, on its own, is no
|
||||
longer differentiating. Revisit the monetization open/paid line with
|
||||
that in mind.
|
||||
- Worth a closer look at **how** CubeSandbox does credential injection
|
||||
and per-sandbox egress tokens (eBPF virtual switch vs. bot-bottle's
|
||||
mitmproxy egress proxy) when hardening bot-bottle's now-shipped
|
||||
credential-custody implementation.
|
||||
|
||||
## Addendum 2026-07-18 (second pass) — agent-tailored policy landscape
|
||||
|
||||
The second-pass question was: how novel is bot-bottle's per-agent,
|
||||
role-tailored sandbox relative to the expanded field?
|
||||
|
||||
**The short answer:** on the isolation + network + role-tailoring
|
||||
combination, bot-bottle remains the only tool in this set. On
|
||||
role-tailored *policy at the tool-call level*, Microsoft AGT and OAP are
|
||||
the most complete answers, but they don't provide isolation; they
|
||||
complement rather than substitute.
|
||||
|
||||
**The competitive picture by axis:**
|
||||
|
||||
- *Agent-tailored egress (declarative, per-role)* — bot-bottle and
|
||||
tilde.run. Cleanroom is per-repo, not per-role. Everyone else is
|
||||
per-session or not addressed.
|
||||
- *Agent-tailored tool-call policy (declarative, per-agent identity)* —
|
||||
Microsoft AGT (YAML policy + DID identity + trust score), OAP
|
||||
(declarative policy rules + cryptographic audit). Neither provides
|
||||
network/filesystem isolation.
|
||||
- *Composable policy (role overlays)* — bot-bottle (`extends:`). No
|
||||
other tool surveyed supports composable role-policy inheritance.
|
||||
- *Isolation + DX (one-command safe yolo)* — bot-bottle and Docker sbx.
|
||||
Docker sbx is proprietary, preset-based, and cloud-agent-specific;
|
||||
it's the first DX-class competitor at microVM isolation strength.
|
||||
|
||||
**What the HN "coarse-grained" complaint maps to:** The complaint is
|
||||
that a VM isolates the filesystem but doesn't know if the agent
|
||||
*should* be sending an email. bot-bottle's bottle/agent split is a
|
||||
structural answer to this: the bottle manifest declares exactly what
|
||||
the role can reach, and the sandbox enforces it at the network layer.
|
||||
Microsoft AGT is the most complete answer at the semantic/tool-call
|
||||
layer. The gap both leave open is *intent classification* — knowing
|
||||
whether a permitted action is consistent with the agent's actual task.
|
||||
See `hn-agent-safety-discourse-july-2026.md` for the blast-radius
|
||||
analysis.
|
||||
|
||||
**Open ideas from new tools (also summarized above):**
|
||||
|
||||
- **Microsoft AGT's trust-score decay** — privilege that reflects
|
||||
observed behaviour rather than static provisioning. Applied to
|
||||
bot-bottle: a bottle that has triggered DLP alerts or supervise holds
|
||||
could auto-downgrade its network preset, or flag the session for
|
||||
closer review. Fits the existing supervise-server architecture.
|
||||
- **Docker sbx's live network TUI** — real-time per-session view of
|
||||
allowed and blocked outbound connections with point-and-click
|
||||
allow/block. `cli.py supervise` is the right surface; adding a
|
||||
live-connections panel would directly address the "I can't see what
|
||||
the agent is doing" gap without any backend changes.
|
||||
- **OAP's cryptographic audit chain** — Ed25519-signed, hash-chained
|
||||
audit records. Currently bot-bottle logs egress decisions but doesn't
|
||||
chain them. A tamper-evident audit record per session would be useful
|
||||
for the compliance use case the CubeSandbox positioning targets.
|
||||
|
||||
@@ -0,0 +1,357 @@
|
||||
# HN discourse on agent sandbox safety — June/July 2026
|
||||
|
||||
A survey of community opinion and notable security disclosures on Hacker
|
||||
News and adjacent sources over June–July 2026. The question: what does
|
||||
the current discourse say about whether sandboxes are sufficient for
|
||||
agentic AI safety, and where does bot-bottle land against the issues
|
||||
being raised?
|
||||
|
||||
Research conducted 2026-07-18.
|
||||
|
||||
## Summary
|
||||
|
||||
The past month marks a turning point in community opinion. Earlier in
|
||||
2026, the debate was mostly "which sandbox tool is best?" By June–July,
|
||||
a cascade of critical CVEs and novel attack classes has shifted the
|
||||
framing to "sandboxes are not enough — what else do you need?" The
|
||||
attacks that drove this shift are structurally distinct: most route
|
||||
through legitimate, trusted channels (Sentry issues, MCP descriptions,
|
||||
README files) rather than exploiting the isolation boundary directly.
|
||||
|
||||
bot-bottle's architecture holds up well against the direct-escape class
|
||||
(Firecracker/Apple Container default backends, credentials never in the
|
||||
agent's env, harness entirely on the host). The remaining gap is prompt
|
||||
injection — attacker-controlled data interpreted as model instructions.
|
||||
Egress controls and prompt injection defenses are orthogonal: egress
|
||||
limits what the agent can *send out*; injection is about what it is
|
||||
*told to do*. The two don't substitute for each other. Inside a tightly-
|
||||
egressed sandbox a successful injection can't exfiltrate to unknown
|
||||
hosts, but it can still corrupt the work product, push malicious commits
|
||||
past a secret scanner, or use allowlisted channels for exfiltration.
|
||||
Those residual risks are addressed below.
|
||||
|
||||
## The sandboxing boom sets the stage
|
||||
|
||||
The preceding months generated a wave of sandbox tooling. A March 28
|
||||
Ask HN thread
|
||||
([#47444917](https://news.ycombinator.com/item?id=47444917)) catalogued
|
||||
the explosion: E2B, AIO Sandbox, AgentSphere, Yolobox, Exe.dev,
|
||||
AgentFence, DenoSandbox, Capsule (WASM), ERA, Vibekit, Daytona, Modal,
|
||||
Nono, and more — all launched within roughly 12 months. A parallel March
|
||||
9 thread ([#47185250](https://news.ycombinator.com/item?id=47185250))
|
||||
surveyed what developers were actually deploying: "containers or YOLO"
|
||||
dominated. The honest community mood was that most teams hadn't solved
|
||||
this and were shipping anyway.
|
||||
|
||||
The March 12 launch of **Agent Safehouse**
|
||||
([#47301085](https://news.ycombinator.com/item?id=47301085), 823 points)
|
||||
crystallised the community framing: a zero-dep `sandbox-exec` wrapper for
|
||||
macOS that attracted the top comment *"I honestly think that sandboxing is
|
||||
currently THE major challenge that needs to be solved for the tech to fully
|
||||
realise its potential."* The creator's own framing — "no dependencies, no
|
||||
daemons, no subscription; the simplicity is the feature" — and Simon
|
||||
Willison's observation that evaluating whether a sandboxing tool works as
|
||||
intended is itself hard, both prefigure the June–July shift in tone. See
|
||||
[`agent-sandbox-landscape.md`](agent-sandbox-landscape.md) for a full
|
||||
per-project breakdown.
|
||||
|
||||
## The June–July attack cascade
|
||||
|
||||
Six attack patterns broke in quick succession. Together they form the
|
||||
argument that the community's framing was wrong: the threat model for
|
||||
agents isn't just "code that escapes its container" — it's also prompt
|
||||
injection, where attacker-controlled data is interpreted as model
|
||||
instructions regardless of whether any isolation boundary was crossed.
|
||||
Sections 2–4 below are all the same attack class; the "trusted channel"
|
||||
label describes the delivery vector, not a different threat.
|
||||
|
||||
### 1. Sandbox escape CVEs (DuneSlide, CVE-2026-39861)
|
||||
|
||||
Cato AI Labs disclosed **DuneSlide** (CVE-2026-50548/50549, CVSS 9.8),
|
||||
a pair of flaws in Cursor 2.x. CVE-2026-50548 abuses the sandbox's
|
||||
`working_directory` parameter to point writes at system files; CVE-26-50549
|
||||
exploits a symlink-resolution fallback that fails open. Both start with
|
||||
a prompt injection and end in sandbox escape — and Cato's framing was
|
||||
blunt: "each CVE defeats a different guardrail; the problem is
|
||||
structural, not a string of one-offs."
|
||||
|
||||
Claude Code's own sandbox had a similar escape this year:
|
||||
**CVE-2026-39861** (symlink flaw). The CurXecute/MCPoison/CVE-2026-26268
|
||||
chain from Cursor added a poisoned Slack message, a swap-after-approval
|
||||
MCP config, and a Git hook as three more entry points in the same
|
||||
attack class.
|
||||
|
||||
All patched, but the pattern holds: any application-level sandbox that
|
||||
takes attacker-influenced values as path parameters is reachable from a
|
||||
prompt injection.
|
||||
|
||||
### 2. Prompt injection via MCP data (Agentjacking)
|
||||
|
||||
Tenet's "Agentjacking" technique planted a fake bug report in Sentry's
|
||||
MCP output. When an agent queries Sentry to fix open issues, the
|
||||
malicious event is rendered as structured content visually
|
||||
indistinguishable from a real Sentry event, and the agent executes the
|
||||
embedded instructions with the developer's full privileges. Hit rate
|
||||
across Claude Code and Cursor: **85%**. The route is entirely through a
|
||||
legitimately-authorized MCP channel — no isolation boundary is crossed;
|
||||
the injection arrives inbound through a channel the sandbox explicitly
|
||||
trusts.
|
||||
|
||||
The Cloud Security Alliance's summary: treat observability, bug-report,
|
||||
and integration data as **untrusted agent input**, not neutral
|
||||
development metadata.
|
||||
|
||||
### 3. README-embedded prompt injection
|
||||
|
||||
A July disclosure showed malicious instructions hidden in `README.md`
|
||||
— a file that receives no trust prompt and requires no elevated access.
|
||||
When asked point-blank whether the repo held hidden instructions, both
|
||||
Claude Sonnet 4.6 and GPT-5.5 said no. A payload written for Sonnet
|
||||
4.6 transferred unchanged to Sonnet 5, Opus 4.8, and GPT-5.5. The
|
||||
attack surface is every repo an agent is asked to work in.
|
||||
|
||||
### 4. Prompt injection via MCP tool descriptions
|
||||
|
||||
Microsoft research (June 30) showed that attacker-controlled MCP tool
|
||||
description fields can silently redirect agent behavior. The injection
|
||||
is embedded in metadata the model reads during tool selection — before
|
||||
any sandbox enforcement or egress check runs, and entirely on the
|
||||
inbound path that egress controls cannot touch.
|
||||
|
||||
### 5. MCP STDIO command injection (10 CVEs)
|
||||
|
||||
OX Security disclosed a systemic command injection class in Anthropic's
|
||||
MCP protocol, covering 10 CVEs across multiple coding agents. The
|
||||
Windsurf case (CVE-2026-30615): processing attacker-controlled HTML
|
||||
causes the agent to auto-register a malicious MCP STDIO server and
|
||||
execute arbitrary commands with no further user interaction.
|
||||
|
||||
### 6. LiteLLM gateway compromise (CVE-2026-40217, CVE-2026-42271)
|
||||
|
||||
CVE-2026-40217 exposes LiteLLM's guardrail sandbox via `exec()` with no
|
||||
source filtering. CVE-2026-42271 (exploited in the wild, added to CISA's
|
||||
KEV catalog) lets callers spawn subprocesses through MCP preview
|
||||
endpoints. The threat extends to any agent routed through a compromised
|
||||
LiteLLM proxy: the proxy can swap model responses for forged tool calls
|
||||
in transit, giving the attacker a reverse shell from the developer's
|
||||
machine.
|
||||
|
||||
## HN community opinion clusters
|
||||
|
||||
**"Move enforcement to the kernel, not the app"** — the Nono Show HN
|
||||
([#46849615](https://news.ycombinator.com/item?id=46849615)) and a
|
||||
kernel-sandbox thread
|
||||
([#47066574](https://news.ycombinator.com/item?id=47066574)) both argued
|
||||
that application-layer sandboxes are inherently bypassable by the code
|
||||
they're sandboxing. The academic framing, from *Red-Teaming the Agentic
|
||||
Red-Team* ([arXiv 2606.24496](https://arxiv.org/pdf/2606.24496)):
|
||||
"enforcement should occur at the OS level via the kernel refusing system
|
||||
calls that violate policy at runtime — not pre-execution argument
|
||||
validation in tool calls."
|
||||
|
||||
**"The harness belongs outside the sandbox"** — a May thread
|
||||
([#47990675](https://news.ycombinator.com/item?id=47990675)) converged
|
||||
on clean architectural separation: harness in one VM, tool execution in
|
||||
another. Top comment: "having the harness in one VM, and tool use applied
|
||||
to user data in another, is about as safe as you can be at present."
|
||||
Several replies described a hypervisor-like policy layer — sitting outside
|
||||
both VMs — as the right long-term model.
|
||||
|
||||
**"Sandboxes are too coarse-grained"** — a Feb thread
|
||||
([#47006445](https://news.ycombinator.com/item?id=47006445)) argued
|
||||
that VMs don't answer the real question: knowing whether an agent
|
||||
*should* be sending an email or making a transaction. "Everything's just
|
||||
in the same big box." This framing picked up traction through June–July
|
||||
as the trusted-channel attacks dominated.
|
||||
|
||||
**"MCP's trust model is the real problem"** — the month's recurring
|
||||
theme. MCP by design gives agents access to authorized external services.
|
||||
Once a trusted channel delivers a malicious payload, filesystem sandboxing
|
||||
is irrelevant. The community call: treat all MCP tool metadata and return
|
||||
values as untrusted input subject to policy validation before ingestion,
|
||||
and disable automatic MCP server loading from untrusted repositories.
|
||||
|
||||
## How bot-bottle addresses these issues
|
||||
|
||||
### What it covers well
|
||||
|
||||
**Direct sandbox escape (CVEs, container breakout)**
|
||||
|
||||
bot-bottle's default backends are Firecracker microVM (KVM Linux) and
|
||||
Apple Container (macOS). Both run the agent in a separate VM with a
|
||||
dedicated kernel — the container-escape CVE class (Dirty Pipe, runc
|
||||
escapes, DuneSlide's path-parameter abuse) requires escaping a real
|
||||
hypervisor boundary, not just a namespace. On the legacy Docker backend,
|
||||
gVisor auto-detection provides a userspace syscall barrier for hosts where
|
||||
neither KVM nor Apple Container is available.
|
||||
|
||||
The bot-bottle process itself runs entirely on the host, outside the VM.
|
||||
This is the "harness outside the sandbox" architecture the HN thread
|
||||
converged on as best practice. The bottle manifest, egress rules, and
|
||||
secrets never enter the agent VM.
|
||||
|
||||
**Credential theft on sandbox escape**
|
||||
|
||||
Even on a successful VM/container escape, the agent has nothing useful
|
||||
to steal. Credentials are injected in-flight by the gateway proxy
|
||||
(`auth.scheme` / `auth.token_ref` in the egress route config) — `printenv`
|
||||
inside the agent shows proxy URLs only. The git-gate similarly holds the
|
||||
upstream SSH credential on the host; the agent pushes through a
|
||||
gitleaks-scanned daemon that forwards clean refs upstream. An escaped
|
||||
agent gets the host filesystem, not the keys.
|
||||
|
||||
**Orphaned-agent credential risk**
|
||||
|
||||
bot-bottle is explicitly ephemeral: when the agent exits, `cli.py` tears
|
||||
down every gateway and both networks — nothing persists between runs. The
|
||||
agent never holds credentials, so there is nothing to orphan.
|
||||
|
||||
**MCP config redirection / STDIO auto-registration**
|
||||
|
||||
The trust boundary at `$HOME` means bottles live only under
|
||||
`~/.bot-bottle/bottles/` — a cloned repo cannot add egress routes or
|
||||
redirect env vars to attacker hosts (the design rationale is in
|
||||
`docs/prds/0011-per-file-md-manifest.md`). Auto-registering a malicious
|
||||
MCP STDIO server from within the agent is still sandboxed by the VM, and
|
||||
any outbound calls from that server must pass the egress allowlist and
|
||||
outbound DLP scanner.
|
||||
|
||||
**Per-agent role tailoring (the "coarse-grained sandbox" complaint)**
|
||||
|
||||
The Feb 2026 HN thread that argued "sandboxes are too coarse-grained"
|
||||
was pointing at a real gap: a VM isolates the filesystem but doesn't
|
||||
know whether an agent *should* be sending email or calling an external
|
||||
API. bot-bottle's bottle/agent split is a structural answer at the
|
||||
network layer — the bottle manifest declares exactly what each role can
|
||||
reach (which hosts, which paths, which HTTP methods), and the egress
|
||||
scanner enforces it. A `gitea-dev` bottle that only lists
|
||||
`gitea.dideric.is` and `api.anthropic.com` structurally cannot send
|
||||
email or reach AWS, not because the model was told not to, but because
|
||||
those routes don't exist.
|
||||
|
||||
The `extends:` composition model means provider-level policy (the Claude
|
||||
auth route) lives in one base bottle and role-specific overlays are
|
||||
stacked on top — no duplication, and changing the base propagates to all
|
||||
derived roles.
|
||||
|
||||
Competitive position on this axis (from `agent-sandbox-landscape.md`):
|
||||
|
||||
| Tool | Agent-tailored policy |
|
||||
|---|---|
|
||||
| **bot-bottle** | Yes — declarative per-role manifest; `extends:` composition; egress + credentials scoped to role |
|
||||
| **tilde.run** | Yes — per-agent DSL RBAC (allow/deny/approve per action/repo/agent), but hosted SaaS |
|
||||
| **Microsoft AGT** | Yes — YAML policy + per-agent DID + trust score, but tool-call level only (no network isolation) |
|
||||
| **OAP** | Yes — declarative pre-action policy + cryptographic audit, but no isolation |
|
||||
| **Cleanroom** | Partial — per-repo `cleanroom.yaml`, not per-role |
|
||||
| **Docker sbx** | No — network presets only |
|
||||
| **Anthropic srt** | No — programmatic per-invocation |
|
||||
| **matchlock / smolmachines / microsandbox** | No |
|
||||
| **agent-safehouse** | Partial — per-agent Seatbelt profiles; no egress |
|
||||
|
||||
Two takeaways: bot-bottle and tilde.run are the only isolation tools
|
||||
with declarative role-tailored policy; Microsoft AGT and OAP are the
|
||||
closest competitors on role-tailoring but operate at the tool-call layer
|
||||
without network/filesystem isolation — complementary, not substitutes.
|
||||
|
||||
**Outbound exfiltration (any injection class)**
|
||||
|
||||
Whatever triggers the agent — README injection, Agentjacking, MCP
|
||||
description poisoning — the final step in most attacks is exfiltration.
|
||||
bot-bottle's egress allowlist is default-deny with a per-bottle host
|
||||
allowlist; unknown hosts get a hard 403. Outbound DLP scanning
|
||||
(`outbound_detectors: [token_patterns, known_secrets]`) catches tokens
|
||||
and secrets in outbound bodies; the `supervise` policy (default for
|
||||
manifest routes) holds the request for operator approval rather than
|
||||
silently blocking it. Together these limit what a successful injection
|
||||
can *do* even if it succeeds at the model layer.
|
||||
|
||||
**LiteLLM / compromised-proxy attacks**
|
||||
|
||||
bot-bottle does not use LiteLLM. The model API route (e.g.
|
||||
`api.anthropic.com`) is an auto-injected provider route on the egress
|
||||
allowlist; the agent dials the gateway, not the model API directly.
|
||||
A compromised third-party proxy is not in the architecture.
|
||||
|
||||
### Where it is weaker
|
||||
|
||||
**Prompt injection**
|
||||
|
||||
Egress controls and prompt injection defenses are orthogonal. Egress
|
||||
limits what the agent can *send out* (outbound leg); prompt injection
|
||||
is about what attacker-controlled data *tells the agent to do* (inbound
|
||||
leg). The two don't substitute for each other and must be treated
|
||||
separately.
|
||||
|
||||
The inbound DLP scanner (`inbound_detectors: [naive_injection_detection]`)
|
||||
is the only runtime defense against injection arriving through allowlisted
|
||||
channels — Sentry MCP responses, MCP tool descriptions, README content.
|
||||
It is explicitly pattern-matching and will not catch a sufficiently
|
||||
crafted payload. There is no semantic / intent-level gate between what
|
||||
the model decides and what the agent executes.
|
||||
|
||||
**Blast radius within the permitted scope**
|
||||
|
||||
Inside a tightly-egressed sandbox a successful injection can't
|
||||
exfiltrate to unknown hosts, but it still has real options:
|
||||
|
||||
- *Work product corruption.* The agent can modify, delete, or backdoor
|
||||
files in the working directory. This is within its permitted scope;
|
||||
egress controls have nothing to say about it.
|
||||
|
||||
- *Malicious commits past the git-gate.* The git-gate scans outbound
|
||||
refs for secrets (gitleaks), not for semantic code intent. A prompt-
|
||||
injected agent can commit subtly malicious code — logic bombs,
|
||||
backdoored auth paths, code that exfiltrates data through the
|
||||
application's own HTTP clients at runtime — that looks clean to a
|
||||
secret scanner.
|
||||
|
||||
- *Exfiltration through allowlisted channels.* If an attacker knows or
|
||||
can predict what hosts are in the egress allowlist, those channels are
|
||||
available for exfiltration. A GitHub remote being allowlisted means
|
||||
"push to an attacker-controlled fork" is viable. A logging endpoint
|
||||
being allowlisted means structured data can leave through it. The
|
||||
outbound DLP scanner catches credential tokens and known secrets but
|
||||
not arbitrary business data.
|
||||
|
||||
- *Dependency installation within the sandbox.* An agent that runs
|
||||
`npm install` or `pip install` on attacker-specified packages executes
|
||||
code inside the sandbox with the same capabilities the agent has:
|
||||
filesystem access, tool calls, calls to allowlisted hosts. Supply chain
|
||||
injection via package names is in the same injection family, triggered
|
||||
by the same prompt-injection path.
|
||||
|
||||
### What would close the remaining gaps
|
||||
|
||||
The blast-radius risks above point at two distinct mitigations that
|
||||
don't yet exist in bot-bottle:
|
||||
|
||||
- *Outbound intent classification.* The egress addon today scans
|
||||
outbound request content for token patterns. What it lacks is
|
||||
awareness of context — it can't distinguish "agent is pushing a
|
||||
legitimate commit" from "agent was injected and is pushing a backdoor."
|
||||
The `supervise` policy is already the right shape for human-in-the-loop
|
||||
review on sensitive outbound actions; extending it with context from
|
||||
the agent's recent tool calls (what files were touched, what was the
|
||||
triggering task) would narrow the gap.
|
||||
|
||||
- *Semantic code review on git push.* gitleaks is the wrong tool for
|
||||
catching injected logic. A review step on outbound commits — even a
|
||||
simple diff summary surfaced in `cli.py supervise` before the push is
|
||||
forwarded — would close the malicious-commit path without requiring
|
||||
the agent to be fully trusted.
|
||||
|
||||
## Sources
|
||||
|
||||
- [Ask HN: The new wave of AI agent sandboxes? (Mar 2026)](https://news.ycombinator.com/item?id=47444917)
|
||||
- [OK, let's survey how everybody is sandboxing AI coding agents (Mar 2026)](https://news.ycombinator.com/item?id=47185250)
|
||||
- [The agent harness belongs outside the sandbox (May 2026)](https://news.ycombinator.com/item?id=47990675)
|
||||
- [Show HN: Nono – Kernel-enforced sandboxing for AI agents (Feb 2026)](https://news.ycombinator.com/item?id=46849615)
|
||||
- [Kernel-enforced sandbox for AI agents, MCP and LLM workloads (Feb 2026)](https://news.ycombinator.com/item?id=47066574)
|
||||
- [Sandboxes will be left in 2026 (Feb 2026)](https://news.ycombinator.com/item?id=47006445)
|
||||
- [Critical Cursor Flaws / DuneSlide – The Hacker News](https://thehackernews.com/2026/07/critical-cursor-flaws-could-let-prompt.html)
|
||||
- [Agentjacking Attack – The Hacker News](https://thehackernews.com/2026/06/agentjacking-attack-tricks-ai-coding.html)
|
||||
- [Friendly Fire: AI Agents Built to Catch Malicious Code – The Hacker News](https://thehackernews.com/2026/07/friendly-fire-ai-agents-built-to-catch.html)
|
||||
- [Microsoft Warns Poisoned MCP Tool Descriptions – The Hacker News](https://thehackernews.com/2026/06/microsoft-warns-poisoned-mcp-tool.html)
|
||||
- [MCP STDIO Command Injection Advisory – OX Security](https://www.ox.security/blog/mcp-supply-chain-advisory-rce-vulnerabilities-across-the-ai-ecosystem/)
|
||||
- [LiteLLM Vulnerability Chain – The Hacker News](https://thehackernews.com/2026/06/litellm-vulnerability-chain-lets-low.html)
|
||||
- [Red-Teaming the Agentic Red-Team (arXiv 2606.24496)](https://arxiv.org/pdf/2606.24496)
|
||||
@@ -1,182 +0,0 @@
|
||||
# Landscape: containerized AI coding agent tools
|
||||
|
||||
Research into whether bot-bottle is redundant with existing projects, and
|
||||
whether it's worth publishing.
|
||||
|
||||
## Summary
|
||||
|
||||
The "AI coding agents in isolated sandboxes" space is active but not saturated.
|
||||
bot-bottle occupies a distinct position: no surveyed project combines all five
|
||||
of its defining features. Publishing is likely worthwhile, with the main risk
|
||||
being claudebox expanding to absorb the same niche.
|
||||
|
||||
**Updated 2026-07-09:** bot-bottle now supports three isolation backends
|
||||
(Docker, Apple `container`, smolmachines/libkrun microVMs) and three built-in
|
||||
agent providers (Claude Code, OpenAI Codex, Pi) with an open plugin system for
|
||||
arbitrary providers. This meaningfully strengthens the differentiation against
|
||||
all surveyed competitors.
|
||||
|
||||
## Closest competitor: claudebox
|
||||
|
||||
[RchGrav/claudebox](https://github.com/RchGrav/claudebox) is the most
|
||||
feature-complete analog. It runs Claude Code in Docker with per-project
|
||||
isolated images, 15+ pre-configured dev-language profiles, and per-project
|
||||
network firewall allowlists. Actively maintained with multiple forks.
|
||||
|
||||
What it lacks: manifest-driven named agents, per-agent env resolution modes
|
||||
(prompt / host-forward / literal), skill directory injection, per-agent system
|
||||
prompts, SSH-agent forwarding without copying private keys, home+project
|
||||
manifest merge.
|
||||
|
||||
## Other surveyed projects
|
||||
|
||||
- **textcortex/claude-code-sandbox → spritz** — evolved toward
|
||||
Kubernetes-native multi-agent infra; not stdlib-first or local-Docker.
|
||||
Original sandbox repo is archived.
|
||||
- **trailofbits/claude-code-devcontainer** — devcontainer config for security
|
||||
audits; not a general agent launcher.
|
||||
- **Several small solo repos** (arezi/claude-sandbox, nkrefman/claude-sandbox,
|
||||
VishalJ99/claude-docker) — lightweight Docker wrappers with no multi-agent
|
||||
config layer.
|
||||
- **Docker's official sandbox templates** — launch-and-run Dockerfiles plus an
|
||||
npm-based runtime; not a manifest-driven fleet manager.
|
||||
|
||||
## Adjacent (different model)
|
||||
|
||||
- **dagger/container-use** (mid-2025) — exposes an MCP server so the *agent*
|
||||
spins up its own containers with Git worktrees. Inverted model vs. bot-bottle
|
||||
(agent controls container rather than being launched into one by a manifest).
|
||||
Still marked early-development.
|
||||
- **E2B, Northflank, Cloudflare Sandbox SDK** — cloud-hosted SaaS sandbox
|
||||
runtimes; fundamentally different architecture.
|
||||
- **superhq.ai / SuperHQ** (v0.4.4, April 2026) — macOS desktop app (Rust/GPUI)
|
||||
that runs Claude Code, Codex, and Pi inside microVMs via Apple's
|
||||
Virtualization.framework (their own shuru-sdk / libkrun). Auth gateway
|
||||
injects API keys on the wire so the sandbox never sees them; tmpfs overlay
|
||||
stages agent writes for diff-and-accept review; mobile remote access via
|
||||
remote.superhq.ai. Early alpha, free on launch, Apple Silicon only.
|
||||
|
||||
Overlap: both projects cover agent isolation, credential proxying, and
|
||||
multi-provider support (Claude Code / Codex / Pi). Differences: SuperHQ is a
|
||||
GUI desktop app with no manifest layer; bot-bottle is a CLI fleet manager with
|
||||
named agents, skills injection, per-agent system prompts, and cross-platform
|
||||
backends (Docker, Apple `container`, smolmachines). SuperHQ's microVM
|
||||
isolation story is now partially matched by bot-bottle's `macos_container` and
|
||||
smolmachines backends. Worth watching — it targets the same security-minded
|
||||
power-user audience and moves fast.
|
||||
|
||||
**Known gap in SuperHQ (user-requested, as of 2026-07-09):** A named user
|
||||
(Brian Cheong, Founder, Dunialabs.io) explicitly called out the absence of
|
||||
per-run audit logging: tool calls and network egress. Bot-bottle covers both:
|
||||
network egress is logged by pipelock/mitmproxy, and per-run op-log/audit state
|
||||
is persisted to SQLite.
|
||||
|
||||
- **OneCLI** ([onecli.sh](https://onecli.sh/)) — YC-backed, GA, open-source
|
||||
(Apache-2.0, Rust) "identity gateway for AI agents": a credential/secret
|
||||
broker that holds API keys and OAuth tokens out of the agent's reach and
|
||||
injects them at the network layer (phantom-token — the agent sees a
|
||||
placeholder, the gateway swaps in the real, AES-256-GCM-encrypted credential
|
||||
at request time). Framework-agnostic and drop-in for any HTTP-calling agent,
|
||||
50+ app integrations, plus a hosted cloud tier with a per-agent dashboard and
|
||||
audit logs. Full technical breakdown in
|
||||
[`agent-credential-proxy-landscape.md`](agent-credential-proxy-landscape.md).
|
||||
|
||||
**How close a competitor:** near-exact on the *single axis of agent secret
|
||||
custody* — the exact thing bot-bottle sells as "the agent never sees real
|
||||
credentials, even via `printenv`." OneCLI does that one job well, is mature
|
||||
and funded, and is *more portable* (it sits in front of anything; bot-bottle
|
||||
only helps agents launched through bot-bottle). Takeaway: bot-bottle should
|
||||
stop treating secret custody as a *unique* differentiator. But OneCLI is
|
||||
**not** a competitor to bot-bottle's actual product — it does no agent
|
||||
sandboxing (containers/microVMs), no fleet/manifest layer, no named agents /
|
||||
skills / per-agent system prompts, no multi-provider launching, no egress
|
||||
firewall.
|
||||
|
||||
**Our edge:** (1) *Isolation is the product, not a proxy.* OneCLI keeps the
|
||||
key out of reach at the network layer, but the agent itself still runs
|
||||
unsandboxed — a hijacked agent behind OneCLI has full run of its host and can
|
||||
exfil captured data through any allowed host. bot-bottle runs the agent inside
|
||||
a kernel/VM-enforced sandbox, injects credentials across that same
|
||||
out-of-process boundary, *and* clamps egress with pipelock — defense in depth
|
||||
vs. a single network layer. (2) *Fleet + manifest model* with named agents,
|
||||
skills, per-agent system prompts, multi-provider and multi-backend — OneCLI
|
||||
has no equivalent. (3) *Trust posture:* OneCLI's managed tier reintroduces a
|
||||
third-party credential custodian, whereas bot-bottle's OSS-runtime +
|
||||
paid-control-plane split keeps custody inside the operator's own boundary —
|
||||
the stronger story for the security-minded self-hoster. (4) *Runs inside your
|
||||
network boundary — local/internal reach.* Because bot-bottle executes the
|
||||
agent on your own host (homelab, corporate LAN, a Tailnet) and egress is a
|
||||
manifest field, giving an agent *scoped* access to **internal** resources — a
|
||||
private Gitea, a LAN database, a Tailscale node — is just another egress-route
|
||||
line, not a networking project (the same move an operator already makes to
|
||||
reach their Tailscale services). OneCLI's OSS core can self-host too, but it's
|
||||
a credential *broker* for outbound API calls, not an agent runtime, and its
|
||||
managed tier + 50+ integrations are oriented at public SaaS — it doesn't put
|
||||
the agent behind your firewall for you. This is a reach advantage, distinct
|
||||
from the isolation ones above, and it's a wedge cloud-first agent products
|
||||
(Devin, Copilot Workspace, OneCLI Cloud) structurally can't match. **Tactical
|
||||
read:**
|
||||
adopt OneCLI's OSS core for the credential slice if building is undesirable
|
||||
(it's mature now); don't build atop its managed tier (competitor, not
|
||||
dependency); re-position bot-bottle on isolation + fleet + self-hosted custody
|
||||
rather than "we hide your secrets."
|
||||
|
||||
## What no found project does
|
||||
|
||||
None combine:
|
||||
1. Named-agent manifest with per-agent env resolution (prompt / host-forward / literal), supporting multiple providers (Claude Code, Codex, Pi, arbitrary plugins)
|
||||
2. Skills directory injection
|
||||
3. Per-agent system prompts
|
||||
4. SSH-agent key forwarding without copying private keys into the container
|
||||
5. Home + project manifest merge
|
||||
6. Pluggable isolation backends: Docker (Linux/macOS), Apple `container` (macOS microVMs), smolmachines/libkrun microVMs
|
||||
7. Per-run audit log: network egress via pipelock/mitmproxy + op-log persisted to SQLite
|
||||
|
||||
**In-flight directions (not yet shipped):**
|
||||
|
||||
- **Forge-native dispatch (issue #317):** Gitea webhook → orchestrator spins up a bottle
|
||||
with the issue body as prompt → agent works → bottle freezes awaiting review comment →
|
||||
rehydrates on comment → tears down on PR close. The issue-to-PR lifecycle concept is not
|
||||
novel (Devin, Copilot Workspace, SWE-agent all do this as cloud services); what's
|
||||
distinct is doing it self-hosted, manifest-driven, inside bot-bottle's isolation
|
||||
primitives.
|
||||
- **Paid web control plane (issue #327):** Browser-based multi-host agent launch and
|
||||
monitoring; account-scoped bottle and agent definitions; secret custody (encrypted at
|
||||
rest, injected into the sidecar at launch, never exposed to the agent or returned by any
|
||||
read API). Monetization model: OSS runtime free, control plane paid — a standard split
|
||||
(HashiCorp, Grafana) applied to a self-hosted agent sandbox. The principled secret
|
||||
custody model (agent never sees real credentials, even via printenv) is more rigorous
|
||||
than most surveyed tools but not unprecedented.
|
||||
|
||||
## Publishing verdict
|
||||
|
||||
Worth publishing. Differentiators that matter to the target audience (power
|
||||
users running parallel AI coding agent sessions with distinct personas/tooling):
|
||||
|
||||
- The Python-stdlib-first, low-dependency design — competitors are npm-based,
|
||||
Rust/GUI, or Kubernetes-native.
|
||||
- Named agents with distinct skills and system prompts, not just language profiles.
|
||||
- Multi-backend isolation: Docker, Apple `container` microVMs, and
|
||||
smolmachines/libkrun — single manifest works across all three.
|
||||
- Multi-provider: Claude Code, Codex, Pi, plus an open plugin system for
|
||||
arbitrary providers.
|
||||
- SSH forwarding without key copying.
|
||||
- Per-run audit log (tool calls + network egress) — an explicitly requested gap
|
||||
in SuperHQ as of 2026-07-09.
|
||||
- Forge-native dispatch and a paid control plane (in flight) bring bot-bottle
|
||||
into the same product category as cloud services like Devin and Copilot
|
||||
Workspace — but self-hosted, with stronger isolation guarantees and a
|
||||
manifest-driven fleet model those services don't have.
|
||||
|
||||
Main risk: claudebox adds manifest/agent config; SuperHQ is moving fast on the
|
||||
GUI / microVM side. The space is moving fast enough that publishing sooner is
|
||||
better if establishing prior art matters.
|
||||
|
||||
Discovery will be slow without active promotion; an Anthropic Discord post or
|
||||
HN "Show HN" would do most of the work.
|
||||
|
||||
## Caveats
|
||||
|
||||
- GitHub search cannot surface private or very new repos comprehensively.
|
||||
- Counts (stars, forks) were not confirmed for every project.
|
||||
- Initial research conducted 2026-05-07; SuperHQ entry added 2026-07-09; the space moves fast.
|
||||
@@ -0,0 +1,204 @@
|
||||
# Scanning for malicious commits at the git-gate — and whether the semantic layer is a paid feature
|
||||
|
||||
A sandboxed agent's most durable way out isn't live exfiltration — it's the
|
||||
**artifact it pushes**. Code committed through the git-gate leaves the
|
||||
sandbox and runs *later*, in CI or production, with none of the bottle's
|
||||
isolation or egress control. gitleaks covers *secrets* on that channel
|
||||
(see [`git-gate` pre-receive](../../bot_bottle/git_gate_render.py)); this
|
||||
note asks the harder question — can we scan for *malicious code* in
|
||||
commits — and the product question the answer raises: is the semantic
|
||||
(LLM) layer a decent paid feature on "git egress," alongside a paid
|
||||
human-review console for the supervise flow?
|
||||
|
||||
## Summary
|
||||
|
||||
- **No scanner reliably detects malicious code.** It's undecidable in
|
||||
general and adversarial in practice — a capable agent hides a backdoor
|
||||
from any pattern scanner (and from human reviewers). So the frame is not
|
||||
"detect malice"; it's **raise the cost, cover the common/obvious cases,
|
||||
and put a human on the high-leverage ones.**
|
||||
- **Layered, ranked by ROI:** (1) supply-chain / malicious-dependency
|
||||
scanning (Socket, OSV-Scanner, GuardDog) — highest ROI, best git fit;
|
||||
(2) fast heuristic / obfuscation / dangerous-sink scanning
|
||||
(Semgrep-on-diff) — gitleaks-shaped; (3) **risk-based human gating** via
|
||||
the existing supervise plane on high-risk file classes; (4) **LLM
|
||||
semantic diff-review** as a best-effort backstop for novel backdoors.
|
||||
- **Architecture:** the git-gate's `pre-receive` is *synchronous* (it
|
||||
blocks the push), so run only fast scanners inline and push heavy
|
||||
analysis (CodeQL, full LLM review) to an **async post-push CI lane** that
|
||||
can't hold the developer hostage.
|
||||
- **Monetization verdict: yes — but the paid unit is the *governed
|
||||
git-egress review bundle*, not the raw scanner.** The pre-receive
|
||||
mechanics and the fast OSS scanners stay free (open-core social
|
||||
contract, same as gitleaks). The sell is the control-plane workflow
|
||||
around them: **managed semantic review + a web-console human-review flow
|
||||
+ RBAC + audit retention + cross-run policy.** A bare per-push scan is
|
||||
borderline "single-run = free" by our own heuristic, and its detection
|
||||
accuracy is not something to sell on. Sell the *workflow and the
|
||||
governance*, not the detector.
|
||||
|
||||
## The threat: temporal escape
|
||||
|
||||
This is the one containment fundamentally can't close. bot-bottle's VM
|
||||
isolation + egress DLP shrink *live* exfiltration, but the git push
|
||||
channel launders a payload past all of it: the agent commits code that
|
||||
looks benign at review time and does its work later, outside the
|
||||
perimeter — the "blind-XSS"-style temporal escape that both the "3 reasons
|
||||
sandboxing won't secure your AI agent" talk and Embroidery's "In sandboxes
|
||||
we (shouldn't) trust" call the sharpest gap. gitleaks gates *secrets* on
|
||||
this channel; malicious *code* is the uncovered surface, and it's the
|
||||
behavioral half of the isolation-vs-behavioral split noted in
|
||||
[`agent-sandbox-landscape.md`](agent-sandbox-landscape.md).
|
||||
|
||||
## Why it's fundamentally hard
|
||||
|
||||
Distinguishing malicious from legitimate code is undecidable in general
|
||||
and adversarial in the specific. The underhanded-code contests show humans
|
||||
hide backdoors from *human reviewers*; a scanner reduces to regex/dataflow
|
||||
heuristics a motivated agent writes around. So "catch malicious commits"
|
||||
is the wrong success metric. The right ones: *how much did we raise the
|
||||
cost*, *did we cover the common and the obvious*, and *did a human see the
|
||||
dangerous changes before they left*.
|
||||
|
||||
## Landscape of git-pluggable approaches (ranked by ROI)
|
||||
|
||||
1. **Supply-chain / malicious-dependency scanning — highest ROI.** The
|
||||
most likely agent-introduced threat is a malicious or typosquatted
|
||||
*dependency*, not a hand-written backdoor — and it's far more
|
||||
detectable. **Socket** (behavioral analysis of packages: install
|
||||
scripts, obfuscation, unexpected network/fs access, typosquats) is the
|
||||
closest thing to malicious-intent detection that actually works;
|
||||
**OSV-Scanner** (known-bad DB, fast) and **GuardDog** (PyPI/npm
|
||||
heuristics) complement it. Trigger only when a commit changes a
|
||||
lockfile/manifest → block or escalate.
|
||||
2. **Fast heuristic / obfuscation / dangerous-sink scanning.** **Semgrep**
|
||||
with a curated ruleset, scanning the *diff* (fast enough for
|
||||
pre-receive). Catches obfuscation (`eval(atob(...))`, base64/hex blobs,
|
||||
minified source), exfil primitives (`curl | sh`, reverse shells,
|
||||
DNS-exfil), and dangerous sinks (`shell=True`, `pickle.loads`, unsafe
|
||||
`yaml.load`, `child_process.exec`). Tune noisy rules to *escalate*, not
|
||||
block.
|
||||
3. **Risk-based human gating — the most robust control, already
|
||||
half-built.** The git-gate already escalates `# gitleaks:allow`
|
||||
findings to a supervisor proposal and blocks the push pending approval
|
||||
(`supervise_gitleaks_allow`, [`git_gate_render.py`](../../bot_bottle/git_gate_render.py)).
|
||||
Extend the same flow to **high-risk file classes**: any commit touching
|
||||
CI/build/deploy scripts, auth/crypto code, egress config, or
|
||||
adding/changing dependencies → route to `./cli.py supervise`. This is
|
||||
attribution/policy, not detection, and it's the strongest thing here —
|
||||
a human on exactly the temporal-escape surfaces.
|
||||
4. **LLM semantic diff-review — the behavioral backstop.** The only
|
||||
approach with a shot at a novel, benign-looking backdoor: a model
|
||||
reviews the diff for a backdoor / exfiltration / obfuscated payload /
|
||||
dangerous behavior. This is the behavioral-detection thesis applied to
|
||||
the *artifact* (the diff) rather than the agent's reasoning. Real
|
||||
caveats: false positives and negatives, latency/cost, it can be
|
||||
prompt-injected by hostile content *in the diff itself*, and it invites
|
||||
over-trust. Useful, never a guarantee.
|
||||
|
||||
**Not for this job:** CodeQL, Trivy, Grype, Bandit. They find *known
|
||||
vulns and insecure patterns* (bugs), not deliberate backdoors, and the
|
||||
powerful ones (CodeQL taint) need a build + database — too heavy for a
|
||||
synchronous gate. They belong in the async CI lane if at all.
|
||||
|
||||
## Fit into bot-bottle's git-gate
|
||||
|
||||
The `pre-receive` hook today is: gitleaks-scan each ref → escalate
|
||||
`# gitleaks:allow` findings to supervise → forward to upstream
|
||||
([`git_gate_render.py`](../../bot_bottle/git_gate_render.py)). The
|
||||
additions slot in cleanly:
|
||||
|
||||
- **Inline (fast), before forward:** a dep-scan phase (on manifest/lockfile
|
||||
change) and a Semgrep-diff phase. Findings block or open a supervise
|
||||
proposal, same shape as gitleaks.
|
||||
- **New supervise tool types** alongside the existing
|
||||
`egress-block/allow`, `gitleaks-allow`, `egress-token-allow`
|
||||
([`supervise_types.py`](../../bot_bottle/supervise_types.py)) — e.g. a
|
||||
`commit-review` proposal for risky-file-class gating and for semantic
|
||||
review. The supervise plane is already the right abstraction; this is
|
||||
another *producer* feeding it, and [`supervise_server.py`](../../bot_bottle/supervise_server.py)
|
||||
(JSON-RPC) is already the console backend.
|
||||
- **Async lane (heavy):** full LLM review + any CodeQL run out of band
|
||||
after the push, feeding the same review/audit surface, so the
|
||||
synchronous gate stays fast.
|
||||
|
||||
## The product question: paid feature on git egress?
|
||||
|
||||
Restating the open-core line bot-bottle runs on: *give away the
|
||||
sandbox/runtime, charge for the control plane; single-run/single-node =
|
||||
free, cross-run aggregation + central enforcement + identity/fleet = paid;
|
||||
the moat is uniform egress audit + secret custody + policy across
|
||||
untrusted agents.*
|
||||
|
||||
Against that line, the split is clean:
|
||||
|
||||
**Free (OSS runtime — the trust funnel):**
|
||||
- the `pre-receive` gate mechanics and gitleaks;
|
||||
- wiring the OSS scanners (Socket CLI / OSV-Scanner / Semgrep);
|
||||
- the CLI supervise flow.
|
||||
Keeping the raw scanners free is the same social contract as gitleaks and
|
||||
preserves the bottom-up distribution funnel.
|
||||
|
||||
**Paid (the governed git-egress bundle — the control plane):**
|
||||
- **Managed semantic diff-review** — hosted inference + a curated,
|
||||
maintained malicious-pattern/policy set. This is *capability* (metered),
|
||||
not *insurance* — the thing individuals actually pay for. Position it as
|
||||
**governed code-egress review**, not "we resell inference" (the
|
||||
monetization notes explicitly warn against reselling compute).
|
||||
- **The web-console supervise/review flow — the strongest anchor.** Turn
|
||||
the CLI `./cli.py supervise` approval into a real review surface:
|
||||
rendered diff + finding context, approve/reject, **who-approved audit
|
||||
trail, RBAC on approvers, mobile/phone-control** (ties to the
|
||||
dashboard/vault north star). This is "central enforcement +
|
||||
identity/fleet = paid" almost verbatim — and it generalizes across
|
||||
*every* supervise proposal (egress block/allow, gitleaks-allow,
|
||||
commit-review), so it's worth building for the whole plane, with the
|
||||
semantic check as one producer.
|
||||
- **Cross-run governance:** fleet-wide policy for what escalates,
|
||||
review-decision history/search/export, and drift alerts.
|
||||
|
||||
**Why it fits the moat rather than bolting on:** a git push *is* an egress
|
||||
channel. A semantic review + human approval + audit on it extends the
|
||||
uniform "egress audit + custody + policy across untrusted agents" wedge to
|
||||
**code artifacts** — the same product, applied to the one channel gitleaks
|
||||
only half-covers. That's on-moat, not a detour.
|
||||
|
||||
**The honest nuance (don't oversell):** a bare per-push LLM scan is
|
||||
arguably *free* by the single-run heuristic, and its detection accuracy is
|
||||
not defensible to charge for. The paid value is the **governance around
|
||||
it** — the console, RBAC, audit retention, cross-run policy — plus the
|
||||
managed capability. Sell the *review-and-approve-and-audit workflow*; let
|
||||
the detector be explicitly best-effort. And per the monetization
|
||||
guardrail, the "anti-corporate" free crowd must not veto these team
|
||||
features: the review console + RBAC + audit *are* the monetization.
|
||||
|
||||
## Recommendation
|
||||
|
||||
1. **Land the free layer first.** Add the dep-scan and Semgrep-diff phases
|
||||
to `pre-receive`, and extend supervise to risky-file-class gating —
|
||||
reuses existing machinery, immediate value, stays OSS.
|
||||
2. **Build the supervise web console** over `supervise_server`'s JSON-RPC
|
||||
(already the Phase-1 move in the monetization path). This is the paid
|
||||
anchor and it serves *all* proposal types, not just commit review.
|
||||
3. **Add managed semantic diff-review as a paid producer** feeding that
|
||||
console — "governed code-egress review," metered, explicitly
|
||||
best-effort on detection.
|
||||
4. **Don't oversell detection.** Market the workflow (review + approve +
|
||||
audit) and the cross-run policy/RBAC, where the value is real and
|
||||
defensible; keep the raw scanners open.
|
||||
|
||||
## Sources / references
|
||||
|
||||
- [`agent-sandbox-landscape.md`](agent-sandbox-landscape.md) — the
|
||||
egress-DLP gap and isolation-vs-behavioral framing.
|
||||
- Git-gate internals: [`git_gate_render.py`](../../bot_bottle/git_gate_render.py),
|
||||
[`supervise_types.py`](../../bot_bottle/supervise_types.py),
|
||||
[`supervise_server.py`](../../bot_bottle/supervise_server.py).
|
||||
- External tools: Socket (socket.dev), OSV-Scanner (google/osv-scanner),
|
||||
GuardDog (DataDog/guarddog), Semgrep (semgrep/semgrep).
|
||||
- Threat framing: "3 reasons sandboxing won't secure your AI agent"
|
||||
(youtube TsYDazwHJ6U); Embroidery, "In sandboxes we (shouldn't) trust."
|
||||
- The authoritative monetization/positioning analysis (the open-core line,
|
||||
the wedge, single-run-free/cross-run-paid) lives in the **separate
|
||||
`bot-bottle-console` repo**, not this one — cited here from memory, not
|
||||
linked.
|
||||
@@ -1,6 +1,6 @@
|
||||
# smolmachines as a VM backend for bot-bottle
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
Evaluation of whether [smolmachines](https://smolmachines.com/) would
|
||||
simplify the macOS agent-VM-isolation work spelled out in
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[build-system]
|
||||
requires = ["setuptools>=68"]
|
||||
build-backend = "setuptools.backends.legacy:build"
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "bot-bottle"
|
||||
|
||||
+32
-10
@@ -1,15 +1,19 @@
|
||||
#!/usr/bin/env bash
|
||||
# Combined unit + integration coverage (see docs/decisions/0004-coverage-policy.md).
|
||||
#
|
||||
# Runs the unit suite, then appends the integration suite (which skips
|
||||
# cleanly when Docker / the backend CLIs are unavailable), and prints one
|
||||
# combined report. The integration suite is what scores the subprocess /
|
||||
# backend orchestration modules, so the number here is the policy's
|
||||
# yardstick — not the unit-only badge.
|
||||
# Two modes:
|
||||
#
|
||||
# Usage:
|
||||
# scripts/coverage.sh # combined report
|
||||
# scripts/coverage.sh critical # also report just the critical modules
|
||||
# scripts/coverage.sh [critical]
|
||||
# Run mode (default, for local dev): executes the unit suite then the
|
||||
# integration suite under coverage and prints a combined report.
|
||||
#
|
||||
# scripts/coverage.sh aggregate [critical]
|
||||
# Aggregate mode (used by CI): combines pre-existing .coverage.* files
|
||||
# produced by individual test jobs and prints a combined report. No tests
|
||||
# are re-executed; no KVM or Docker dependency.
|
||||
#
|
||||
# Pass "critical" as the last argument in either mode to also report just the
|
||||
# critical modules (ADR 0004 target: 90%).
|
||||
set -euo pipefail
|
||||
|
||||
cd "$(dirname "$0")/.."
|
||||
@@ -21,13 +25,31 @@ PY="${PYTHON:-python3}"
|
||||
# README "core coverage" badge can't drift; comma-join it for --include.
|
||||
CRITICAL=$(grep -vE '^[[:space:]]*(#|$)' scripts/critical-modules.txt | paste -sd, -)
|
||||
|
||||
if [ "${1:-}" = "aggregate" ]; then
|
||||
# Aggregate mode: combine .coverage.* artifacts already in the workspace.
|
||||
echo "== combining coverage artifacts ==" >&2
|
||||
"$PY" -m coverage combine
|
||||
|
||||
echo "== combined report ==" >&2
|
||||
"$PY" -m coverage report -m
|
||||
|
||||
if [ "${2:-}" = "critical" ]; then
|
||||
echo "== critical modules (ADR 0004 target: 90%) ==" >&2
|
||||
"$PY" -m coverage report --include="$CRITICAL"
|
||||
fi
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Run mode (default): execute both suites under coverage in this process.
|
||||
rm -f .coverage
|
||||
|
||||
echo "== unit ==" >&2
|
||||
"$PY" -m coverage run -m unittest discover -t . -s tests/unit
|
||||
|
||||
echo "== integration (skips without Docker) ==" >&2
|
||||
"$PY" -m coverage run --append -m unittest discover -t . -s tests/integration
|
||||
echo "== integration (firecracker; skips docker tests) ==" >&2
|
||||
BOT_BOTTLE_BACKEND=firecracker SKIP_DOCKER_TESTS=1 \
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_DIR="${BOT_BOTTLE_CI_INFRA_ARTIFACT_DIR:-}" \
|
||||
"$PY" -m coverage run --append -m unittest discover -t . -s tests/integration
|
||||
|
||||
echo "== combined report ==" >&2
|
||||
"$PY" -m coverage report -m
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Enforce the repository's issue/PR metadata policy in Gitea Actions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
ISSUE_REFERENCE = re.compile(
|
||||
r"(?im)\b(?:close[sd]?|fix(?:e[sd])?|resolve[sd]?|part\s+of|"
|
||||
r"related\s+to|refs?|references)\s+#(\d+)\b"
|
||||
)
|
||||
TRIAGE_LABEL = "Status/Needs Triage"
|
||||
|
||||
|
||||
def deliberate_issue_numbers(title: str, body: str) -> set[int]:
|
||||
"""Return same-repository issue numbers referenced intentionally."""
|
||||
return {int(match) for match in ISSUE_REFERENCE.findall(f"{title}\n{body}")}
|
||||
|
||||
|
||||
class GiteaApi:
|
||||
"""Small API client using the Actions-provided repository token."""
|
||||
|
||||
def __init__(self, api_url: str, repository: str, token: str) -> None:
|
||||
self.base = f"{api_url.rstrip('/')}/repos/{repository}"
|
||||
self.token = token
|
||||
|
||||
def request(self, method: str, path: str, payload: object | None = None) -> Any:
|
||||
data = None if payload is None else json.dumps(payload).encode()
|
||||
request = urllib.request.Request(
|
||||
f"{self.base}{path}",
|
||||
data=data,
|
||||
method=method,
|
||||
headers={
|
||||
"Authorization": f"token {self.token}",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=15) as response:
|
||||
if response.status == 204:
|
||||
return None
|
||||
return json.load(response)
|
||||
|
||||
|
||||
def check_pull_request(event: dict[str, Any], api: GiteaApi) -> list[str]:
|
||||
"""Return policy violations for a pull_request event."""
|
||||
pull = event["pull_request"]
|
||||
errors: list[str] = []
|
||||
labels = pull.get("labels") or []
|
||||
if labels:
|
||||
errors.append(
|
||||
"PRs must be unlabeled; put tracker metadata on the linked issue "
|
||||
f"(found: {', '.join(label['name'] for label in labels)})."
|
||||
)
|
||||
|
||||
numbers = deliberate_issue_numbers(pull.get("title", ""), pull.get("body", ""))
|
||||
if not numbers:
|
||||
errors.append(
|
||||
"PR must reference an issue with Closes/Fixes/Resolves #N, "
|
||||
"Part of #N, Related to #N, Refs #N, or References #N."
|
||||
)
|
||||
return errors
|
||||
|
||||
real_issues = 0
|
||||
for number in sorted(numbers):
|
||||
try:
|
||||
item = api.request("GET", f"/issues/{number}")
|
||||
except urllib.error.HTTPError as error:
|
||||
if error.code == 404:
|
||||
errors.append(f"Referenced issue #{number} does not exist.")
|
||||
continue
|
||||
raise
|
||||
if item.get("pull_request") is not None:
|
||||
errors.append(f"#{number} is a pull request, not an issue.")
|
||||
else:
|
||||
real_issues += 1
|
||||
|
||||
if not real_issues and not errors:
|
||||
errors.append("PR must reference at least one real issue.")
|
||||
return errors
|
||||
|
||||
|
||||
def ensure_issue_label(event: dict[str, Any], api: GiteaApi) -> bool:
|
||||
"""Apply the triage label if an issue event leaves the issue unlabeled."""
|
||||
issue = event["issue"]
|
||||
if issue.get("pull_request") is not None or issue.get("labels"):
|
||||
return False
|
||||
labels = api.request("GET", "/labels?limit=100")
|
||||
triage = next((label for label in labels if label["name"] == TRIAGE_LABEL), None)
|
||||
if triage is None:
|
||||
raise RuntimeError(f"repository label {TRIAGE_LABEL!r} does not exist")
|
||||
api.request("POST", f"/issues/{issue['number']}/labels", {"labels": [triage["id"]]})
|
||||
return True
|
||||
|
||||
|
||||
def _load_event(path: str) -> dict[str, Any]:
|
||||
return json.loads(Path(path).read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("command", choices=("check-pr", "label-issue"))
|
||||
parser.add_argument("--event", default=os.environ.get("GITHUB_EVENT_PATH"))
|
||||
args = parser.parse_args()
|
||||
if not args.event:
|
||||
parser.error("--event or GITHUB_EVENT_PATH is required")
|
||||
|
||||
api = GiteaApi(
|
||||
os.environ["GITHUB_API_URL"],
|
||||
os.environ["GITHUB_REPOSITORY"],
|
||||
os.environ["GITHUB_TOKEN"],
|
||||
)
|
||||
event = _load_event(args.event)
|
||||
if args.command == "check-pr":
|
||||
errors = check_pull_request(event, api)
|
||||
if errors:
|
||||
print("\n".join(f"::error::{error}" for error in errors))
|
||||
return 1
|
||||
print("PR tracker policy passed.")
|
||||
return 0
|
||||
|
||||
changed = ensure_issue_label(event, api)
|
||||
print(f"Applied {TRIAGE_LABEL}." if changed else "Issue already has a label.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Resolved paths to system binaries used by subprocess-based tests.
|
||||
|
||||
NixOS and other non-FHS hosts don't populate ``/bin`` (there is no
|
||||
``/bin/sleep``), so tests that spawn real short-lived helper processes
|
||||
must resolve the binary from ``PATH`` rather than hardcoding an FHS path.
|
||||
Import the resolved constant (e.g. ``SLEEP``) instead of writing
|
||||
``/bin/sleep`` inline.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
|
||||
|
||||
def resolve(name: str, fallback: str) -> str:
|
||||
"""Absolute path to ``name`` from ``PATH``; ``fallback`` on FHS hosts
|
||||
where the binary isn't on ``PATH`` but lives at a known ``/bin`` path."""
|
||||
return shutil.which(name) or fallback
|
||||
|
||||
|
||||
# Real ``sleep`` binary. ``/bin/sleep`` is absent on NixOS; resolve from
|
||||
# PATH so subprocess tests run instead of erroring with FileNotFoundError.
|
||||
SLEEP = resolve("sleep", "/bin/sleep")
|
||||
+29
-9
@@ -2,24 +2,44 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import unittest
|
||||
|
||||
|
||||
def docker_available() -> bool:
|
||||
if os.environ.get("SKIP_DOCKER_TESTS"):
|
||||
return False
|
||||
if shutil.which("docker") is None:
|
||||
return False
|
||||
return (
|
||||
subprocess.run(
|
||||
["docker", "info"],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
check=False,
|
||||
).returncode
|
||||
== 0
|
||||
)
|
||||
try:
|
||||
return (
|
||||
subprocess.run(
|
||||
["docker", "info"],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
check=False,
|
||||
timeout=5,
|
||||
).returncode
|
||||
== 0
|
||||
)
|
||||
except subprocess.TimeoutExpired:
|
||||
return False
|
||||
|
||||
|
||||
def skip_unless_docker(reason: str = "docker unreachable"):
|
||||
return unittest.skipUnless(docker_available(), reason)
|
||||
|
||||
|
||||
def skip_unless_docker_or_firecracker(
|
||||
reason: str = "neither Docker nor Firecracker selected",
|
||||
):
|
||||
"""Skip a backend-agnostic test unless one supported backend can run.
|
||||
|
||||
Firecracker does not require the host Docker daemon. The KVM coverage job
|
||||
deliberately sets ``SKIP_DOCKER_TESTS`` to exclude Docker-only integration
|
||||
classes while still exercising this path.
|
||||
"""
|
||||
firecracker_selected = os.environ.get("BOT_BOTTLE_BACKEND") == "firecracker"
|
||||
return unittest.skipUnless(firecracker_selected or docker_available(), reason)
|
||||
|
||||
@@ -86,13 +86,12 @@ class TestGatewayImage(unittest.TestCase):
|
||||
self.assertIn("Mitmproxy", out)
|
||||
|
||||
def test_python_imports_supervise_module(self):
|
||||
# The bundle's supervise daemon imports `supervise` as a
|
||||
# same-directory sibling of `supervise_server`. Probe the
|
||||
# import resolves with `python3 -c` from /app (the
|
||||
# Dockerfile's WORKDIR).
|
||||
# The supervise daemon is installed as part of the bot_bottle
|
||||
# package (5ad3449), not as flat sibling modules under /app.
|
||||
# Probe that the package imports resolve inside the image.
|
||||
rc, out = self._run_in_image(
|
||||
"python3", "-c",
|
||||
"import supervise; import supervise_server; print('ok')",
|
||||
"from bot_bottle import supervise, supervise_server; print('ok')",
|
||||
)
|
||||
self.assertEqual(0, rc, msg=out)
|
||||
self.assertIn("ok", out)
|
||||
|
||||
@@ -31,7 +31,7 @@ from pathlib import Path
|
||||
from bot_bottle.backend import BottleSpec, get_bottle_backend
|
||||
from bot_bottle.bottle_state import cleanup_state
|
||||
from bot_bottle.manifest import ManifestIndex
|
||||
from tests._docker import skip_unless_docker
|
||||
from tests._docker import skip_unless_docker_or_firecracker
|
||||
|
||||
|
||||
# Secrets planted in the bottle env as literals (agents substitute via
|
||||
@@ -67,13 +67,14 @@ _DUMMY_HOST_KEY = (
|
||||
)
|
||||
|
||||
|
||||
@skip_unless_docker()
|
||||
@skip_unless_docker_or_firecracker()
|
||||
@unittest.skipIf(
|
||||
os.environ.get("GITEA_ACTIONS") == "true",
|
||||
"skipped under act_runner: egress_tls_init uses a host bind mount "
|
||||
"the runner container can't see, and the network topology hides "
|
||||
"sibling-gateway visibility — same constraint as the other "
|
||||
"bottle-bringup integration tests",
|
||||
os.environ.get("GITEA_ACTIONS") == "true"
|
||||
and os.environ.get("BOT_BOTTLE_BACKEND") != "firecracker",
|
||||
"skipped under act_runner unless BOT_BOTTLE_BACKEND=firecracker: "
|
||||
"egress_tls_init uses a host bind mount the runner container can't "
|
||||
"see, and the network topology hides sibling-gateway visibility — "
|
||||
"these constraints don't apply on the self-hosted KVM runner",
|
||||
)
|
||||
class TestSandboxEscape(unittest.TestCase):
|
||||
"""End-to-end attacks against a real bottle. The bottle stays
|
||||
@@ -90,11 +91,9 @@ class TestSandboxEscape(unittest.TestCase):
|
||||
|
||||
@classmethod
|
||||
def setUpClass(cls) -> None:
|
||||
# Docker is always required (the agent + companion containers run under it,
|
||||
# and VM backends still use it for the gateway); the
|
||||
# class-level @skip_unless_docker already covers that. Pin
|
||||
# Docker when BOT_BOTTLE_BACKEND is unset to preserve the
|
||||
# Docker-backed CI path.
|
||||
# Pin Docker when BOT_BOTTLE_BACKEND is unset to preserve the
|
||||
# Docker-backed CI path. Firecracker uses its persistent infra VM for
|
||||
# the shared gateway and therefore does not require host Docker.
|
||||
cls._backend_name = os.environ.get("BOT_BOTTLE_BACKEND", "docker")
|
||||
|
||||
# Throwaway static key for the git-gate fixture. It need not
|
||||
|
||||
@@ -9,15 +9,11 @@ import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from bot_bottle.agent_provider import (
|
||||
CLAUDE_HOST_CREDENTIAL_HOSTS,
|
||||
CODEX_HOST_CREDENTIAL_HOSTS,
|
||||
build_agent_provision_plan,
|
||||
prompt_args,
|
||||
)
|
||||
from bot_bottle.egress import (
|
||||
CLAUDE_HOST_CREDENTIAL_TOKEN_REF,
|
||||
CODEX_HOST_CREDENTIAL_TOKEN_REF,
|
||||
)
|
||||
from bot_bottle.egress import CODEX_HOST_CREDENTIAL_TOKEN_REF
|
||||
|
||||
|
||||
def _jwt(exp: int) -> str:
|
||||
@@ -296,67 +292,6 @@ class TestAgentProviderRuntime(unittest.TestCase):
|
||||
)
|
||||
self.assertEqual({}, plan.provisioned_env)
|
||||
|
||||
def test_claude_forward_host_credentials_populates_egress_route(self):
|
||||
access_token = "sk-ant-oat01-test-key" # gitleaks:allow
|
||||
with tempfile.TemporaryDirectory(prefix="bb-provider.") as tmp:
|
||||
home = Path(tmp) / "host-claude"
|
||||
cred_dir = home / ".claude"
|
||||
cred_dir.mkdir(parents=True)
|
||||
(cred_dir / ".credentials.json").write_text(json.dumps({
|
||||
"claudeAiOauth": {"accessToken": access_token},
|
||||
}))
|
||||
plan = build_agent_provision_plan(
|
||||
template="claude",
|
||||
dockerfile="",
|
||||
state_dir=Path(tmp),
|
||||
instance_name="bot-bottle-test",
|
||||
prompt_file=Path(tmp) / "prompt.txt",
|
||||
forward_host_credentials=True,
|
||||
host_env={"HOME": str(home)},
|
||||
)
|
||||
self.assertEqual(1, len(plan.egress_routes))
|
||||
route = plan.egress_routes[0]
|
||||
self.assertIn(route.host, CLAUDE_HOST_CREDENTIAL_HOSTS)
|
||||
self.assertEqual("Bearer", route.auth_scheme)
|
||||
self.assertEqual(CLAUDE_HOST_CREDENTIAL_TOKEN_REF, route.token_ref)
|
||||
self.assertEqual("egress-placeholder", plan.env_vars["CLAUDE_CODE_OAUTH_TOKEN"])
|
||||
self.assertEqual(frozenset({"CLAUDE_CODE_OAUTH_TOKEN"}), plan.hidden_env_names)
|
||||
|
||||
def test_claude_forward_host_credentials_populates_provisioned_env(self):
|
||||
access_token = "sk-ant-oat01-test-key" # gitleaks:allow
|
||||
with tempfile.TemporaryDirectory(prefix="bb-provider.") as tmp:
|
||||
home = Path(tmp) / "host-claude"
|
||||
cred_dir = home / ".claude"
|
||||
cred_dir.mkdir(parents=True)
|
||||
(cred_dir / ".credentials.json").write_text(json.dumps({
|
||||
"claudeAiOauth": {"accessToken": access_token},
|
||||
}))
|
||||
plan = build_agent_provision_plan(
|
||||
template="claude",
|
||||
dockerfile="",
|
||||
state_dir=Path(tmp),
|
||||
instance_name="bot-bottle-test",
|
||||
prompt_file=Path(tmp) / "prompt.txt",
|
||||
forward_host_credentials=True,
|
||||
host_env={"HOME": str(home)},
|
||||
)
|
||||
self.assertEqual(
|
||||
{CLAUDE_HOST_CREDENTIAL_TOKEN_REF: access_token},
|
||||
plan.provisioned_env,
|
||||
)
|
||||
|
||||
def test_claude_without_forward_host_credentials_has_empty_provisioned_env(self):
|
||||
with tempfile.TemporaryDirectory(prefix="bb-provider.") as tmp:
|
||||
plan = build_agent_provision_plan(
|
||||
template="claude",
|
||||
dockerfile="",
|
||||
state_dir=Path(tmp),
|
||||
instance_name="bot-bottle-test",
|
||||
prompt_file=Path(tmp) / "prompt.txt",
|
||||
forward_host_credentials=False,
|
||||
)
|
||||
self.assertEqual({}, plan.provisioned_env)
|
||||
|
||||
def test_pi_plan_writes_default_ollama_models(self):
|
||||
with tempfile.TemporaryDirectory(prefix="bb-provider.") as tmp:
|
||||
plan = build_agent_provision_plan(
|
||||
|
||||
@@ -57,14 +57,16 @@ class TestGetBottleBackend(unittest.TestCase):
|
||||
return self._available
|
||||
|
||||
# No macOS container and the host can't run firecracker (no
|
||||
# KVM / not Linux) → docker is the last resort.
|
||||
# KVM / not Linux) → docker fallback with a prompt. Simulate
|
||||
# the user picking "d" (use docker anyway).
|
||||
with patch.dict(os.environ, {}, clear=True), \
|
||||
patch.object(backend_mod.FirecrackerBottleBackend,
|
||||
"is_host_capable", classmethod(lambda cls: False)), \
|
||||
patch.object(backend_mod, "_backends", {
|
||||
"macos-container": _FakeBackend("macos-container", False),
|
||||
"docker": _FakeBackend("docker", True),
|
||||
}):
|
||||
}), \
|
||||
patch.object(backend_mod, "read_tty_line", return_value="d"):
|
||||
b = get_bottle_backend()
|
||||
self.assertEqual("docker", b.name)
|
||||
|
||||
@@ -96,6 +98,145 @@ class TestGetBottleBackend(unittest.TestCase):
|
||||
with self.assertRaises(SystemExit):
|
||||
get_bottle_backend("nonexistent")
|
||||
|
||||
def test_no_backend_available_dies(self):
|
||||
# No VM and no docker → print install instructions and die.
|
||||
class _FakeBackend:
|
||||
def __init__(self, name: str, available: bool) -> None:
|
||||
self.name = name
|
||||
self._available = available
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return self._available
|
||||
|
||||
with patch.dict(os.environ, {}, clear=True), \
|
||||
patch.object(backend_mod.FirecrackerBottleBackend,
|
||||
"is_host_capable", classmethod(lambda cls: False)), \
|
||||
patch.object(backend_mod, "_backends", {
|
||||
"macos-container": _FakeBackend("macos-container", False),
|
||||
"docker": _FakeBackend("docker", False),
|
||||
}), \
|
||||
patch.object(backend_mod, "die", side_effect=SystemExit("die")):
|
||||
with self.assertRaises(SystemExit):
|
||||
get_bottle_backend()
|
||||
|
||||
def test_docker_fallback_user_quits(self):
|
||||
# VM unavailable, docker available, user picks "q" → die.
|
||||
class _FakeBackend:
|
||||
def __init__(self, name: str, available: bool) -> None:
|
||||
self.name = name
|
||||
self._available = available
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return self._available
|
||||
|
||||
with patch.dict(os.environ, {}, clear=True), \
|
||||
patch.object(backend_mod.FirecrackerBottleBackend,
|
||||
"is_host_capable", classmethod(lambda cls: False)), \
|
||||
patch.object(backend_mod, "_backends", {
|
||||
"macos-container": _FakeBackend("macos-container", False),
|
||||
"docker": _FakeBackend("docker", True),
|
||||
}), \
|
||||
patch.object(backend_mod, "read_tty_line", return_value="q"), \
|
||||
patch.object(backend_mod, "die", side_effect=SystemExit("die")):
|
||||
with self.assertRaises(SystemExit):
|
||||
get_bottle_backend()
|
||||
|
||||
|
||||
def test_docker_fallback_non_interactive_dies(self):
|
||||
# prompt=False: headless/CI contexts must not block on a TTY read.
|
||||
class _FakeBackend:
|
||||
def __init__(self, name: str, available: bool) -> None:
|
||||
self.name = name
|
||||
self._available = available
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return self._available
|
||||
|
||||
with patch.dict(os.environ, {}, clear=True), \
|
||||
patch.object(backend_mod.FirecrackerBottleBackend,
|
||||
"is_host_capable", classmethod(lambda cls: False)), \
|
||||
patch.object(backend_mod, "_backends", {
|
||||
"macos-container": _FakeBackend("macos-container", False),
|
||||
"docker": _FakeBackend("docker", True),
|
||||
}), \
|
||||
patch.object(backend_mod, "die", side_effect=SystemExit("die")):
|
||||
with self.assertRaises(SystemExit):
|
||||
get_bottle_backend(prompt=False)
|
||||
|
||||
def test_docker_fallback_user_picks_install(self):
|
||||
# User picks [i] → print install instructions then die.
|
||||
class _FakeBackend:
|
||||
def __init__(self, name: str, available: bool) -> None:
|
||||
self.name = name
|
||||
self._available = available
|
||||
|
||||
def is_available(self) -> bool:
|
||||
return self._available
|
||||
|
||||
with patch.dict(os.environ, {}, clear=True), \
|
||||
patch.object(backend_mod.FirecrackerBottleBackend,
|
||||
"is_host_capable", classmethod(lambda cls: False)), \
|
||||
patch.object(backend_mod, "_backends", {
|
||||
"macos-container": _FakeBackend("macos-container", False),
|
||||
"docker": _FakeBackend("docker", True),
|
||||
}), \
|
||||
patch.object(backend_mod, "read_tty_line", return_value="i"), \
|
||||
patch.object(backend_mod, "_print_vm_install_instructions") as mock_inst, \
|
||||
patch.object(backend_mod, "die", side_effect=SystemExit("die")):
|
||||
with self.assertRaises(SystemExit):
|
||||
get_bottle_backend()
|
||||
mock_inst.assert_called_once()
|
||||
|
||||
|
||||
class TestReadTtyLine(unittest.TestCase):
|
||||
"""Unit tests for the shared read_tty_line helper in bot_bottle.util."""
|
||||
|
||||
def test_reads_from_dev_tty(self):
|
||||
from unittest.mock import mock_open
|
||||
from bot_bottle.util import read_tty_line
|
||||
|
||||
m = mock_open(read_data="hello\n")
|
||||
with patch("builtins.open", m):
|
||||
result = read_tty_line()
|
||||
self.assertEqual("hello", result)
|
||||
m.assert_called_once_with("/dev/tty", "r", encoding="utf-8")
|
||||
|
||||
def test_falls_back_to_stdin_on_oserror(self):
|
||||
import io
|
||||
from bot_bottle.util import read_tty_line
|
||||
import sys as _sys
|
||||
|
||||
with patch("builtins.open", side_effect=OSError("no tty")), \
|
||||
patch.object(_sys, "stdin", io.StringIO("world\n")):
|
||||
result = read_tty_line()
|
||||
self.assertEqual("world", result)
|
||||
|
||||
|
||||
class TestPrintVmInstallInstructions(unittest.TestCase):
|
||||
"""Unit tests for _print_vm_install_instructions platform branches."""
|
||||
|
||||
def test_linux_prints_firecracker_instructions(self):
|
||||
from bot_bottle.backend import _print_vm_install_instructions
|
||||
|
||||
with patch.object(backend_mod, "_platform_vm_suggestion",
|
||||
return_value="firecracker"), \
|
||||
patch.object(backend_mod, "info") as mock_info:
|
||||
_print_vm_install_instructions()
|
||||
|
||||
messages = [str(c[0][0]) for c in mock_info.call_args_list]
|
||||
self.assertTrue(any("Firecracker" in m for m in messages))
|
||||
|
||||
def test_macos_prints_apple_container_instructions(self):
|
||||
from bot_bottle.backend import _print_vm_install_instructions
|
||||
|
||||
with patch.object(backend_mod, "_platform_vm_suggestion",
|
||||
return_value="macos-container"), \
|
||||
patch.object(backend_mod, "info") as mock_info:
|
||||
_print_vm_install_instructions()
|
||||
|
||||
messages = [str(c[0][0]) for c in mock_info.call_args_list]
|
||||
self.assertTrue(any("Apple Container" in m for m in messages))
|
||||
|
||||
|
||||
class TestKnownBackendNames(unittest.TestCase):
|
||||
def test_returns_backends_sorted(self):
|
||||
|
||||
@@ -215,6 +215,18 @@ class TestDockerSetupStatus(unittest.TestCase):
|
||||
with patch.object(dk.shutil, "which", return_value=None):
|
||||
self.assertFalse(dk._daemon_reachable())
|
||||
|
||||
def test_daemon_reachable_true_when_daemon_responds(self):
|
||||
with patch.object(dk.shutil, "which", return_value="/usr/bin/docker"), \
|
||||
patch.object(dk.subprocess, "run",
|
||||
return_value=subprocess.CompletedProcess([], 0)):
|
||||
self.assertTrue(dk._daemon_reachable())
|
||||
|
||||
def test_daemon_reachable_false_on_timeout(self):
|
||||
with patch.object(dk.shutil, "which", return_value="/usr/bin/docker"), \
|
||||
patch.object(dk.subprocess, "run",
|
||||
side_effect=subprocess.TimeoutExpired(["docker", "info"], 5)):
|
||||
self.assertFalse(dk._daemon_reachable())
|
||||
|
||||
def test_status_reports_missing_docker(self):
|
||||
with patch.object(dk.shutil, "which", return_value=None):
|
||||
rc, out = _cap(dk.status)
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
"""Unit: `cli.py cleanup` walks every backend (issue follow-up).
|
||||
"""Unit: `cli.py cleanup` walks every available backend.
|
||||
|
||||
Asserts cmd_cleanup queries each backend's `prepare_cleanup`,
|
||||
combines the y/N output, and runs each backend's `cleanup` when
|
||||
the operator confirms. Mocks the backends and stdin."""
|
||||
Asserts cmd_cleanup queries each available backend's `prepare_cleanup`,
|
||||
combines the y/N output, and runs each backend's `cleanup` when the
|
||||
operator confirms. Unavailable backends (e.g. macos-container on Linux)
|
||||
are skipped so that `cleanup` never errors on a platform where a backend
|
||||
is not installed. Mocks the backends and stdin."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -21,7 +23,7 @@ def _make_backend(empty: bool = True):
|
||||
|
||||
|
||||
class TestCmdCleanup(unittest.TestCase):
|
||||
def test_iterates_every_backend(self):
|
||||
def test_iterates_every_available_backend(self):
|
||||
docker, docker_plan = _make_backend(empty=False)
|
||||
fc, fc_plan = _make_backend(empty=False)
|
||||
backends_by_name = {"docker": docker, "firecracker": fc}
|
||||
@@ -32,6 +34,8 @@ class TestCmdCleanup(unittest.TestCase):
|
||||
), patch.object(
|
||||
cmd, "get_bottle_backend",
|
||||
side_effect=lambda name: backends_by_name[name], # type: ignore
|
||||
), patch.object(
|
||||
cmd, "has_backend", return_value=True,
|
||||
), patch.object(
|
||||
cmd, "_prompt_yes", return_value=True,
|
||||
):
|
||||
@@ -42,6 +46,32 @@ class TestCmdCleanup(unittest.TestCase):
|
||||
docker.cleanup.assert_called_once_with(docker_plan)
|
||||
fc.cleanup.assert_called_once_with(fc_plan)
|
||||
|
||||
def test_skips_unavailable_backends(self):
|
||||
# macos-container is not available on Linux — must be silently skipped.
|
||||
docker, docker_plan = _make_backend(empty=False)
|
||||
macos = MagicMock()
|
||||
backends_by_name = {"docker": docker}
|
||||
|
||||
def _has(name: str) -> bool:
|
||||
return name == "docker"
|
||||
|
||||
with patch.object(
|
||||
cmd, "known_backend_names",
|
||||
return_value=("docker", "macos-container"),
|
||||
), patch.object(
|
||||
cmd, "get_bottle_backend",
|
||||
side_effect=lambda name: backends_by_name[name], # type: ignore
|
||||
), patch.object(
|
||||
cmd, "has_backend", side_effect=_has,
|
||||
), patch.object(
|
||||
cmd, "_prompt_yes", return_value=True,
|
||||
):
|
||||
self.assertEqual(0, cmd.cmd_cleanup([]))
|
||||
|
||||
docker.prepare_cleanup.assert_called_once()
|
||||
docker.cleanup.assert_called_once_with(docker_plan)
|
||||
macos.prepare_cleanup.assert_not_called()
|
||||
|
||||
def test_short_circuits_when_all_empty(self):
|
||||
docker, _ = _make_backend(empty=True)
|
||||
fc, _ = _make_backend(empty=True)
|
||||
@@ -53,6 +83,8 @@ class TestCmdCleanup(unittest.TestCase):
|
||||
), patch.object(
|
||||
cmd, "get_bottle_backend",
|
||||
side_effect=lambda name: backends_by_name[name], # type: ignore
|
||||
), patch.object(
|
||||
cmd, "has_backend", return_value=True,
|
||||
), patch.object(
|
||||
cmd, "_prompt_yes",
|
||||
) as prompt:
|
||||
@@ -72,6 +104,8 @@ class TestCmdCleanup(unittest.TestCase):
|
||||
), patch.object(
|
||||
cmd, "get_bottle_backend",
|
||||
side_effect=lambda name: backends_by_name[name], # type: ignore
|
||||
), patch.object(
|
||||
cmd, "has_backend", return_value=True,
|
||||
), patch.object(
|
||||
cmd, "_prompt_yes", return_value=False,
|
||||
):
|
||||
@@ -92,6 +126,8 @@ class TestCmdCleanup(unittest.TestCase):
|
||||
), patch.object(
|
||||
cmd, "get_bottle_backend",
|
||||
side_effect=lambda name: backends_by_name[name], # type: ignore
|
||||
), patch.object(
|
||||
cmd, "has_backend", return_value=True,
|
||||
), patch.object(
|
||||
cmd, "_prompt_yes", return_value=True,
|
||||
):
|
||||
|
||||
@@ -95,5 +95,60 @@ class TestMainDispatch(unittest.TestCase):
|
||||
self.assertEqual(130, main(["x"]))
|
||||
|
||||
|
||||
class TestMigrationGate(unittest.TestCase):
|
||||
"""The dispatcher's schema-migration gate (cli/__init__.py)."""
|
||||
|
||||
def setUp(self) -> None:
|
||||
# Force the "schema out of date" branch for every test here.
|
||||
patcher = patch.object(StoreManager, "is_migrated", return_value=False)
|
||||
patcher.start()
|
||||
self.addCleanup(patcher.stop)
|
||||
self.addCleanup(StoreManager.reset)
|
||||
|
||||
def test_store_command_blocks_when_stdin_cannot_confirm(self) -> None:
|
||||
# Non-TTY stdin at EOF (as in CI): the [y/N] prompt reads "" and the
|
||||
# command is refused rather than migrating silently.
|
||||
ran: list[bool] = []
|
||||
|
||||
def handler(_rest: list[str]) -> int:
|
||||
ran.append(True)
|
||||
return 0
|
||||
|
||||
with patch.dict(climod.COMMANDS, {"list": handler}), \
|
||||
patch("sys.stdin", io.StringIO("")), \
|
||||
patch("sys.stderr", io.StringIO()):
|
||||
self.assertEqual(1, main(["list"]))
|
||||
self.assertEqual([], ran, "gated command must not dispatch")
|
||||
|
||||
def test_backend_command_skips_gate(self) -> None:
|
||||
# `backend` provisions/probes the host and never opens the store, so
|
||||
# it must run even on an unmigrated DB with unanswerable stdin.
|
||||
ran: list[bool] = []
|
||||
|
||||
def handler(_rest: list[str]) -> int:
|
||||
ran.append(True)
|
||||
return 0
|
||||
|
||||
with patch.dict(climod.COMMANDS, {"backend": handler}), \
|
||||
patch("sys.stdin", io.StringIO("")), \
|
||||
patch("sys.stderr", io.StringIO()):
|
||||
self.assertEqual(0, main(["backend", "status"]))
|
||||
self.assertEqual([True], ran, "exempt command must dispatch")
|
||||
|
||||
def test_store_command_migrates_on_confirmation(self) -> None:
|
||||
migrated: list[bool] = []
|
||||
|
||||
def handler(_rest: list[str]) -> int:
|
||||
return 0
|
||||
|
||||
with patch.dict(climod.COMMANDS, {"list": handler}), \
|
||||
patch.object(StoreManager, "migrate",
|
||||
side_effect=lambda: migrated.append(True)), \
|
||||
patch("sys.stdin", io.StringIO("y\n")), \
|
||||
patch("sys.stderr", io.StringIO()):
|
||||
self.assertEqual(0, main(["list"]))
|
||||
self.assertEqual([True], migrated, "confirmed gate must migrate")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -1,61 +1,29 @@
|
||||
"""Unit: `cli.py start --backend=<name>` flag (issue #77).
|
||||
"""Unit: backend resolution priority — explicit name > env var.
|
||||
|
||||
Asserts that the flag wins over the env var, that the env var is
|
||||
the fallback, and that the choices are pulled from the backend
|
||||
registry (so adding a backend lights up in argparse without code
|
||||
edits)."""
|
||||
The `--backend` flag has been removed from `cli.py start`; backend
|
||||
selection is driven by BOT_BOTTLE_BACKEND or auto-selection only.
|
||||
`get_bottle_backend` still accepts an explicit name so that `resume`
|
||||
(which records the backend from a prior session) and the `backend`
|
||||
sub-command can pass one directly."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
|
||||
from bot_bottle.backend import known_backend_names
|
||||
|
||||
|
||||
class TestStartBackendFlag(unittest.TestCase):
|
||||
"""The flag is wired by `cmd_start`'s argparse and threaded
|
||||
through `prepare_with_preflight(backend_name=...)`. Rather than
|
||||
drive the whole start flow (which builds containers), we test
|
||||
the argparse shape and the resolution function separately."""
|
||||
|
||||
def _build_parser(self):
|
||||
# Mirror the parser definition from `cmd_start` so this
|
||||
# test doesn't have to invoke the full command.
|
||||
parser = argparse.ArgumentParser(prog="cb start")
|
||||
parser.add_argument(
|
||||
"--backend",
|
||||
choices=known_backend_names(),
|
||||
default=None,
|
||||
)
|
||||
parser.add_argument("name")
|
||||
return parser
|
||||
|
||||
def test_flag_recognized(self):
|
||||
args = self._build_parser().parse_args(["--backend=firecracker", "researcher"])
|
||||
self.assertEqual("firecracker", args.backend)
|
||||
self.assertEqual("researcher", args.name)
|
||||
|
||||
def test_flag_default_none_means_env_or_default_backend(self):
|
||||
args = self._build_parser().parse_args(["researcher"])
|
||||
self.assertIsNone(args.backend)
|
||||
|
||||
def test_invalid_backend_rejected_by_argparse(self):
|
||||
parser = self._build_parser()
|
||||
with self.assertRaises(SystemExit):
|
||||
parser.parse_args(["--backend=garbage", "researcher"])
|
||||
|
||||
def test_resolution_priority_explicit_over_env(self):
|
||||
# Independent assertion that get_bottle_backend (where
|
||||
# `--backend` ultimately threads to) prefers the explicit
|
||||
# name over BOT_BOTTLE_BACKEND.
|
||||
class TestBackendResolutionPriority(unittest.TestCase):
|
||||
def test_explicit_name_overrides_env_var(self):
|
||||
from bot_bottle.backend import get_bottle_backend
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_BACKEND": "firecracker"}):
|
||||
self.assertEqual("docker", get_bottle_backend("docker").name)
|
||||
|
||||
def test_env_var_used_when_no_explicit_name(self):
|
||||
from bot_bottle.backend import get_bottle_backend
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_BACKEND": "firecracker"}):
|
||||
self.assertEqual("firecracker", get_bottle_backend().name)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -175,14 +175,6 @@ class TestCmdStartHeadless(unittest.TestCase):
|
||||
)
|
||||
self.assertEqual("researcher-2", self._spec().label)
|
||||
|
||||
# -- backend wiring ------------------------------------------------
|
||||
|
||||
def test_backend_flag_forwarded(self):
|
||||
start_mod.cmd_start(
|
||||
["--headless", "--backend=docker", "researcher", "--bottle", "claude",
|
||||
"--prompt", "Do it"]
|
||||
)
|
||||
self.assertEqual("docker", self._launch_mock.call_args[1]["backend_name"])
|
||||
|
||||
|
||||
class TestPrepareWithPreflight(unittest.TestCase):
|
||||
|
||||
@@ -57,6 +57,14 @@ class TestCmdStartSelector(unittest.TestCase):
|
||||
self._bottle_picker_mock = self._bottle_picker_patch.start()
|
||||
self._bottle_picker_mock.return_value = ["claude"] # default: one bottle selected
|
||||
|
||||
# name_color_modal opens /dev/tty and blocks on keyboard input on
|
||||
# self-hosted runners that have a real controlling terminal. Stub it
|
||||
# out like the other tui pickers so tests don't wait for a keypress.
|
||||
self._modal_patch = patch.object(
|
||||
tui_mod, "name_color_modal", return_value=("researcher", ""),
|
||||
)
|
||||
self._modal_patch.start()
|
||||
|
||||
self._env_patch = patch.dict(os.environ, {}, clear=False)
|
||||
self._env_patch.start()
|
||||
os.environ.pop("BOT_BOTTLE_BACKEND", None)
|
||||
@@ -66,6 +74,7 @@ class TestCmdStartSelector(unittest.TestCase):
|
||||
self._launch_patch.stop()
|
||||
self._agent_picker_patch.stop()
|
||||
self._bottle_picker_patch.stop()
|
||||
self._modal_patch.stop()
|
||||
self._env_patch.stop()
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
@@ -73,7 +82,7 @@ class TestCmdStartSelector(unittest.TestCase):
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def test_explicit_agent_skips_agent_picker(self):
|
||||
rc = start_mod.cmd_start(["--backend=docker", "researcher"])
|
||||
rc = start_mod.cmd_start(["researcher"])
|
||||
self.assertEqual(0, rc)
|
||||
self._agent_picker_mock.assert_not_called()
|
||||
self._bottle_picker_mock.assert_called_once()
|
||||
@@ -91,7 +100,7 @@ class TestCmdStartSelector(unittest.TestCase):
|
||||
|
||||
def test_agent_absent_shows_agent_picker(self):
|
||||
self._agent_picker_mock.return_value = "researcher"
|
||||
rc = start_mod.cmd_start(["--backend=docker"])
|
||||
rc = start_mod.cmd_start([])
|
||||
self.assertEqual(0, rc)
|
||||
self._agent_picker_mock.assert_called_once()
|
||||
call_kwargs = self._agent_picker_mock.call_args
|
||||
@@ -102,7 +111,7 @@ class TestCmdStartSelector(unittest.TestCase):
|
||||
|
||||
def test_agent_picker_cancel_skips_bottle_picker(self):
|
||||
self._agent_picker_mock.return_value = None
|
||||
rc = start_mod.cmd_start(["--backend=docker"])
|
||||
rc = start_mod.cmd_start([])
|
||||
self.assertEqual(0, rc)
|
||||
self._bottle_picker_mock.assert_not_called()
|
||||
self._launch_mock.assert_not_called()
|
||||
@@ -159,16 +168,6 @@ class TestCmdStartSelector(unittest.TestCase):
|
||||
# Backend wiring
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def test_explicit_backend_forwarded(self):
|
||||
start_mod.cmd_start(["--backend=docker", "researcher"])
|
||||
_, kwargs = self._launch_mock.call_args
|
||||
self.assertEqual("docker", kwargs["backend_name"])
|
||||
|
||||
def test_absent_backend_uses_default(self):
|
||||
start_mod.cmd_start(["researcher"])
|
||||
_, kwargs = self._launch_mock.call_args
|
||||
self.assertIsNone(kwargs["backend_name"])
|
||||
|
||||
def test_bot_bottle_backend_env_skips_backend_picker(self):
|
||||
os.environ["BOT_BOTTLE_BACKEND"] = "docker"
|
||||
try:
|
||||
|
||||
@@ -1,186 +0,0 @@
|
||||
"""Unit: host Claude auth extraction."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from bot_bottle.contrib.claude.claude_auth import (
|
||||
claude_auth_path,
|
||||
claude_host_access_token,
|
||||
)
|
||||
from bot_bottle.log import Die
|
||||
|
||||
|
||||
def _cred_json(access_token: str, **extra: object) -> str:
|
||||
payload: dict[str, object] = {"claudeAiOauth": {"accessToken": access_token, **extra}}
|
||||
return json.dumps(payload)
|
||||
|
||||
|
||||
class TestClaudeHostAccessToken(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.TemporaryDirectory(prefix="bb-claude-auth.")
|
||||
self.home = Path(self.tmp.name)
|
||||
self.cred_dir = self.home / ".claude"
|
||||
self.cred_dir.mkdir()
|
||||
self.auth_path = self.cred_dir / ".credentials.json"
|
||||
|
||||
def tearDown(self):
|
||||
self.tmp.cleanup()
|
||||
|
||||
def _write(self, payload: dict) -> None: # type: ignore[no-untyped-def]
|
||||
self.auth_path.write_text(json.dumps(payload))
|
||||
|
||||
def test_auth_path_uses_home_env(self):
|
||||
self.assertEqual(
|
||||
self.auth_path,
|
||||
claude_auth_path({"HOME": str(self.home)}),
|
||||
)
|
||||
|
||||
# --- file-based (Linux) ---
|
||||
|
||||
def test_file_returns_access_token(self):
|
||||
key = "sk-ant-oat01-real-key" # gitleaks:allow
|
||||
self._write({"claudeAiOauth": {"accessToken": key}})
|
||||
out = claude_host_access_token({"HOME": str(self.home)})
|
||||
self.assertEqual(key, out)
|
||||
|
||||
def test_file_missing_claude_ai_oauth_dies(self):
|
||||
self._write({"hasCompletedOnboarding": True})
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(self.home)})
|
||||
|
||||
def test_file_missing_access_token_dies(self):
|
||||
self._write({"claudeAiOauth": {"expiresAt": 2000000000000}})
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(self.home)})
|
||||
|
||||
def test_file_empty_access_token_dies(self):
|
||||
self._write({"claudeAiOauth": {"accessToken": ""}})
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(self.home)})
|
||||
|
||||
def test_file_expired_token_dies(self):
|
||||
# expiresAt is milliseconds; 1_000_000 ms is year 1970
|
||||
self._write({
|
||||
"claudeAiOauth": {"accessToken": "sk-ant-oat01-x", "expiresAt": 1_000_000}, # gitleaks:allow
|
||||
})
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token(
|
||||
{"HOME": str(self.home)},
|
||||
now=datetime(2026, 1, 1, tzinfo=timezone.utc),
|
||||
)
|
||||
|
||||
def test_file_future_expiry_is_accepted(self):
|
||||
key = "sk-ant-oat01-y" # gitleaks:allow
|
||||
# 2_000_000_000_000 ms ≈ year 2033
|
||||
self._write({
|
||||
"claudeAiOauth": {"accessToken": key, "expiresAt": 2_000_000_000_000},
|
||||
})
|
||||
out = claude_host_access_token(
|
||||
{"HOME": str(self.home)},
|
||||
now=datetime(2026, 1, 1, tzinfo=timezone.utc),
|
||||
)
|
||||
self.assertEqual(key, out)
|
||||
|
||||
def test_file_absent_expiry_is_accepted(self):
|
||||
key = "sk-ant-oat01-z" # gitleaks:allow
|
||||
self._write({"claudeAiOauth": {"accessToken": key}})
|
||||
out = claude_host_access_token({"HOME": str(self.home)})
|
||||
self.assertEqual(key, out)
|
||||
|
||||
def test_file_non_json_dies(self):
|
||||
self.auth_path.write_text("not json {{{")
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(self.home)})
|
||||
|
||||
def test_file_json_array_root_dies(self):
|
||||
self.auth_path.write_text("[]")
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(self.home)})
|
||||
|
||||
def test_file_extra_fields_are_ignored(self):
|
||||
key = "sk-ant-oat01-real" # gitleaks:allow
|
||||
self._write({
|
||||
"claudeAiOauth": {
|
||||
"accessToken": key,
|
||||
"refreshToken": "sk-ant-ort01-secret", # gitleaks:allow
|
||||
"scopes": ["user:inference"],
|
||||
"expiresAt": 2_000_000_000_000,
|
||||
},
|
||||
})
|
||||
out = claude_host_access_token({"HOME": str(self.home)})
|
||||
self.assertEqual(key, out)
|
||||
|
||||
# --- macOS Keychain fallback ---
|
||||
|
||||
def _home_without_creds(self) -> Path:
|
||||
"""A home dir that has .claude/ but no .credentials.json."""
|
||||
empty = self.home / "no-creds"
|
||||
(empty / ".claude").mkdir(parents=True)
|
||||
return empty
|
||||
|
||||
def _mock_keychain(self, stdout: str, returncode: int = 0) -> MagicMock:
|
||||
mock = MagicMock()
|
||||
mock.returncode = returncode
|
||||
mock.stdout = stdout
|
||||
return mock
|
||||
|
||||
def test_keychain_used_when_file_absent(self):
|
||||
key = "sk-ant-oat01-keychain" # gitleaks:allow
|
||||
home = self._home_without_creds()
|
||||
with patch(
|
||||
"bot_bottle.contrib.claude.claude_auth.subprocess.run",
|
||||
return_value=self._mock_keychain(_cred_json(key)),
|
||||
), patch(
|
||||
"bot_bottle.contrib.claude.claude_auth.sys.platform", "darwin",
|
||||
):
|
||||
out = claude_host_access_token({"HOME": str(home)})
|
||||
self.assertEqual(key, out)
|
||||
|
||||
def test_keychain_failure_when_file_absent_dies(self):
|
||||
home = self._home_without_creds()
|
||||
with patch(
|
||||
"bot_bottle.contrib.claude.claude_auth.subprocess.run",
|
||||
return_value=self._mock_keychain("", returncode=44),
|
||||
), patch(
|
||||
"bot_bottle.contrib.claude.claude_auth.sys.platform", "darwin",
|
||||
):
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(home)})
|
||||
|
||||
def test_no_file_no_keychain_on_linux_dies(self):
|
||||
home = self._home_without_creds()
|
||||
with patch("bot_bottle.contrib.claude.claude_auth.sys.platform", "linux"):
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(home)})
|
||||
|
||||
def test_keychain_non_json_dies(self):
|
||||
home = self._home_without_creds()
|
||||
with patch(
|
||||
"bot_bottle.contrib.claude.claude_auth.subprocess.run",
|
||||
return_value=self._mock_keychain("not-json"),
|
||||
), patch(
|
||||
"bot_bottle.contrib.claude.claude_auth.sys.platform", "darwin",
|
||||
):
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(home)})
|
||||
|
||||
def test_keychain_security_not_found_dies(self):
|
||||
home = self._home_without_creds()
|
||||
with patch(
|
||||
"bot_bottle.contrib.claude.claude_auth.subprocess.run",
|
||||
side_effect=FileNotFoundError,
|
||||
), patch(
|
||||
"bot_bottle.contrib.claude.claude_auth.sys.platform", "darwin",
|
||||
):
|
||||
with self.assertRaises(Die):
|
||||
claude_host_access_token({"HOME": str(home)})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Unit: DbStore._connection() context manager and is_migrated()."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlite3
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from bot_bottle.db_store import DbStore
|
||||
from bot_bottle.migrations import TableMigrations
|
||||
|
||||
|
||||
def _store(tmp: Path) -> DbStore:
|
||||
migrations = TableMigrations("test", ["CREATE TABLE items (id INTEGER PRIMARY KEY)"])
|
||||
return DbStore(tmp / "test.db", migrations)
|
||||
|
||||
|
||||
class TestDbStoreIsMigrated(unittest.TestCase):
|
||||
def test_returns_false_when_db_absent(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
store = _store(Path(d))
|
||||
self.assertFalse(store.is_migrated())
|
||||
|
||||
def test_returns_false_when_schema_versions_missing(self):
|
||||
# DB file exists but has no schema_versions table → OperationalError → False.
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
store = _store(Path(d))
|
||||
conn = sqlite3.connect(store.db_path)
|
||||
conn.close()
|
||||
self.assertFalse(store.is_migrated())
|
||||
|
||||
def test_returns_true_after_migrate(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
store = _store(Path(d))
|
||||
store.migrate()
|
||||
self.assertTrue(store.is_migrated())
|
||||
|
||||
def test_returns_false_when_behind(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
migrations = TableMigrations(
|
||||
"test",
|
||||
[
|
||||
"CREATE TABLE items (id INTEGER PRIMARY KEY)",
|
||||
"ALTER TABLE items ADD COLUMN name TEXT",
|
||||
],
|
||||
)
|
||||
store = DbStore(Path(d) / "test.db", migrations)
|
||||
# Apply only the first migration manually.
|
||||
conn = sqlite3.connect(store.db_path)
|
||||
with conn:
|
||||
TableMigrations("test", [migrations.migrations[0]]).apply(conn)
|
||||
conn.close()
|
||||
self.assertFalse(store.is_migrated())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,35 @@
|
||||
"""Tests for integration-test backend selection helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
from tests._docker import skip_unless_docker_or_firecracker
|
||||
|
||||
|
||||
class TestSkipUnlessDockerOrFirecracker(unittest.TestCase):
|
||||
def test_firecracker_runs_when_docker_tests_are_disabled(self):
|
||||
with patch.dict(
|
||||
os.environ,
|
||||
{"BOT_BOTTLE_BACKEND": "firecracker", "SKIP_DOCKER_TESTS": "1"},
|
||||
clear=True,
|
||||
):
|
||||
decorated = skip_unless_docker_or_firecracker()(type("Case", (), {}))
|
||||
|
||||
self.assertFalse(getattr(decorated, "__unittest_skip__", False))
|
||||
|
||||
def test_non_firecracker_still_skips_when_docker_tests_are_disabled(self):
|
||||
with patch.dict(
|
||||
os.environ,
|
||||
{"BOT_BOTTLE_BACKEND": "docker", "SKIP_DOCKER_TESTS": "1"},
|
||||
clear=True,
|
||||
):
|
||||
decorated = skip_unless_docker_or_firecracker()(type("Case", (), {}))
|
||||
|
||||
self.assertTrue(getattr(decorated, "__unittest_skip__", False))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -35,9 +35,12 @@ class TestNetpoolSlots(unittest.TestCase):
|
||||
def test_slot_ip_math_31_pairs(self):
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_FC_IP_BASE": "100.64.0.0"}):
|
||||
s0, s1 = netpool.slot(0), netpool.slot(1)
|
||||
self.assertEqual(("bbfc0", "100.64.0.0", "100.64.0.1"),
|
||||
# Iface names track netpool's (env-driven) prefix — the KVM CI runner
|
||||
# overrides it for its isolated pool, so don't hardcode "bbfc".
|
||||
pfx = netpool.IFACE_PREFIX
|
||||
self.assertEqual((f"{pfx}0", "100.64.0.0", "100.64.0.1"),
|
||||
(s0.iface, s0.host_ip, s0.guest_ip))
|
||||
self.assertEqual(("bbfc1", "100.64.0.2", "100.64.0.3"),
|
||||
self.assertEqual((f"{pfx}1", "100.64.0.2", "100.64.0.3"),
|
||||
(s1.iface, s1.host_ip, s1.guest_ip))
|
||||
|
||||
def test_guest_cidr_is_31(self):
|
||||
@@ -180,11 +183,14 @@ class TestNetpoolOverlap(unittest.TestCase):
|
||||
self.assertEqual("tailscale0", conflicts[0].dev)
|
||||
|
||||
def test_ignores_own_taps_and_default(self):
|
||||
# The "own tap" route uses netpool's (env-driven) iface name, so the
|
||||
# test still exercises the self-ignore path on the KVM CI runner, whose
|
||||
# BOT_BOTTLE_FC_IFACE_PREFIX differs from the default.
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_FC_IP_BASE": "10.243.0.0",
|
||||
"BOT_BOTTLE_FC_POOL_SIZE": "8"}), \
|
||||
self._routes([
|
||||
{"dst": "default", "dev": "enp4s0"},
|
||||
{"dst": "10.243.0.0/31", "dev": "bbfc0"},
|
||||
{"dst": "10.243.0.0/31", "dev": netpool.slot(0).iface},
|
||||
{"dst": "192.168.1.0/24", "dev": "enp4s0"},
|
||||
]):
|
||||
self.assertEqual([], netpool.overlapping_routes())
|
||||
@@ -203,7 +209,7 @@ class TestNetpoolAllocation(unittest.TestCase):
|
||||
# over (and here, exhaust the pool).
|
||||
slot, lock = netpool.allocate("first")
|
||||
self.addCleanup(lock.close)
|
||||
self.assertEqual("bbfc0", slot.iface)
|
||||
self.assertEqual(netpool.slot(0).iface, slot.iface)
|
||||
with patch.object(netpool, "die",
|
||||
side_effect=SystemExit("exhausted")):
|
||||
with self.assertRaises(SystemExit):
|
||||
@@ -323,9 +329,14 @@ class TestNetpoolDefaultsSingleSource(unittest.TestCase):
|
||||
with patch.dict(os.environ, {}, clear=True):
|
||||
self.assertEqual(int(d["BOT_BOTTLE_FC_POOL_SIZE"]), netpool.pool_size())
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_IP_BASE"], netpool.ip_base())
|
||||
# Module constants resolve through the same shared file.
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_IFACE_PREFIX"], netpool.IFACE_PREFIX)
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_NFT_TABLE"], netpool.NFT_TABLE)
|
||||
# The module's *defaults* come from the same shared file. Assert the
|
||||
# parsed defaults (not IFACE_PREFIX/NFT_TABLE, which layer a live env
|
||||
# override on top — the KVM CI runner sets those for its isolated pool,
|
||||
# which would otherwise mask this single-source check).
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_IFACE_PREFIX"],
|
||||
netpool._DEFAULTS["BOT_BOTTLE_FC_IFACE_PREFIX"])
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_NFT_TABLE"],
|
||||
netpool._DEFAULTS["BOT_BOTTLE_FC_NFT_TABLE"])
|
||||
|
||||
def test_env_var_overrides_the_shared_default(self):
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_FC_IP_BASE": "10.99.0.0"}):
|
||||
|
||||
@@ -63,9 +63,12 @@ class TestNetpoolProbes(unittest.TestCase):
|
||||
self.assertEqual(2, ok.call_count)
|
||||
|
||||
def test_missing_taps(self):
|
||||
# Derive the expected iface from netpool's (env-driven) config rather
|
||||
# than hardcoding "bbfc1": the KVM CI runner sets BOT_BOTTLE_FC_* for
|
||||
# its isolated pool, so the prefix there is not the default.
|
||||
with patch.dict("os.environ", {"BOT_BOTTLE_FC_POOL_SIZE": "2"}), \
|
||||
patch.object(netpool, "tap_present", side_effect=[True, False]):
|
||||
self.assertEqual(["bbfc1"], netpool.missing_taps())
|
||||
self.assertEqual([netpool.slot(1).iface], netpool.missing_taps())
|
||||
|
||||
def test_orch_slot_is_top_of_ip_base_16(self):
|
||||
# Dedicated orchestrator link: /31 at the top of the IP_BASE /16,
|
||||
|
||||
@@ -24,7 +24,7 @@ class TestBuildAgentRootfsDir(unittest.TestCase):
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
|
||||
def test_cache_hit_skips_rebuild(self):
|
||||
digest = image_builder._dockerfile_hash(self.dockerfile)
|
||||
digest = image_builder._rootfs_digest(self.dockerfile)
|
||||
base = self.cache / "rootfs" / f"agent-{digest}"
|
||||
base.mkdir(parents=True)
|
||||
(base / ".bb-ready").write_text("ok\n")
|
||||
@@ -55,6 +55,17 @@ class TestBuildAgentRootfsDir(unittest.TestCase):
|
||||
image_builder._dockerfile_hash(other),
|
||||
)
|
||||
|
||||
def test_rootfs_digest_tracks_dockerfile_and_init(self):
|
||||
# Same Dockerfile, different injected init -> different rootfs key, so
|
||||
# an init fix (e.g. /tmp perms) rebuilds instead of reusing a stale
|
||||
# rootfs; different Dockerfiles also differ.
|
||||
base = image_builder._rootfs_digest(self.dockerfile)
|
||||
with patch.object(image_builder.util, "_GUEST_INIT", "#!/bin/sh\n# changed\n"):
|
||||
self.assertNotEqual(base, image_builder._rootfs_digest(self.dockerfile))
|
||||
other = self.cache / "Dockerfile2"
|
||||
other.write_text("FROM python:3.12-slim\n")
|
||||
self.assertNotEqual(base, image_builder._rootfs_digest(other))
|
||||
|
||||
|
||||
class TestSmokeTest(unittest.TestCase):
|
||||
def test_empty_argv_is_noop(self):
|
||||
|
||||
@@ -38,7 +38,9 @@ class TestBuildInfraRootfs(unittest.TestCase):
|
||||
# and exports PATH so gateway_init's subprocess daemons find python3.
|
||||
init = build.call_args.kwargs["init_script"]
|
||||
self.assertIn("bot_bottle.orchestrator", init)
|
||||
self.assertIn("gateway_init.py", init)
|
||||
# Gateway launches via the installed package (there is no
|
||||
# /app/gateway_init.py file since the daemons moved into bot_bottle).
|
||||
self.assertIn("bot_bottle.gateway_init", init)
|
||||
self.assertIn("export PATH=", init)
|
||||
# Persistent registry volume mounted at the DB dir before the CP starts.
|
||||
self.assertIn("/dev/vdb", init)
|
||||
@@ -98,7 +100,11 @@ class TestRegistryVolume(unittest.TestCase):
|
||||
class TestEnsureBuilt(unittest.TestCase):
|
||||
def test_default_pulls_artifact_without_docker(self):
|
||||
# PRD 0069 Stage 2: the launch host pulls the prebuilt rootfs; no Docker.
|
||||
with patch.object(infra_vm.docker_mod, "build_image") as build, \
|
||||
# Pin BOT_BOTTLE_INFRA_BUILD off: the coverage CI job exports it =local
|
||||
# for the integration suite, and that ambient value would otherwise send
|
||||
# this default-path test down the local Docker-build branch.
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_INFRA_BUILD": ""}), \
|
||||
patch.object(infra_vm.docker_mod, "build_image") as build, \
|
||||
patch.object(infra_vm.infra_artifact, "ensure_artifact_gz") as pull:
|
||||
infra_vm.ensure_built()
|
||||
build.assert_not_called()
|
||||
@@ -138,27 +144,58 @@ class TestWaitForHealth(unittest.TestCase):
|
||||
|
||||
|
||||
class TestEnsureRunningSingleton(unittest.TestCase):
|
||||
def test_adopts_when_healthy(self):
|
||||
# A healthy control plane + existing key -> adopt (no boot), vm=None.
|
||||
with patch.object(infra_vm, "_health_ok", return_value=True), \
|
||||
patch.object(infra_vm, "_infra_dir") as d, \
|
||||
patch.object(infra_vm, "boot") as boot:
|
||||
keydir = MagicMock()
|
||||
(keydir / "id_ed25519").exists.return_value = True
|
||||
d.return_value = keydir
|
||||
infra = infra_vm.ensure_running()
|
||||
def test_adopts_when_healthy_and_version_matches(self):
|
||||
# Healthy control plane + existing key + matching version marker
|
||||
# -> adopt (no boot), vm=None.
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = Path(td)
|
||||
(d / "id_ed25519").write_text("k")
|
||||
(d / "booted-version").write_text("v-current\n")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d), \
|
||||
patch.object(infra_vm, "_expected_version", return_value="v-current"), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=True), \
|
||||
patch.object(infra_vm, "boot") as boot:
|
||||
infra = infra_vm.ensure_running()
|
||||
boot.assert_not_called()
|
||||
self.assertIsNone(infra.vm)
|
||||
|
||||
def test_reboots_when_version_stale(self):
|
||||
# Healthy control plane but the running VM booted an OLDER image
|
||||
# (marker mismatch) -> reboot rather than adopt stale code.
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = Path(td)
|
||||
(d / "id_ed25519").write_text("k")
|
||||
(d / "booted-version").write_text("v-old\n")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d), \
|
||||
patch.object(infra_vm, "_expected_version", return_value="v-current"), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=True), \
|
||||
patch.object(infra_vm, "stop") as stop, \
|
||||
patch.object(infra_vm, "ensure_built"), \
|
||||
patch.object(infra_vm, "wait_for_health"), \
|
||||
patch.object(infra_vm, "boot") as boot:
|
||||
boot.return_value = infra_vm.InfraVm(
|
||||
guest_ip="10.243.255.1", private_key=Path("/k"), vm=MagicMock())
|
||||
infra_vm.ensure_running()
|
||||
stop.assert_called_once() # dislodge the outdated VM
|
||||
boot.assert_called_once()
|
||||
# The fresh boot records the current version for the next launcher.
|
||||
self.assertEqual("v-current\n", (d / "booted-version").read_text())
|
||||
|
||||
def test_boots_when_unhealthy(self):
|
||||
with patch.object(infra_vm, "_health_ok", return_value=False), \
|
||||
patch.object(infra_vm, "stop") as stop, \
|
||||
patch.object(infra_vm, "ensure_built") as built, \
|
||||
patch.object(infra_vm, "boot") as boot, \
|
||||
patch.object(infra_vm, "wait_for_health") as wait:
|
||||
boot.return_value = infra_vm.InfraVm(
|
||||
guest_ip="10.243.255.1", private_key=Path("/k"), vm=MagicMock())
|
||||
infra_vm.ensure_running()
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=Path(td)), \
|
||||
patch.object(infra_vm, "_expected_version", return_value="v-current"), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=False), \
|
||||
patch.object(infra_vm, "stop") as stop, \
|
||||
patch.object(infra_vm, "ensure_built") as built, \
|
||||
patch.object(infra_vm, "boot") as boot, \
|
||||
patch.object(infra_vm, "wait_for_health") as wait:
|
||||
boot.return_value = infra_vm.InfraVm(
|
||||
guest_ip="10.243.255.1", private_key=Path("/k"), vm=MagicMock())
|
||||
infra_vm.ensure_running()
|
||||
stop.assert_called_once() # clear a stale VM first
|
||||
built.assert_called_once()
|
||||
boot.assert_called_once()
|
||||
@@ -185,5 +222,74 @@ class TestKillPidfile(unittest.TestCase):
|
||||
kill.assert_not_called()
|
||||
|
||||
|
||||
class TestAdoptable(unittest.TestCase):
|
||||
def _dir(self, td: str, *, key: bool = True, version: str | None = None) -> Path:
|
||||
d = Path(td)
|
||||
if key:
|
||||
(d / "id_ed25519").write_text("k")
|
||||
if version is not None:
|
||||
(d / "booted-version").write_text(version + "\n")
|
||||
return d
|
||||
|
||||
def test_true_when_key_version_and_health(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = self._dir(td, version="v1")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=True):
|
||||
self.assertTrue(infra_vm._adoptable(d / "id_ed25519", "u", "v1"))
|
||||
|
||||
def test_false_when_key_missing(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = self._dir(td, key=False, version="v1")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d):
|
||||
self.assertFalse(infra_vm._adoptable(d / "id_ed25519", "u", "v1"))
|
||||
|
||||
def test_false_when_no_version_marker(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = self._dir(td) # key present, no booted-version
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d):
|
||||
self.assertFalse(infra_vm._adoptable(d / "id_ed25519", "u", "v1"))
|
||||
|
||||
def test_false_when_version_mismatch(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = self._dir(td, version="v-old")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=True):
|
||||
self.assertFalse(infra_vm._adoptable(d / "id_ed25519", "u", "v1"))
|
||||
|
||||
|
||||
class TestKillInfraFirecrackers(unittest.TestCase):
|
||||
def _fake_proc(self, root: Path, pid: int, comm: str, cmdline: list[str]) -> None:
|
||||
p = root / str(pid)
|
||||
p.mkdir()
|
||||
(p / "comm").write_text(comm + "\n")
|
||||
(p / "cmdline").write_bytes(b"\0".join(a.encode() for a in cmdline) + b"\0")
|
||||
|
||||
def test_kills_only_matching_infra_firecracker(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td, \
|
||||
tempfile.TemporaryDirectory() as proc:
|
||||
infra_dir = Path(td)
|
||||
cfg = str(infra_dir / "config.json")
|
||||
root = Path(proc)
|
||||
# target: firecracker bound to the infra config -> killed
|
||||
self._fake_proc(root, 111, "firecracker",
|
||||
["firecracker", "--no-api", "--config-file", cfg])
|
||||
# a firecracker for a different (interactive) VM -> spared
|
||||
self._fake_proc(root, 222, "firecracker",
|
||||
["firecracker", "--config-file", "/home/u/other.json"])
|
||||
# a non-firecracker process on the same config path -> spared
|
||||
self._fake_proc(root, 333, "python3", ["python3", cfg])
|
||||
(root / "not-a-pid").mkdir()
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=infra_dir), \
|
||||
patch.object(infra_vm.os, "kill") as kill:
|
||||
infra_vm._kill_infra_firecrackers(proc_root=root)
|
||||
kill.assert_called_once_with(111, infra_vm.signal.SIGKILL)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -2,9 +2,9 @@
|
||||
|
||||
Tests both the helper functions in `bot_bottle.gateway_init`
|
||||
and the supervisor's end-to-end signal / exit-code behavior. The
|
||||
end-to-end tests use real subprocesses (`/bin/sleep`,
|
||||
`/bin/sh -c '...'`) — short-lived, no docker required — so they
|
||||
run under `tests/unit/` rather than `tests/integration/`."""
|
||||
end-to-end tests use real subprocesses (`sleep`, `/bin/sh -c '...'`) —
|
||||
short-lived, no docker required — so they run under `tests/unit/`
|
||||
rather than `tests/integration/`."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -25,6 +25,7 @@ from bot_bottle.gateway_init import (
|
||||
_env_for_daemon,
|
||||
_selected_daemons,
|
||||
)
|
||||
from tests._bin import SLEEP
|
||||
|
||||
|
||||
class TestEnvForDaemon(unittest.TestCase):
|
||||
@@ -182,7 +183,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
# up and the supervisor never set shutdown_at.
|
||||
specs = [
|
||||
_DaemonSpec("crasher", ("/bin/sh", "-c", "exit 1")),
|
||||
_DaemonSpec("longrun", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("longrun", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -214,7 +215,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
# signal-killed longrun's negative returncode.
|
||||
specs = [
|
||||
_DaemonSpec("crasher", ("/bin/sh", "-c", "exit 1")),
|
||||
_DaemonSpec("longrun", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("longrun", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -259,7 +260,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
)
|
||||
specs = [
|
||||
_DaemonSpec("egress", sighup_marker),
|
||||
_DaemonSpec("other", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("other", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -282,7 +283,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_forward_signal_unknown_daemon_no_op(self):
|
||||
specs = [_DaemonSpec("a", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("a", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
delivered = sup.forward_signal(signal.SIGHUP, "ghost")
|
||||
@@ -294,8 +295,8 @@ class TestSupervisor(unittest.TestCase):
|
||||
# Restart one daemon; the other (supervise, the MCP server
|
||||
# in production) must remain untouched.
|
||||
specs = [
|
||||
_DaemonSpec("git-gate", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("supervise", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("git-gate", (SLEEP, "30")),
|
||||
_DaemonSpec("supervise", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -319,8 +320,8 @@ class TestSupervisor(unittest.TestCase):
|
||||
|
||||
def test_request_restart_is_drained_by_tick(self):
|
||||
specs = [
|
||||
_DaemonSpec("git-gate", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("supervise", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("git-gate", (SLEEP, "30")),
|
||||
_DaemonSpec("supervise", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -343,7 +344,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_repeated_restart_requests_coalesce(self):
|
||||
specs = [_DaemonSpec("git-gate", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("git-gate", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
time.sleep(0.1)
|
||||
@@ -366,7 +367,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_request_restart_unknown_daemon_no_op(self):
|
||||
specs = [_DaemonSpec("a", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("a", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
ok = sup.request_restart("ghost")
|
||||
@@ -376,7 +377,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_restart_unknown_daemon_no_op(self):
|
||||
specs = [_DaemonSpec("a", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("a", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
ok = sup.restart_daemon("ghost")
|
||||
@@ -385,7 +386,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_restart_during_shutdown_is_no_op(self):
|
||||
specs = [_DaemonSpec("git-gate", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("git-gate", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
sup.request_shutdown(reason="test")
|
||||
@@ -395,7 +396,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_pending_restart_dropped_during_shutdown(self):
|
||||
specs = [_DaemonSpec("git-gate", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("git-gate", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
time.sleep(0.1)
|
||||
@@ -413,8 +414,8 @@ class TestSupervisor(unittest.TestCase):
|
||||
# both should receive SIGTERM and exit. Signal-only
|
||||
# shutdown clamps to a zero supervisor exit code.
|
||||
specs = [
|
||||
_DaemonSpec("a", ("/bin/sleep", "60")),
|
||||
_DaemonSpec("b", ("/bin/sleep", "60")),
|
||||
_DaemonSpec("a", (SLEEP, "60")),
|
||||
_DaemonSpec("b", (SLEEP, "60")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -449,7 +450,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self.assertEqual(0, rc)
|
||||
|
||||
def test_idempotent_shutdown_requests(self):
|
||||
specs = [_DaemonSpec("a", ("/bin/sleep", "60"))]
|
||||
specs = [_DaemonSpec("a", (SLEEP, "60"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
time.sleep(0.1)
|
||||
@@ -470,7 +471,7 @@ class TestMainEndToEnd(unittest.TestCase):
|
||||
|
||||
@classmethod
|
||||
def setUpClass(cls):
|
||||
for p in ("/bin/sh", "/bin/sleep"):
|
||||
for p in ("/bin/sh", SLEEP):
|
||||
if not Path(p).exists():
|
||||
raise unittest.SkipTest(f"missing {p}")
|
||||
|
||||
@@ -485,8 +486,8 @@ class TestMainEndToEnd(unittest.TestCase):
|
||||
"import os, runpy, sys\n"
|
||||
"from bot_bottle import gateway_init as si\n"
|
||||
"si._DAEMONS = (\n"
|
||||
" si._DaemonSpec('alpha', ('/bin/sleep','30')),\n"
|
||||
" si._DaemonSpec('beta', ('/bin/sleep','30')),\n"
|
||||
f" si._DaemonSpec('alpha', ({SLEEP!r},'30')),\n"
|
||||
f" si._DaemonSpec('beta', ({SLEEP!r},'30')),\n"
|
||||
")\n"
|
||||
"sys.exit(si.main([]))\n"
|
||||
)
|
||||
|
||||
@@ -168,16 +168,18 @@ class TestHookRender(unittest.TestCase):
|
||||
# Stdin is buffered to a tempfile so both phases can re-read.
|
||||
self.assertIn("refs_file=$(mktemp)", hook)
|
||||
|
||||
def test_new_ref_scan_scoped_to_incoming_commits(self):
|
||||
# A new branch (old=all-zeros) must scan only commits new to the
|
||||
# gate, not the full ancestry — otherwise historical findings
|
||||
# block every new-branch push (PRD 0028 / issue #106).
|
||||
def test_scan_scoped_to_incoming_commits(self):
|
||||
# Every non-delete push scans only commits new to the gate, not
|
||||
# the full ancestry and not the `$old..$new` delta — otherwise
|
||||
# historical fixtures block new-branch pushes (PRD 0028 / #106)
|
||||
# and a rebase/force-push onto an advanced main drags in main's
|
||||
# history incl. the sandbox-escape fixtures (#346).
|
||||
hook = git_gate_render_hook()
|
||||
self.assertIn('log_opts="$new --not --all"', hook)
|
||||
# The old over-broad full-ancestry range must be gone.
|
||||
# Neither the full-ancestry range nor the ancestry-blind delta
|
||||
# range may survive.
|
||||
self.assertNotIn('log_opts="$new"', hook)
|
||||
# Existing-branch delta scan is unchanged.
|
||||
self.assertIn('log_opts="$old..$new"', hook)
|
||||
self.assertNotIn('log_opts="$old..$new"', hook)
|
||||
|
||||
def test_forward_ssh_is_non_interactive_and_bounded(self):
|
||||
# No prompt (BatchMode) and a connect timeout, so an unreachable
|
||||
|
||||
@@ -50,7 +50,12 @@ class _CacheMixin(unittest.TestCase):
|
||||
self._env = mock.patch.dict(
|
||||
os.environ,
|
||||
{"BOT_BOTTLE_FC_CACHE": self._tmp.name,
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_TOKEN": ""},
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_TOKEN": "",
|
||||
# Pin the candidate-dir override off: the coverage CI job exports a
|
||||
# candidate dir for the integration suite, and an ambient value
|
||||
# would send these registry-pull tests down the local-bundle path.
|
||||
# Cases that exercise the candidate path set it explicitly.
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": ""},
|
||||
clear=False,
|
||||
)
|
||||
self._env.start()
|
||||
@@ -93,6 +98,31 @@ class TestVersionInputs(unittest.TestCase):
|
||||
(pkg / "netpool.defaults.env").write_text("FOO=1\n")
|
||||
for name in ("Dockerfile.orchestrator", "Dockerfile.gateway", "Dockerfile.infra"):
|
||||
(root / name).write_text(f"FROM scratch # {name}\n")
|
||||
(root / "pyproject.toml").write_text("[project]\nname = 'bot-bottle'\n")
|
||||
|
||||
def test_pyproject_toml_change_bumps_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._fake_repo(root)
|
||||
before = ia.infra_artifact_version("init", repo_root=root)
|
||||
(root / "pyproject.toml").write_text(
|
||||
"[project]\nname = 'bot-bottle'\ndependencies = ['httpx']\n")
|
||||
after = ia.infra_artifact_version("init", repo_root=root)
|
||||
self.assertNotEqual(before, after)
|
||||
|
||||
def test_dropbear_change_bumps_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._fake_repo(root)
|
||||
dropbear = root / "dropbear"
|
||||
dropbear.write_bytes(b"dropbear-v1")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_FC_DROPBEAR": str(dropbear),
|
||||
}):
|
||||
before = ia.infra_artifact_version("init", repo_root=root)
|
||||
dropbear.write_bytes(b"dropbear-v2")
|
||||
after = ia.infra_artifact_version("init", repo_root=root)
|
||||
self.assertNotEqual(before, after)
|
||||
|
||||
def test_non_python_file_change_bumps_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
@@ -118,6 +148,58 @@ class TestVersionInputs(unittest.TestCase):
|
||||
|
||||
|
||||
class TestEnsureArtifact(_CacheMixin):
|
||||
def test_uses_verified_ci_candidate_without_network(self) -> None:
|
||||
version = "deadbeef00000000"
|
||||
gz = _gz(b"candidate ext4")
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
(root / "version.txt").write_text(version + "\n")
|
||||
(root / "rootfs.ext4.gz").write_bytes(gz)
|
||||
digest = hashlib.sha256(gz).hexdigest()
|
||||
(root / "rootfs.ext4.gz.sha256").write_text(
|
||||
f"{digest} rootfs.ext4.gz\n")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": str(root),
|
||||
}), mock.patch.object(ia.urllib.request, "urlopen") as net:
|
||||
path = ia.ensure_artifact_gz(version)
|
||||
self.assertEqual(root / "rootfs.ext4.gz", path)
|
||||
net.assert_not_called()
|
||||
|
||||
def test_rejects_candidate_for_another_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
(root / "version.txt").write_text("wrong\n")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": str(root),
|
||||
}):
|
||||
with self.assertRaises(Die) as ctx:
|
||||
ia.ensure_artifact_gz("expected")
|
||||
self.assertIn("version mismatch", str(ctx.exception.message))
|
||||
|
||||
def test_rejects_incomplete_candidate(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
(root / "version.txt").write_text("v1\n")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": str(root),
|
||||
}):
|
||||
with self.assertRaises(Die) as ctx:
|
||||
ia.ensure_artifact_gz("v1")
|
||||
self.assertIn("incomplete", str(ctx.exception.message))
|
||||
|
||||
def test_rejects_candidate_checksum_mismatch(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
(root / "version.txt").write_text("v1\n")
|
||||
(root / "rootfs.ext4.gz").write_bytes(b"bad")
|
||||
(root / "rootfs.ext4.gz.sha256").write_text("0" * 64 + " rootfs.ext4.gz\n")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": str(root),
|
||||
}):
|
||||
with self.assertRaises(Die) as ctx:
|
||||
ia.ensure_artifact_gz("v1")
|
||||
self.assertIn("checksum mismatch", str(ctx.exception.message))
|
||||
|
||||
def test_downloads_verifies_and_caches(self) -> None:
|
||||
version = "deadbeef00000000"
|
||||
gz = _gz(b"fake ext4 bytes")
|
||||
|
||||
@@ -17,10 +17,10 @@ from unittest.mock import patch
|
||||
from bot_bottle.backend.macos_container.bottle import MacosContainerBottle
|
||||
from bot_bottle.backend.macos_container.bottle_plan import MacosContainerBottlePlan
|
||||
from bot_bottle.backend.macos_container.consolidated_launch import GatewayEndpoint
|
||||
from bot_bottle.backend.macos_container.gateway_hosts import GATEWAY_HOSTNAME
|
||||
from bot_bottle.backend.macos_container.launch import (
|
||||
_agent_run_argv,
|
||||
_identity_proxy_env,
|
||||
_proxy_url,
|
||||
)
|
||||
from bot_bottle.manifest import ManifestIndex
|
||||
|
||||
@@ -104,19 +104,38 @@ class TestAgentRunArgv(unittest.TestCase):
|
||||
read back after start."""
|
||||
self.assertNotIn("--ip", self.argv)
|
||||
|
||||
def test_run_time_proxy_carries_no_identity_token(self) -> None:
|
||||
"""The token is minted by registration, which happens after this run —
|
||||
so it cannot be here. `/resolve` denies the token-less pair (#366),
|
||||
which is the safe direction; the real value arrives at exec time."""
|
||||
joined = " ".join(self.argv)
|
||||
self.assertIn(f"HTTP_PROXY={_proxy_url('192.168.128.3')}", joined)
|
||||
self.assertNotIn("bottle:", joined)
|
||||
def test_run_time_env_sets_no_proxy_vars_at_all(self) -> None:
|
||||
"""Regression: the proxy vars must exist *only* in the exec-time env.
|
||||
|
||||
`container exec --env` appends rather than replaces, so a token-less
|
||||
`HTTPS_PROXY` here would survive next to the token-bearing one and
|
||||
leave two entries in the agent's `environ`. Resolution is then
|
||||
runtime-specific — Node reads the last, Rust's `std::env::var` reads
|
||||
the first — so Codex proxied without its identity token and every
|
||||
request fail-closed at `/resolve`.
|
||||
"""
|
||||
for entry in self.argv:
|
||||
self.assertFalse(
|
||||
entry.upper().startswith(("HTTP_PROXY=", "HTTPS_PROXY=")),
|
||||
f"run-time env must not set a proxy var, got {entry!r}",
|
||||
)
|
||||
|
||||
def test_run_time_env_carries_no_identity_token(self) -> None:
|
||||
"""The token is minted by registration, which happens after this run,
|
||||
so it cannot be here under any name."""
|
||||
self.assertNotIn("bottle:", " ".join(self.argv))
|
||||
|
||||
def test_gateway_bypasses_the_proxy(self) -> None:
|
||||
"""git-http + supervise live on the gateway and must be reached
|
||||
directly, not through its own egress proxy."""
|
||||
entry = next(a for a in self.argv if a.startswith("NO_PROXY="))
|
||||
self.assertIn("192.168.128.3", entry)
|
||||
self.assertIn(GATEWAY_HOSTNAME, entry)
|
||||
|
||||
def test_no_proxy_names_the_gateway_and_never_addresses_it(self) -> None:
|
||||
"""NO_PROXY is baked into the run-time env, so an address here is as
|
||||
unfixable as the proxy URL if the gateway moves."""
|
||||
entry = next(a for a in self.argv if a.startswith("NO_PROXY="))
|
||||
self.assertNotIn("192.168.128.3", entry)
|
||||
|
||||
def test_forwarded_secrets_stay_off_argv(self) -> None:
|
||||
"""Bare name → inherited from the run process env, so the value never
|
||||
@@ -134,7 +153,7 @@ class TestIdentityTokenDelivery(unittest.TestCase):
|
||||
def test_exec_env_carries_the_token_as_proxy_credentials(self) -> None:
|
||||
env = _identity_proxy_env(_endpoint(), "s3cret")
|
||||
self.assertEqual(
|
||||
"http://bottle:s3cret@192.168.128.3:9099", env["HTTP_PROXY"],
|
||||
f"http://bottle:s3cret@{GATEWAY_HOSTNAME}:9099", env["HTTP_PROXY"],
|
||||
)
|
||||
self.assertEqual(env["HTTP_PROXY"], env["https_proxy"])
|
||||
|
||||
@@ -202,7 +221,7 @@ class TestPlanIdentityToken(unittest.TestCase):
|
||||
self.assertIn("HTTP_PROXY", argv)
|
||||
self.assertNotIn("s3cret", " ".join(argv))
|
||||
self.assertEqual(
|
||||
"http://bottle:s3cret@192.168.128.3:9099", kwargs["env"]["HTTP_PROXY"],
|
||||
f"http://bottle:s3cret@{GATEWAY_HOSTNAME}:9099", kwargs["env"]["HTTP_PROXY"],
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -272,6 +272,30 @@ resolver #2
|
||||
),
|
||||
)
|
||||
|
||||
def test_exec_container_as_root_selects_root_user(self):
|
||||
completed = util.subprocess.CompletedProcess(
|
||||
args=[], returncode=0, stdout="", stderr="",
|
||||
)
|
||||
with patch.object(util, "_run_container_op", return_value=completed) as run:
|
||||
util.exec_container_as_root("bot-bottle-demo", ["true"])
|
||||
|
||||
run.assert_called_once_with([
|
||||
"container", "exec", "--user", "root", "bot-bottle-demo", "true",
|
||||
])
|
||||
|
||||
def test_exec_container_as_root_reports_failure(self):
|
||||
failed = util.subprocess.CompletedProcess(
|
||||
args=[], returncode=1, stdout="", stderr="permission denied\n",
|
||||
)
|
||||
with patch.object(util, "_run_container_op", return_value=failed), \
|
||||
patch.object(util, "die", side_effect=SystemExit("die")) as die:
|
||||
with self.assertRaises(SystemExit):
|
||||
util.exec_container_as_root("bot-bottle-demo", ["true"])
|
||||
|
||||
die.assert_called_once_with(
|
||||
"container exec (root) in bot-bottle-demo failed: permission denied",
|
||||
)
|
||||
|
||||
|
||||
def _completed(stdout: str, returncode: int = 0):
|
||||
return util.subprocess.CompletedProcess(args=[], returncode=returncode, stdout=stdout, stderr="")
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
"""Unit: stable gateway name via each bottle's /etc/hosts (issue #443).
|
||||
|
||||
The gateway's address moves whenever the infra container is recreated. Agents
|
||||
name it instead of addressing it, and the name resolves through `/etc/hosts` —
|
||||
a file, so it stays rewritable while the bottle runs, unlike the `environ` the
|
||||
proxy URL is delivered in.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import unittest
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import patch
|
||||
|
||||
from bot_bottle.backend.macos_container.gateway_hosts import (
|
||||
GATEWAY_HOSTNAME,
|
||||
refresh_gateway_host,
|
||||
set_gateway_host,
|
||||
)
|
||||
|
||||
_MOD = "bot_bottle.backend.macos_container.gateway_hosts"
|
||||
|
||||
|
||||
class TestSetGatewayHost(unittest.TestCase):
|
||||
def _script(self, exec_root: object) -> str:
|
||||
argv = exec_root.call_args.args[1] # type: ignore[attr-defined]
|
||||
self.assertEqual(["sh", "-c"], argv[:2])
|
||||
return argv[2]
|
||||
|
||||
def test_writes_the_address_against_the_stable_name(self) -> None:
|
||||
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
|
||||
set_gateway_host("bot-bottle-demo", "192.168.128.19")
|
||||
self.assertEqual("bot-bottle-demo", ex.call_args.args[0])
|
||||
script = self._script(ex)
|
||||
self.assertIn("192.168.128.19", script)
|
||||
self.assertIn(GATEWAY_HOSTNAME, script)
|
||||
|
||||
def test_runs_as_root_so_the_agent_cannot_repoint_itself(self) -> None:
|
||||
"""The agent runs as `node`. If it could rewrite /etc/hosts it could
|
||||
aim its own gateway name elsewhere, so the write must go through the
|
||||
root-only helper."""
|
||||
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
|
||||
set_gateway_host("bot-bottle-demo", "10.0.0.1")
|
||||
ex.assert_called_once()
|
||||
|
||||
def test_is_idempotent_by_removing_its_own_line_first(self) -> None:
|
||||
"""Re-pointing must replace the managed entry, not append a second one
|
||||
— two entries for the same name would resolve by luck of ordering."""
|
||||
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
|
||||
set_gateway_host("bot-bottle-demo", "10.0.0.1")
|
||||
self.assertIn("grep -v", self._script(ex))
|
||||
|
||||
def test_preserves_the_rest_of_the_hosts_file(self) -> None:
|
||||
"""localhost and the container's own name must survive the rewrite."""
|
||||
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
|
||||
set_gateway_host("bot-bottle-demo", "10.0.0.1")
|
||||
script = self._script(ex)
|
||||
# Filter-and-append, never a truncating write of just our line.
|
||||
self.assertIn("/etc/hosts >", script)
|
||||
self.assertIn(">> /tmp/.bb-hosts", script)
|
||||
|
||||
def test_keeps_the_original_inode(self) -> None:
|
||||
"""`cat >` rather than `mv`: a pre-created /etc/hosts must keep its
|
||||
ownership and mode, not be replaced by a root-owned copy."""
|
||||
script = None
|
||||
with patch(f"{_MOD}.container_mod.exec_container_as_root") as ex:
|
||||
set_gateway_host("bot-bottle-demo", "10.0.0.1")
|
||||
script = self._script(ex)
|
||||
self.assertIn("cat /tmp/.bb-hosts > /etc/hosts", script)
|
||||
self.assertNotIn("mv ", script)
|
||||
|
||||
|
||||
class TestRefreshGatewayHost(unittest.TestCase):
|
||||
"""The re-attach sweep: bottles stranded by an earlier gateway restart get
|
||||
re-pointed in place instead of needing a relaunch."""
|
||||
|
||||
def _agents(self, *slugs: str) -> list[SimpleNamespace]:
|
||||
return [SimpleNamespace(slug=s) for s in slugs]
|
||||
|
||||
def test_repoints_every_running_bottle(self) -> None:
|
||||
with patch(f"{_MOD}.enumerate_active", return_value=self._agents("a", "b")), \
|
||||
patch(f"{_MOD}.set_gateway_host") as setter:
|
||||
updated = refresh_gateway_host("192.168.128.19")
|
||||
self.assertEqual(["bot-bottle-a", "bot-bottle-b"], updated)
|
||||
self.assertEqual(
|
||||
[("bot-bottle-a", "192.168.128.19"), ("bot-bottle-b", "192.168.128.19")],
|
||||
[c.args for c in setter.call_args_list],
|
||||
)
|
||||
|
||||
def test_one_failing_bottle_does_not_stop_the_sweep(self) -> None:
|
||||
"""A container that is already exiting must not block the repair of
|
||||
its neighbours, nor fail the launch that triggered the sweep."""
|
||||
def _flaky(name: str, _ip: str) -> None:
|
||||
if name == "bot-bottle-a":
|
||||
raise RuntimeError("container is exiting")
|
||||
|
||||
with patch(f"{_MOD}.enumerate_active", return_value=self._agents("a", "b")), \
|
||||
patch(f"{_MOD}.set_gateway_host", side_effect=_flaky), \
|
||||
patch(f"{_MOD}.warn") as warn:
|
||||
updated = refresh_gateway_host("10.0.0.1")
|
||||
self.assertEqual(["bot-bottle-b"], updated)
|
||||
warn.assert_called_once()
|
||||
|
||||
def test_no_running_bottles_is_a_clean_no_op(self) -> None:
|
||||
with patch(f"{_MOD}.enumerate_active", return_value=[]), \
|
||||
patch(f"{_MOD}.set_gateway_host") as setter:
|
||||
self.assertEqual([], refresh_gateway_host("10.0.0.1"))
|
||||
setter.assert_not_called()
|
||||
|
||||
|
||||
|
||||
class TestLaunchWiring(unittest.TestCase):
|
||||
"""Ordering matters: the name must resolve before anything execs, and the
|
||||
stranded-bottle sweep must run once the gateway is known to be up."""
|
||||
|
||||
def test_launch_sets_the_host_entry_before_reading_the_source_ip(self) -> None:
|
||||
"""The agent's every URL names the gateway, so the entry has to exist
|
||||
before the first connection — which means before the agent execs."""
|
||||
import inspect
|
||||
|
||||
from bot_bottle.backend.macos_container import launch
|
||||
|
||||
src = inspect.getsource(launch)
|
||||
set_at = src.index("set_gateway_host(plan.container_name")
|
||||
exec_at = src.index("wait_container_ipv4_on_network")
|
||||
self.assertLess(set_at, exec_at)
|
||||
|
||||
def test_launch_refreshes_stranded_bottles_after_ensure_gateway(self) -> None:
|
||||
import inspect
|
||||
|
||||
from bot_bottle.backend.macos_container import launch
|
||||
|
||||
src = inspect.getsource(launch)
|
||||
ensure_at = src.index("endpoint = ensure_gateway()")
|
||||
refresh_at = src.index("refresh_gateway_host(endpoint.gateway_ip)")
|
||||
self.assertLess(ensure_at, refresh_at)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -46,7 +46,9 @@ class TestInfraRun(unittest.TestCase):
|
||||
argv = self._run_container(MacosInfraService(repo_root=Path("/r")))
|
||||
script = argv[-1]
|
||||
self.assertIn("bot_bottle.orchestrator", script)
|
||||
self.assertIn("gateway_init.py", script)
|
||||
# Gateway launches via the installed package (there is no
|
||||
# /app/gateway_init.py file since the daemons moved into bot_bottle).
|
||||
self.assertIn("bot_bottle.gateway_init", script)
|
||||
self.assertIn("127.0.0.1", script) # they reach each other on loopback
|
||||
|
||||
def test_db_is_a_container_only_volume(self) -> None:
|
||||
|
||||
@@ -80,19 +80,11 @@ class TestAgentProviderHostCredentials(unittest.TestCase):
|
||||
"forward_host_credentials": "yes",
|
||||
})
|
||||
|
||||
def test_forward_host_credentials_allowed_for_claude(self):
|
||||
b = _provider_config_bottle({
|
||||
"template": "claude",
|
||||
"forward_host_credentials": True,
|
||||
})
|
||||
self.assertTrue(b.agent_provider.forward_host_credentials)
|
||||
|
||||
def test_forward_host_credentials_and_auth_token_rejected_together(self):
|
||||
def test_forward_host_credentials_rejected_for_claude(self):
|
||||
with self.assertRaises(ManifestError):
|
||||
_provider_config_bottle({
|
||||
"template": "claude",
|
||||
"forward_host_credentials": True,
|
||||
"auth_token": "SOME_TOKEN",
|
||||
})
|
||||
|
||||
def test_auth_token_defaults_empty(self):
|
||||
|
||||
@@ -86,22 +86,10 @@ class TestAgentProviderValidation(unittest.TestCase):
|
||||
"b", {"forward_host_credentials": True, "template": "weird"}
|
||||
)
|
||||
|
||||
def test_forward_creds_pi_template_rejected(self) -> None:
|
||||
def test_forward_creds_non_codex_template(self) -> None:
|
||||
with self.assertRaises(ManifestError):
|
||||
ManifestAgentProvider.from_dict(
|
||||
"b", {"forward_host_credentials": True, "template": "pi"}
|
||||
)
|
||||
|
||||
def test_forward_creds_claude_allowed(self) -> None:
|
||||
p = ManifestAgentProvider.from_dict(
|
||||
"b", {"forward_host_credentials": True, "template": "claude"}
|
||||
)
|
||||
self.assertTrue(p.forward_host_credentials)
|
||||
|
||||
def test_forward_creds_and_auth_token_rejected(self) -> None:
|
||||
with self.assertRaises(ManifestError):
|
||||
ManifestAgentProvider.from_dict(
|
||||
"b", {"forward_host_credentials": True, "auth_token": "T", "template": "claude"}
|
||||
"b", {"forward_host_credentials": True, "template": "claude"}
|
||||
)
|
||||
|
||||
def test_valid_claude_auth_token(self) -> None:
|
||||
|
||||
@@ -6,8 +6,11 @@ read it into memory. Network is mocked; no Docker, no real build.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
from email.message import Message
|
||||
import tempfile
|
||||
import unittest
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
@@ -58,5 +61,99 @@ class TestPut(unittest.TestCase):
|
||||
self.assertEqual(b"abc123 rootfs\n", captured[0].data)
|
||||
|
||||
|
||||
class TestPublishBundle(unittest.TestCase):
|
||||
def _bundle(self, root: Path, version: str) -> None:
|
||||
payload = b"candidate"
|
||||
(root / "version.txt").write_text(version + "\n")
|
||||
(root / "rootfs.ext4.gz").write_bytes(payload)
|
||||
digest = hashlib.sha256(payload).hexdigest()
|
||||
(root / "rootfs.ext4.gz.sha256").write_text(
|
||||
f"{digest} rootfs.ext4.gz\n")
|
||||
|
||||
def test_existing_identical_artifact_is_success(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "v1")
|
||||
sha = (root / "rootfs.ext4.gz.sha256").read_bytes()
|
||||
response = mock.MagicMock()
|
||||
response.__enter__.return_value.read.return_value = sha
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="v1"
|
||||
), mock.patch.object(
|
||||
pub.urllib.request, "urlopen", return_value=response
|
||||
), mock.patch.object(pub, "_put") as put:
|
||||
self.assertEqual("v1", pub._publish_bundle(root, "token"))
|
||||
put.assert_not_called()
|
||||
|
||||
def test_partial_artifact_is_replaced(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "v1")
|
||||
missing = urllib.error.HTTPError("u", 404, "missing", Message(), None)
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="v1"
|
||||
), mock.patch.object(
|
||||
pub.urllib.request, "urlopen", side_effect=missing
|
||||
), mock.patch.object(pub, "_delete") as delete, \
|
||||
mock.patch.object(pub, "_put") as put:
|
||||
pub._publish_bundle(root, "token")
|
||||
self.assertEqual(3, delete.call_count)
|
||||
self.assertEqual(3, put.call_count)
|
||||
|
||||
def test_rejects_bundle_for_different_checkout(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "old")
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="new"
|
||||
):
|
||||
with self.assertRaises(SystemExit) as ctx:
|
||||
pub._publish_bundle(root, "token")
|
||||
self.assertIn("does not match checkout", str(ctx.exception))
|
||||
|
||||
def test_rejects_bad_bundle_checksum(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "v1")
|
||||
(root / "rootfs.ext4.gz").write_bytes(b"tampered")
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="v1"
|
||||
):
|
||||
with self.assertRaises(SystemExit) as ctx:
|
||||
pub._publish_bundle(root, "token")
|
||||
self.assertIn("checksum mismatch", str(ctx.exception))
|
||||
|
||||
def test_registry_lookup_failure_is_reported(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "v1")
|
||||
failure = urllib.error.URLError("offline")
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="v1"
|
||||
), mock.patch.object(pub.urllib.request, "urlopen", side_effect=failure):
|
||||
with self.assertRaises(SystemExit) as ctx:
|
||||
pub._publish_bundle(root, "token")
|
||||
self.assertIn("registry unreachable", str(ctx.exception))
|
||||
|
||||
|
||||
class TestMain(unittest.TestCase):
|
||||
def test_output_builds_candidate_and_records_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d) / "candidate"
|
||||
with mock.patch.object(
|
||||
pub, "build_artifact", return_value=("v1", root / "g", root / "s")
|
||||
) as build:
|
||||
self.assertEqual(0, pub.main(["--output", str(root)]))
|
||||
build.assert_called_once_with(root)
|
||||
self.assertEqual("v1\n", (root / "version.txt").read_text())
|
||||
|
||||
def test_publish_dir_publishes_existing_candidate(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d, \
|
||||
mock.patch.object(pub.infra_artifact, "_config", return_value=("", "", "t")), \
|
||||
mock.patch.object(pub, "_publish_bundle", return_value="v1") as publish:
|
||||
self.assertEqual(0, pub.main(["--publish-dir", d]))
|
||||
publish.assert_called_once_with(Path(d), "t")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -32,7 +32,9 @@ from bot_bottle.supervise_server import (
|
||||
_RpcError,
|
||||
_RpcInternalError,
|
||||
_response_timeout_from_env,
|
||||
format_pending_response_text,
|
||||
format_response_text,
|
||||
handle_check_proposal,
|
||||
handle_initialize,
|
||||
handle_tools_call,
|
||||
handle_tools_list,
|
||||
@@ -218,6 +220,7 @@ class TestHandleToolsList(unittest.TestCase):
|
||||
_sv.TOOL_EGRESS_ALLOW,
|
||||
_sv.TOOL_EGRESS_BLOCK,
|
||||
_sv.TOOL_LIST_EGRESS_ROUTES,
|
||||
_sv.TOOL_CHECK_PROPOSAL,
|
||||
]),
|
||||
sorted(names),
|
||||
)
|
||||
@@ -484,9 +487,10 @@ class TestFormatResponseText(unittest.TestCase):
|
||||
|
||||
class TestFormatPendingResponseText(unittest.TestCase):
|
||||
def test_formats_timeout_message(self):
|
||||
text = supervise_server.format_pending_response_text(12.5)
|
||||
text = supervise_server.format_pending_response_text("prop-9", 12.5)
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn("12.5s", text)
|
||||
self.assertIn("proposal_id: prop-9", text)
|
||||
|
||||
|
||||
# --- End-to-end HTTP sanity ------------------------------------------------
|
||||
@@ -685,5 +689,129 @@ class TestResolvedRoutesPayload(unittest.TestCase):
|
||||
_handler(None)._resolved_routes_payload()
|
||||
|
||||
|
||||
class TestNonBlockingSupervise(unittest.TestCase):
|
||||
"""PRD prd-new / issue #412: pending responses carry the proposal id, and
|
||||
`check-proposal` polls a queued proposal without blocking or re-proposing."""
|
||||
|
||||
_ROUTES = "routes:\n - host: example.com\n"
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory(prefix="supervise-nonblock-test.")
|
||||
self._home_patch = use_bottle_root(Path(self._tmp.name) / ".bot-bottle")
|
||||
self.config = ServerConfig(bottle_slug="dev")
|
||||
_qs.QueueStore("dev").migrate()
|
||||
_as.AuditStore().migrate()
|
||||
|
||||
def tearDown(self):
|
||||
self._home_patch()
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _seed_proposal(self) -> "_sv.Proposal":
|
||||
p = _sv.Proposal.new(
|
||||
bottle_slug="dev",
|
||||
tool=_sv.TOOL_EGRESS_ALLOW,
|
||||
proposed_file=self._ROUTES,
|
||||
justification="need example.com",
|
||||
current_file_hash=_sv.sha256_hex(self._ROUTES),
|
||||
)
|
||||
_sv.write_proposal(p)
|
||||
return p
|
||||
|
||||
def _check(self, proposal_id: str) -> dict[str, object]:
|
||||
return handle_check_proposal({"arguments": {"proposal_id": proposal_id}}, self.config)
|
||||
|
||||
# --- pending response carries the id ---
|
||||
|
||||
def test_pending_text_includes_id_and_pointer(self):
|
||||
text = format_pending_response_text("abc-123", 30.0)
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn("proposal_id: abc-123", text)
|
||||
self.assertIn("check-proposal", text)
|
||||
|
||||
def test_tools_call_timeout_returns_pending_with_id_and_stays_queued(self):
|
||||
# No responder → the grace window expires → pending, not blocked forever.
|
||||
result = handle_tools_call(
|
||||
{
|
||||
"name": _sv.TOOL_EGRESS_ALLOW,
|
||||
"arguments": {"routes_yaml": self._ROUTES, "justification": "x"},
|
||||
},
|
||||
ServerConfig(bottle_slug="dev", response_timeout_seconds=0.05),
|
||||
)
|
||||
self.assertFalse(result["isError"]) # type: ignore[index]
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: pending", text)
|
||||
pending = _sv.list_pending_proposals("dev")
|
||||
self.assertEqual(1, len(pending)) # still queued, not archived
|
||||
self.assertIn(pending[0].id, text) # agent got the id to poll
|
||||
|
||||
# --- check-proposal poll ---
|
||||
|
||||
def test_check_returns_approved_and_archives(self):
|
||||
p = self._seed_proposal()
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=p.id, status=_sv.STATUS_APPROVED, notes="ok"))
|
||||
result = self._check(p.id)
|
||||
self.assertFalse(result["isError"])
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: approved", text)
|
||||
self.assertIn("notes: ok", text)
|
||||
with self.assertRaises(FileNotFoundError): # archived on read
|
||||
_sv.read_proposal("dev", p.id)
|
||||
|
||||
def test_check_rejected_sets_isError(self):
|
||||
p = self._seed_proposal()
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=p.id, status=_sv.STATUS_REJECTED, notes="no"))
|
||||
result = self._check(p.id)
|
||||
self.assertTrue(result["isError"])
|
||||
self.assertIn("status: rejected", result["content"][0]["text"]) # type: ignore[index]
|
||||
|
||||
def test_check_pending_when_no_decision_yet(self):
|
||||
p = self._seed_proposal()
|
||||
result = self._check(p.id)
|
||||
self.assertFalse(result["isError"])
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn(p.id, text)
|
||||
self.assertEqual(1, len(_sv.list_pending_proposals("dev"))) # not archived
|
||||
|
||||
def test_check_unknown_id_is_error(self):
|
||||
result = self._check("no-such-proposal")
|
||||
self.assertTrue(result["isError"])
|
||||
self.assertIn("status: unknown", result["content"][0]["text"]) # type: ignore[index]
|
||||
|
||||
def test_check_missing_id_raises(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": {}}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_check_empty_id_raises(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": {"proposal_id": " "}}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_check_arguments_must_be_object(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": []}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_full_nonblocking_round_trip(self):
|
||||
# 1. tools/call times out → pending with id
|
||||
result = handle_tools_call(
|
||||
{
|
||||
"name": _sv.TOOL_EGRESS_ALLOW,
|
||||
"arguments": {"routes_yaml": self._ROUTES, "justification": "x"},
|
||||
},
|
||||
ServerConfig(bottle_slug="dev", response_timeout_seconds=0.05),
|
||||
)
|
||||
pid = _sv.list_pending_proposals("dev")[0].id
|
||||
self.assertIn(pid, result["content"][0]["text"]) # type: ignore[index]
|
||||
# 2. operator decides out-of-band
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=pid, status=_sv.STATUS_APPROVED, notes="ok"))
|
||||
# 3. agent resumes by polling — no re-proposing
|
||||
poll = self._check(pid)
|
||||
self.assertFalse(poll["isError"])
|
||||
self.assertIn("status: approved", poll["content"][0]["text"]) # type: ignore[index]
|
||||
self.assertEqual([], _sv.list_pending_proposals("dev")) # resolved + archived
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
import unittest
|
||||
from unittest.mock import Mock
|
||||
|
||||
from scripts.tracker_policy import (
|
||||
TRIAGE_LABEL,
|
||||
check_pull_request,
|
||||
deliberate_issue_numbers,
|
||||
ensure_issue_label,
|
||||
)
|
||||
|
||||
|
||||
class TestDeliberateIssueNumbers(unittest.TestCase):
|
||||
def test_accepts_completing_and_noncompleting_forms(self):
|
||||
self.assertEqual(
|
||||
deliberate_issue_numbers("Fixes #12", "Part of #14; refs #15"),
|
||||
{12, 14, 15},
|
||||
)
|
||||
|
||||
def test_does_not_treat_incidental_number_as_link(self):
|
||||
self.assertEqual(deliberate_issue_numbers("Audit #12", "See PR #14"), set())
|
||||
|
||||
|
||||
class TestCheckPullRequest(unittest.TestCase):
|
||||
def test_accepts_unlabelled_pr_linked_to_real_issue(self):
|
||||
api = Mock()
|
||||
api.request.return_value = {"number": 12, "pull_request": None}
|
||||
event = {"pull_request": {"title": "Change", "body": "Part of #12", "labels": []}}
|
||||
self.assertEqual(check_pull_request(event, api), [])
|
||||
|
||||
def test_rejects_labels_and_pr_reference(self):
|
||||
api = Mock()
|
||||
api.request.return_value = {"number": 12, "pull_request": {}}
|
||||
event = {
|
||||
"pull_request": {
|
||||
"title": "Change",
|
||||
"body": "Closes #12",
|
||||
"labels": [{"name": "Kind/Bug"}],
|
||||
}
|
||||
}
|
||||
errors = check_pull_request(event, api)
|
||||
self.assertEqual(len(errors), 2)
|
||||
self.assertIn("unlabeled", errors[0])
|
||||
self.assertIn("not an issue", errors[1])
|
||||
|
||||
|
||||
class TestEnsureIssueLabel(unittest.TestCase):
|
||||
def test_adds_triage_label_to_unlabelled_issue(self):
|
||||
api = Mock()
|
||||
api.request.side_effect = [[{"id": 55, "name": TRIAGE_LABEL}], None]
|
||||
event = {"issue": {"number": 405, "labels": [], "pull_request": None}}
|
||||
self.assertTrue(ensure_issue_label(event, api))
|
||||
api.request.assert_any_call("POST", "/issues/405/labels", {"labels": [55]})
|
||||
|
||||
def test_leaves_labelled_issue_unchanged(self):
|
||||
api = Mock()
|
||||
event = {"issue": {"number": 405, "labels": [{"name": "Kind/Documentation"}]}}
|
||||
self.assertFalse(ensure_issue_label(event, api))
|
||||
api.request.assert_not_called()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user