Compare commits
43 Commits
e1f10fb9da
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
| 44479f328e | |||
| 2de223a33b | |||
| af1690ab22 | |||
| 09debcf4f0 | |||
| fa11ad9a4a | |||
| ad6471af12 | |||
| 137df6f853 | |||
| 4252ca3562 | |||
| 701f5bf5e3 | |||
| d589c08d9d | |||
| 559dc03bb5 | |||
| 9172bf3a42 | |||
| 0adbf25977 | |||
| d1aec706e3 | |||
| a589604aa0 | |||
| 4c01e31e96 | |||
| 6f885af4b4 | |||
| 127ba49372 | |||
| 0d696674e3 | |||
| 626f07efa6 | |||
| d117460192 | |||
| e72ec71047 | |||
| 7aff69fbe0 | |||
| 1d91db3e31 | |||
| 686ca0d74b | |||
| 6d44a1be0a | |||
| 32e85de16f | |||
| a1d2c4a500 | |||
| a6fe31a424 | |||
| 41b2b24b36 | |||
| 37045ca147 | |||
| 9b54cfa854 | |||
| c193b04338 | |||
| c7ab3e0957 | |||
| 034f774529 | |||
| 5b359fe8d2 | |||
| 015ff52eda | |||
| 4302678f3e | |||
| 3a6fbad057 | |||
| a800a417d9 | |||
| 293218035d | |||
| 727eafe0f9 | |||
| 1ec114b6d7 |
@@ -24,8 +24,19 @@ jobs:
|
||||
|
||||
- name: Run pylint
|
||||
run: |
|
||||
# Run pylint on all Python files in the repo
|
||||
find . -name '*.py' -not -path './.venv/*' -not -path './.git/*' | xargs pylint --fail-under=8.0
|
||||
# Pylint's normal exit code is nonzero for any emitted finding,
|
||||
# regardless of --fail-under. Preserve the full report but enforce
|
||||
# the aggregate score this workflow promises.
|
||||
set +e
|
||||
find . -name '*.py' -not -path './.venv/*' -not -path './.git/*' \
|
||||
| xargs pylint --fail-under=8.0 \
|
||||
| tee /tmp/pylint-output.txt
|
||||
set -e
|
||||
SCORE=$(sed -n \
|
||||
's/^Your code has been rated at \([-0-9.]*\)\/10.*/\1/p' \
|
||||
/tmp/pylint-output.txt | tail -1)
|
||||
test -n "$SCORE"
|
||||
awk -v score="$SCORE" 'BEGIN { exit !(score >= 8.0) }'
|
||||
|
||||
- name: Run pyright
|
||||
run: |
|
||||
|
||||
+114
-6
@@ -25,15 +25,68 @@ on:
|
||||
- '.gitea/workflows/**.yml'
|
||||
- 'scripts/**'
|
||||
- 'README.md'
|
||||
# Dockerfiles and pyproject.toml are baked into the infra rootfs; a
|
||||
# change here alters what the integration/coverage jobs build locally.
|
||||
- 'Dockerfile*'
|
||||
- 'pyproject.toml'
|
||||
pull_request:
|
||||
paths:
|
||||
- '**.py'
|
||||
- '.gitea/workflows/**.yml'
|
||||
- 'scripts/**'
|
||||
- 'README.md'
|
||||
- 'Dockerfile*'
|
||||
- 'pyproject.toml'
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
stage-firecracker-inputs:
|
||||
runs-on: [self-hosted, kvm]
|
||||
# Same guard as the other KVM-runner jobs: don't spin the privileged
|
||||
# runner for fork PRs (this only copies a non-secret static binary, but
|
||||
# keep the posture consistent — build-infra/integration/coverage all
|
||||
# depend on it, so gating here gates the whole Firecracker chain).
|
||||
if: >-
|
||||
github.event_name == 'push' ||
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'pull_request' &&
|
||||
github.event.pull_request.head.repo.full_name == github.repository)
|
||||
steps:
|
||||
- name: Stage the provisioned static dropbear
|
||||
run: |
|
||||
mkdir -p firecracker-inputs
|
||||
cp /var/cache/bot-bottle-fc/dropbear firecracker-inputs/dropbear
|
||||
|
||||
- name: Upload Firecracker build inputs
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: firecracker-inputs/
|
||||
|
||||
build-infra:
|
||||
needs: stage-firecracker-inputs
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Download Firecracker build inputs
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: firecracker-inputs
|
||||
|
||||
- name: Build infra candidate from this checkout
|
||||
env:
|
||||
BOT_BOTTLE_FC_DROPBEAR: ${{ github.workspace }}/firecracker-inputs/dropbear
|
||||
run: python3 -m bot_bottle.backend.firecracker.publish_infra --output infra-candidate
|
||||
|
||||
- name: Upload infra candidate
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate/
|
||||
|
||||
unit:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
@@ -83,9 +136,10 @@ jobs:
|
||||
# PRs don't execute untrusted code on the privileged runner.
|
||||
#
|
||||
# Runner prerequisites (provision once; see README "Firecracker on Linux"):
|
||||
# `firecracker` on PATH, `/dev/kvm` accessible, Docker, cached kernel +
|
||||
# `firecracker` on PATH, `/dev/kvm` accessible, cached kernel +
|
||||
# static dropbear, and the pool as a persistent systemd unit.
|
||||
integration-firecracker:
|
||||
needs: build-infra
|
||||
runs-on: [self-hosted, kvm]
|
||||
if: >-
|
||||
github.event_name == 'push' ||
|
||||
@@ -105,12 +159,23 @@ jobs:
|
||||
# range overlap; it prints the exact `backend setup` fix.
|
||||
python3 cli.py backend status --backend=firecracker
|
||||
|
||||
- name: Install dev requirements
|
||||
run: python3 -m pip install --user -r requirements-dev.txt
|
||||
- name: Download the candidate built from this checkout
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate
|
||||
|
||||
- name: Replace the persistent infra VM with the candidate
|
||||
run: python3 -c 'from bot_bottle.backend.firecracker import infra_vm; infra_vm.stop()'
|
||||
|
||||
# No dev-requirements install: the integration suite runs on stdlib
|
||||
# `unittest` (pylint/pyright are lint.yml's concern, not this job's),
|
||||
# and the self-hosted runner's Nix python env has no `pip` module
|
||||
# (`python3 -m pip` → "No module named pip"). Nothing to install.
|
||||
- name: Run integration tests (firecracker)
|
||||
env:
|
||||
BOT_BOTTLE_BACKEND: firecracker
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_DIR: ${{ github.workspace }}/infra-candidate
|
||||
run: python3 -m unittest discover -t . -s tests/integration -v
|
||||
|
||||
# Combined unit+integration coverage + the diff-coverage gate (the hard
|
||||
@@ -129,7 +194,13 @@ jobs:
|
||||
#
|
||||
# See #414 for the planned follow-up: artifact-based coverage combination
|
||||
# (run tests once in their respective jobs, combine .coverage files here).
|
||||
#
|
||||
# build-infra creates one candidate from the checkout. This job boots that
|
||||
# same candidate after integration-firecracker has exercised it; the main
|
||||
# push path publishes the identical bytes only after every required job.
|
||||
coverage:
|
||||
needs: [build-infra, integration-firecracker]
|
||||
timeout-minutes: 15
|
||||
runs-on: [self-hosted, kvm]
|
||||
if: >-
|
||||
github.event_name == 'push' ||
|
||||
@@ -151,15 +222,52 @@ jobs:
|
||||
# range overlap; it prints the exact `backend setup` fix.
|
||||
python3 cli.py backend status --backend=firecracker
|
||||
|
||||
- name: Install dev requirements
|
||||
run: python3 -m pip install --user -r requirements-dev.txt
|
||||
- name: Download the candidate already exercised by integration
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate
|
||||
|
||||
# No dev-requirements install: `coverage` is already provided by the
|
||||
# self-hosted runner's Nix python env, and that env has no `pip`
|
||||
# module to install into anyway. `scripts/coverage.sh` +
|
||||
# `diff_coverage.py` need only `coverage` (not pylint/pyright).
|
||||
- name: Combined coverage (unit + integration, incl. firecracker)
|
||||
env:
|
||||
BOT_BOTTLE_BACKEND: firecracker
|
||||
BOT_BOTTLE_CI_INFRA_ARTIFACT_DIR: ${{ github.workspace }}/infra-candidate
|
||||
run: PYTHON=python3 bash scripts/coverage.sh critical
|
||||
|
||||
- name: Diff-coverage gate (changed lines >= 90%)
|
||||
run: |
|
||||
git fetch --no-tags origin main:refs/remotes/origin/main
|
||||
python3 scripts/diff_coverage.py --base origin/main --min 90
|
||||
|
||||
publish-infra:
|
||||
needs: [stage-firecracker-inputs, build-infra, unit, integration-docker, integration-firecracker, coverage]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
steps:
|
||||
- name: Checkout the tested revision
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Download the tested candidate
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate
|
||||
|
||||
# publish_infra re-derives the version from the checkout to confirm the
|
||||
# bundle matches before uploading, and the version hashes the dropbear
|
||||
# bytes. Stage the SAME dropbear build-infra used, or the recheck
|
||||
# computes a "<missing>"-dropbear version and rejects the candidate.
|
||||
- name: Download the staged dropbear (matches build-infra's version)
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: firecracker-inputs
|
||||
|
||||
- name: Publish the tested candidate
|
||||
env:
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_TOKEN: ${{ secrets.BOT_BOTTLE_INFRA_ARTIFACT_TOKEN }}
|
||||
BOT_BOTTLE_FC_DROPBEAR: ${{ github.workspace }}/firecracker-inputs/dropbear
|
||||
run: python3 -m bot_bottle.backend.firecracker.publish_infra --publish-dir infra-candidate
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
name: tracker-policy-issues
|
||||
|
||||
on:
|
||||
issues:
|
||||
types: [opened, unlabeled]
|
||||
|
||||
jobs:
|
||||
label-issue:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
issues: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Ensure the issue has a label
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: python3 scripts/tracker_policy.py label-issue
|
||||
@@ -0,0 +1,18 @@
|
||||
name: tracker-policy-pr
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, edited, reopened, synchronize, labeled, unlabeled]
|
||||
|
||||
jobs:
|
||||
check-pr:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
issues: read
|
||||
pull-requests: read
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Require an unlabeled PR linked to an issue
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: python3 scripts/tracker_policy.py check-pr
|
||||
+3
-2
@@ -98,6 +98,9 @@ RUN pip install --no-cache-dir /src/
|
||||
|
||||
# mitmdump -s requires a file path, not a module. Write a one-line shim that
|
||||
# re-exports `addons` from the installed package; mitmdump finds it there.
|
||||
# WORKDIR here also creates /app so the shim + COPYs below can write into it
|
||||
# (nothing created /app before this point).
|
||||
WORKDIR /app
|
||||
RUN printf 'from bot_bottle.egress_addon import addons\n' > /app/egress_addon.py
|
||||
COPY bot_bottle/egress_entrypoint.sh /app/egress-entrypoint.sh
|
||||
RUN chmod +x /app/egress-entrypoint.sh
|
||||
@@ -117,8 +120,6 @@ RUN mkdir -p \
|
||||
# subset the bottle uses.
|
||||
EXPOSE 8888 9099 9418 9420 9100
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# PID 1 is the supervisor. It owns signal handling and exit-code
|
||||
# propagation; no `exec` chain in the entrypoint itself.
|
||||
ENTRYPOINT ["python3", "-m", "bot_bottle.gateway_init"]
|
||||
|
||||
@@ -90,7 +90,7 @@ BOT_BOTTLE_BACKEND=firecracker ./cli.py start <agent>
|
||||
|
||||
> **NixOS:** enable `virtualisation.docker`, ensure the KVM module is loaded (`boot.kernelModules = [ "kvm-intel" ];` or `kvm-amd`), and add your user to the `kvm` and `docker` groups. For the network pool, consume the flake module — `imports = [ inputs.bot-bottle.nixosModules.firecracker-netpool ]; services.bot-bottle-firecracker = { enable = true; owner = "you"; };` — then `nixos-rebuild switch` (imperative nft/TAP rules don't survive a rebuild; channel users can `imports = [ <bot-bottle>/nix/firecracker-netpool.nix ]`). `firecracker` isn't in nixpkgs by default as a user binary — install the release binary (pin the version) and put it on `PATH`.
|
||||
|
||||
> **CI:** the coverage gate (`.gitea/workflows/test.yml` → `coverage` job) runs on a self-hosted runner labelled `kvm`, because the Firecracker backend's VM/SSH orchestration is exercised only by the integration suite, which needs `/dev/kvm` + the provisioned pool (a container runner would skip it and read as uncovered). Provision that runner exactly like a normal Firecracker host — `firecracker` on `PATH`, `/dev/kvm`, Docker, the cached guest kernel + static dropbear, and the pool installed as the persistent systemd unit — then register it with the `kvm` label. The unit/lint jobs still run on `ubuntu-latest`.
|
||||
> **CI:** the coverage gate (`.gitea/workflows/test.yml` → `coverage` job) runs on a self-hosted runner labelled `kvm`, because the Firecracker backend's VM/SSH orchestration is exercised only by the integration suite, which needs `/dev/kvm` + the provisioned pool (a container runner would skip it and read as uncovered). Provision that runner exactly like a normal Firecracker host — `firecracker` on `PATH`, `/dev/kvm`, the cached guest kernel + static dropbear, and the pool installed as the persistent systemd unit — then register it with the `kvm` label. A Docker-capable hosted job builds the candidate once; KVM tests boot those exact bytes, and a successful main run publishes them. The unit/lint jobs still run on `ubuntu-latest`.
|
||||
|
||||
```sh
|
||||
./cli.py start <agent> # builds the image on first run, drops you into claude
|
||||
@@ -173,6 +173,15 @@ When an outbound DLP detector matches a token, the route's `dlp.outbound_on_matc
|
||||
|
||||
More examples in `examples/`. Full design lives under `docs/prds/`; the trust-boundary rationale is in `docs/prds/0011-per-file-md-manifest.md`.
|
||||
|
||||
## Tracker policy
|
||||
|
||||
Issues are the canonical work items and own all tracker labels; every issue
|
||||
must have at least one. Pull requests stay unlabeled and deliberately reference
|
||||
an issue with `Closes #…`, `Part of #…`, or another form defined in
|
||||
[`ADR 0005`](docs/decisions/0005-issues-own-tracker-metadata.md). Gitea Actions
|
||||
enforces the convention for new work from 2026-07-18 onward. Earlier closed
|
||||
PRs are grandfathered rather than given artificial retrospective issues.
|
||||
|
||||
## Trademarks
|
||||
|
||||
bot-bottle is an independent project and is not affiliated with, endorsed by, or sponsored by Anthropic, PBC. "Claude" and "Claude Code" are trademarks of Anthropic, PBC; the project name uses "claude" descriptively to indicate that the tool runs Claude Code inside a sandbox.
|
||||
|
||||
@@ -42,7 +42,7 @@ class AuditStore(DbStore):
|
||||
super().__init__(db_path or host_db_path(), migrations)
|
||||
|
||||
def write_audit_entry(self, entry: AuditEntry) -> Path:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO supervise_audit_entries (
|
||||
@@ -66,7 +66,7 @@ class AuditStore(DbStore):
|
||||
def read_audit_entries(self, component: str, slug: str) -> list[AuditEntry]:
|
||||
if not self.db_path.is_file():
|
||||
return []
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT * FROM supervise_audit_entries
|
||||
|
||||
@@ -27,10 +27,14 @@ def _docker_on_path() -> bool:
|
||||
def _daemon_reachable() -> bool:
|
||||
if not _docker_on_path():
|
||||
return False
|
||||
return subprocess.run(
|
||||
["docker", "info"],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False,
|
||||
).returncode == 0
|
||||
try:
|
||||
return subprocess.run(
|
||||
["docker", "info"],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
check=False, timeout=5,
|
||||
).returncode == 0
|
||||
except subprocess.TimeoutExpired:
|
||||
return False
|
||||
|
||||
|
||||
def _print_install_pointer() -> None:
|
||||
|
||||
@@ -38,25 +38,39 @@ _BUILD_TIMEOUT_SECONDS = 900.0
|
||||
|
||||
|
||||
def _dockerfile_hash(dockerfile: Path) -> str:
|
||||
"""Cache key: the Dockerfile's content. The shipped agent Dockerfiles
|
||||
COPY nothing from the build context (see .dockerignore), so their content
|
||||
fully determines the image; a Dockerfile that adds COPY will want the
|
||||
"""The Dockerfile's content hash. The shipped agent Dockerfiles COPY
|
||||
nothing from the build context (see .dockerignore), so their content fully
|
||||
determines the built image; a Dockerfile that adds COPY will want the
|
||||
context folded in here too."""
|
||||
return hashlib.sha256(dockerfile.read_bytes()).hexdigest()[:16]
|
||||
|
||||
|
||||
def _rootfs_digest(dockerfile: Path) -> str:
|
||||
"""Cache key for the built AND boot-injected agent rootfs. Two inputs
|
||||
determine the on-disk rootfs: the Dockerfile (the image) and the guest init
|
||||
injected into it (`util._GUEST_INIT`). Folding the init in means a fix to
|
||||
it — e.g. making /tmp world-writable — busts the cache instead of silently
|
||||
reusing a stale rootfs built with the old init."""
|
||||
h = hashlib.sha256()
|
||||
h.update(_dockerfile_hash(dockerfile).encode())
|
||||
h.update(b"\0")
|
||||
h.update(util._GUEST_INIT.encode())
|
||||
return h.hexdigest()[:16]
|
||||
|
||||
|
||||
def build_agent_rootfs_dir(
|
||||
dockerfile: Path, *, image_tag: str, smoke_test: tuple[str, ...] = (),
|
||||
) -> Path:
|
||||
"""Build `dockerfile` in the infra VM (buildah, no host docker), export its
|
||||
rootfs, inject the guest boot bits, and return the cached base dir — the
|
||||
same shape `util.build_rootfs_ext4` consumes. Cached by Dockerfile content,
|
||||
so a repeat launch skips the rebuild.
|
||||
same shape `util.build_rootfs_ext4` consumes. Cached by Dockerfile content
|
||||
+ injected guest init, so a repeat launch skips the rebuild but an init or
|
||||
Dockerfile change rebuilds.
|
||||
|
||||
`smoke_test` (the provider's declared argv, e.g. `("claude","--version")`)
|
||||
is run in the freshly built image before export, catching an npm
|
||||
silent-failure image at build time rather than at first agent use."""
|
||||
digest = _dockerfile_hash(dockerfile)
|
||||
digest = _rootfs_digest(dockerfile)
|
||||
base = util.cache_dir() / "rootfs" / f"agent-{digest}"
|
||||
if (base / ".bb-ready").is_file():
|
||||
info(f"using cached agent rootfs {base.name}")
|
||||
|
||||
@@ -85,6 +85,11 @@ def infra_artifact_version(init_script: str, *, repo_root: Path = _REPO_ROOT) ->
|
||||
h.update(name.encode())
|
||||
h.update(b"\0")
|
||||
h.update((repo_root / name).read_bytes())
|
||||
h.update(b"pyproject.toml\0")
|
||||
h.update((repo_root / "pyproject.toml").read_bytes())
|
||||
h.update(b"dropbear\0")
|
||||
dropbear = util.dropbear_path()
|
||||
h.update(dropbear.read_bytes() if dropbear.is_file() else b"<missing>")
|
||||
h.update(b"init\0")
|
||||
h.update(init_script.encode())
|
||||
return h.hexdigest()[:16]
|
||||
@@ -111,6 +116,7 @@ def artifact_url(version: str, filename: str) -> str:
|
||||
|
||||
_GZ_NAME = "rootfs.ext4.gz"
|
||||
_SHA_NAME = "rootfs.ext4.gz.sha256"
|
||||
_CANDIDATE_DIR_ENV = "BOT_BOTTLE_INFRA_ARTIFACT_DIR"
|
||||
|
||||
|
||||
def _cache_root(version: str) -> Path:
|
||||
@@ -160,6 +166,33 @@ def ensure_artifact_gz(version: str) -> Path:
|
||||
"""The verified, cached `rootfs.ext4.gz` for `version` — downloading it (and
|
||||
its `.sha256`) once, then reusing it. Fail-closed on a checksum mismatch:
|
||||
the partial is removed and we die rather than boot an unverified rootfs."""
|
||||
candidate_dir = os.environ.get(_CANDIDATE_DIR_ENV, "").strip()
|
||||
if candidate_dir:
|
||||
root = Path(candidate_dir)
|
||||
version_file = root / "version.txt"
|
||||
# Guard the read so a missing version.txt is a clean error, not a raw
|
||||
# FileNotFoundError.
|
||||
if not version_file.is_file():
|
||||
die(f"infra candidate bundle is incomplete: {root}")
|
||||
declared = version_file.read_text(encoding="utf-8").strip()
|
||||
if declared != version:
|
||||
die(
|
||||
f"infra candidate version mismatch: expected {version}, "
|
||||
f"bundle contains {declared or '<empty>'}"
|
||||
)
|
||||
gz = root / _GZ_NAME
|
||||
sha = root / _SHA_NAME
|
||||
if not gz.is_file() or not sha.is_file():
|
||||
die(f"infra candidate bundle is incomplete: {root}")
|
||||
expected = sha.read_text().split()[0].strip().lower()
|
||||
actual = _sha256_file(gz)
|
||||
if actual != expected:
|
||||
die(
|
||||
f"infra candidate checksum mismatch for {version}:\n"
|
||||
f" expected {expected}\n actual {actual}"
|
||||
)
|
||||
return gz
|
||||
|
||||
root = _cache_root(version)
|
||||
root.mkdir(parents=True, exist_ok=True)
|
||||
gz = root / _GZ_NAME
|
||||
|
||||
@@ -161,20 +161,23 @@ def ensure_running() -> InfraVm:
|
||||
slot = netpool.orch_slot()
|
||||
url = f"http://{slot.guest_ip}:{CONTROL_PLANE_PORT}"
|
||||
key = _infra_dir() / "id_ed25519"
|
||||
if key.exists() and _health_ok(url):
|
||||
want = _expected_version()
|
||||
if _adoptable(key, url, want):
|
||||
info(f"adopting running infra VM at {url}")
|
||||
return InfraVm(guest_ip=slot.guest_ip, private_key=key)
|
||||
|
||||
with _singleton_lock():
|
||||
# Re-check under the lock: another launcher may have booted it while
|
||||
# we waited for the lock (double-checked, so we adopt not re-boot).
|
||||
if key.exists() and _health_ok(url):
|
||||
if _adoptable(key, url, want):
|
||||
info(f"adopting running infra VM at {url}")
|
||||
return InfraVm(guest_ip=slot.guest_ip, private_key=key)
|
||||
stop() # clear a stale/hung VM holding the link before booting fresh
|
||||
# Clear a stale/hung/OUTDATED VM holding the link before booting fresh.
|
||||
stop()
|
||||
ensure_built()
|
||||
infra = boot()
|
||||
wait_for_health(infra)
|
||||
_record_booted_version(want)
|
||||
return infra
|
||||
|
||||
|
||||
@@ -193,9 +196,15 @@ def _singleton_lock() -> Generator[None, None, None]:
|
||||
|
||||
|
||||
def stop() -> None:
|
||||
"""Stop the infra VM singleton (idempotent — absent is success)."""
|
||||
"""Stop the infra VM singleton (idempotent — absent is success). Reaps the
|
||||
recorded VMM AND any orphaned firecracker still bound to the infra config —
|
||||
the PID file drifts after crashes / out-of-band kills, and a survivor would
|
||||
hold the orchestrator TAP so the next boot dies with "tap … Resource busy".
|
||||
Drops the version marker so a stopped VM is never treated as adoptable."""
|
||||
_kill_pidfile()
|
||||
_kill_infra_firecrackers()
|
||||
_pid_file().unlink(missing_ok=True)
|
||||
_version_file().unlink(missing_ok=True)
|
||||
|
||||
|
||||
def boot() -> InfraVm:
|
||||
@@ -238,6 +247,36 @@ def _pid_file() -> Path:
|
||||
return _infra_dir() / "vm.pid"
|
||||
|
||||
|
||||
def _version_file() -> Path:
|
||||
"""Records the infra-artifact version the *running* VM booted from, so a
|
||||
later launcher can tell whether the singleton it found is the current code.
|
||||
Without it, a healthy VM built from an older image gets adopted forever and
|
||||
the new code never boots — every infra change would need an out-of-band
|
||||
kill to dislodge the stale VM (and races whatever launched next)."""
|
||||
return _infra_dir() / "booted-version"
|
||||
|
||||
|
||||
def _expected_version() -> str:
|
||||
return infra_artifact.infra_artifact_version(_infra_init())
|
||||
|
||||
|
||||
def _adoptable(key: Path, url: str, want: str) -> bool:
|
||||
"""Adopt a running infra VM only if it booted from the CURRENT version and
|
||||
its control plane is healthy. A missing/mismatched marker means a prior
|
||||
launcher booted an older infra image — reboot rather than reuse stale code."""
|
||||
if not key.exists():
|
||||
return False
|
||||
try:
|
||||
booted = _version_file().read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
return False
|
||||
return booted == want and _health_ok(url)
|
||||
|
||||
|
||||
def _record_booted_version(version: str) -> None:
|
||||
_version_file().write_text(version + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
# The registry "volume": a host-side ext4 file attached to the infra VM as a
|
||||
# second virtio-block device (guest /dev/vdb), mounted at the control plane's
|
||||
# DB dir. It outlives the ephemeral rootfs, so the bottle registry survives an
|
||||
@@ -309,6 +348,28 @@ def _kill_pidfile() -> None:
|
||||
pass
|
||||
|
||||
|
||||
def _kill_infra_firecrackers(proc_root: Path = Path("/proc")) -> None:
|
||||
"""SIGKILL any firecracker VMM whose `--config-file` is this host's infra
|
||||
config, independent of the PID file — reaps orphans it lost track of so the
|
||||
orchestrator TAP is free to rebind. Scoped to the infra config path, so the
|
||||
interactive pool's agent/infra VMs (other config paths) are untouched."""
|
||||
cfg = str(_infra_dir() / "config.json")
|
||||
for entry in proc_root.iterdir():
|
||||
if not entry.name.isdigit():
|
||||
continue
|
||||
try:
|
||||
if (entry / "comm").read_text().strip() != "firecracker":
|
||||
continue
|
||||
args = (entry / "cmdline").read_bytes().split(b"\0")
|
||||
except OSError:
|
||||
continue # process vanished / not ours
|
||||
if any(a.decode("utf-8", "replace") == cfg for a in args):
|
||||
try:
|
||||
os.kill(int(entry.name), signal.SIGKILL)
|
||||
except (OSError, ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def _health_ok(url: str) -> bool:
|
||||
try:
|
||||
with urllib.request.urlopen(f"{url}/health", timeout=1.0) as resp:
|
||||
@@ -433,7 +494,7 @@ BOT_BOTTLE_ROOT=/var/lib/bot-bottle python3 -m bot_bottle.orchestrator \\
|
||||
BOT_BOTTLE_GATEWAY_DAEMONS=egress,git-http,supervise \\
|
||||
BOT_BOTTLE_ORCHESTRATOR_URL=http://127.0.0.1:{CONTROL_PLANE_PORT} \\
|
||||
SUPERVISE_DB_PATH=/var/lib/bot-bottle/db/bot-bottle.db \\
|
||||
python3 /app/gateway_init.py &
|
||||
python3 -m bot_bottle.gateway_init &
|
||||
|
||||
# Reap as PID 1; children are backgrounded, so `wait` blocks.
|
||||
while : ; do wait ; done
|
||||
|
||||
@@ -11,7 +11,8 @@ The `<version>` is `infra_artifact.infra_artifact_version(...)`, the content
|
||||
hash of the rootfs inputs, so a launch host at the same code checkout resolves
|
||||
the exact artifact this produced.
|
||||
|
||||
python3 -m bot_bottle.backend.firecracker.publish_infra [--dry-run] [--force]
|
||||
python3 -m bot_bottle.backend.firecracker.publish_infra --output DIR
|
||||
python3 -m bot_bottle.backend.firecracker.publish_infra --publish-dir DIR
|
||||
|
||||
Auth: a token with `write:package` on the target owner, from
|
||||
`BOT_BOTTLE_INFRA_ARTIFACT_TOKEN`.
|
||||
@@ -24,7 +25,6 @@ import gzip
|
||||
import hashlib
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
@@ -131,36 +131,79 @@ def build_artifact(out_dir: Path) -> tuple[str, Path, Path]:
|
||||
return version, gz, sha
|
||||
|
||||
|
||||
def _publish_bundle(root: Path, token: str) -> str:
|
||||
version_file = root / "version.txt"
|
||||
# Guard the read so a missing version.txt is a clean error, not a raw
|
||||
# FileNotFoundError.
|
||||
if not version_file.is_file():
|
||||
raise SystemExit(f"incomplete artifact bundle: {root}")
|
||||
version = version_file.read_text(encoding="utf-8").strip()
|
||||
expected = infra_artifact.infra_artifact_version(infra_vm._infra_init())
|
||||
if version != expected:
|
||||
raise SystemExit(
|
||||
f"artifact bundle version {version!r} does not match checkout {expected!r}"
|
||||
)
|
||||
gz = root / "rootfs.ext4.gz"
|
||||
sha = root / "rootfs.ext4.gz.sha256"
|
||||
if not gz.is_file() or not sha.is_file():
|
||||
raise SystemExit(f"incomplete artifact bundle: {root}")
|
||||
expected_sha = sha.read_text().split()[0].strip().lower()
|
||||
if _sha256(gz) != expected_sha:
|
||||
raise SystemExit("artifact bundle checksum mismatch")
|
||||
|
||||
gz_url = infra_artifact.artifact_url(version, gz.name)
|
||||
sha_url = infra_artifact.artifact_url(version, sha.name)
|
||||
about_url = infra_artifact.artifact_url(version, _ABOUT_NAME)
|
||||
|
||||
# Publishing is idempotent. If this exact complete artifact is already
|
||||
# present, a test-only main commit is a no-op. Otherwise clear any partial
|
||||
# upload left by an interrupted prior attempt and upload the complete set.
|
||||
try:
|
||||
with urllib.request.urlopen(infra_artifact._open(sha_url)) as resp:
|
||||
remote_sha = resp.read().decode("utf-8").split()[0].strip().lower()
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code != 404:
|
||||
raise SystemExit(f"checking existing artifact failed (HTTP {e.code})")
|
||||
remote_sha = ""
|
||||
except urllib.error.URLError as e:
|
||||
raise SystemExit(f"registry unreachable: {sha_url} ({e.reason})")
|
||||
if remote_sha == expected_sha:
|
||||
print(f"infra rootfs {version} already published")
|
||||
return version
|
||||
|
||||
for url in (gz_url, sha_url, about_url):
|
||||
_delete(url, token)
|
||||
_put(gz_url, gz, token)
|
||||
_put(sha_url, sha.read_bytes(), token)
|
||||
_put(about_url, _ABOUT_TEXT.encode(), token)
|
||||
return version
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="publish_infra", description="Build + publish the infra rootfs artifact.")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="build the artifact but do not upload")
|
||||
parser.add_argument("--force", action="store_true",
|
||||
help="overwrite an already-published artifact of this version")
|
||||
mode = parser.add_mutually_exclusive_group(required=True)
|
||||
mode.add_argument("--output", type=Path,
|
||||
help="build a candidate bundle in DIR without publishing")
|
||||
mode.add_argument("--publish-dir", type=Path,
|
||||
help="publish an already-built and tested candidate bundle")
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
_, _, token = infra_artifact._config()
|
||||
if not args.dry_run and not token:
|
||||
if args.publish_dir is not None and not token:
|
||||
raise SystemExit(
|
||||
"no publish token: set BOT_BOTTLE_INFRA_ARTIFACT_TOKEN to a token "
|
||||
"with write:package")
|
||||
|
||||
with tempfile.TemporaryDirectory(prefix="bb-publish-infra.") as tmp:
|
||||
version, gz, sha = build_artifact(Path(tmp))
|
||||
gz_url = infra_artifact.artifact_url(version, gz.name)
|
||||
sha_url = infra_artifact.artifact_url(version, sha.name)
|
||||
about_url = infra_artifact.artifact_url(version, _ABOUT_NAME)
|
||||
if args.dry_run:
|
||||
print(f"dry-run: would upload -> {gz_url}")
|
||||
return 0
|
||||
if args.force:
|
||||
_delete(gz_url, token)
|
||||
_delete(sha_url, token)
|
||||
_delete(about_url, token)
|
||||
_put(gz_url, gz, token) # streamed from disk (hundreds of MB)
|
||||
_put(sha_url, sha.read_bytes(), token) # tiny, in-memory is fine
|
||||
_put(about_url, _ABOUT_TEXT.encode(), token) # package description
|
||||
if args.output is not None:
|
||||
args.output.mkdir(parents=True, exist_ok=True)
|
||||
version, _gz, _sha = build_artifact(args.output)
|
||||
(args.output / "version.txt").write_text(version + "\n", encoding="utf-8")
|
||||
print(f"built infra rootfs candidate {version}")
|
||||
return 0
|
||||
|
||||
assert args.publish_dir is not None
|
||||
version = _publish_bundle(args.publish_dir, token)
|
||||
print(f"published infra rootfs {version}")
|
||||
return 0
|
||||
|
||||
|
||||
@@ -368,6 +368,11 @@ mount -t devtmpfs dev /dev 2>/dev/null
|
||||
mkdir -p /dev/pts && mount -t devpts devpts /dev/pts 2>/dev/null
|
||||
mount -o remount,rw / 2>/dev/null
|
||||
|
||||
# /tmp must be world-writable + sticky. The rootless rootfs build can land
|
||||
# it 0755/root-owned, leaving the agent (uid 1000 node) unable to create
|
||||
# scratch dirs there — git worktrees, build temp, `git init /tmp/...`, etc.
|
||||
mkdir -p /tmp && chmod 1777 /tmp
|
||||
|
||||
# Install the per-bottle SSH pubkey from the kernel cmdline.
|
||||
KEY=$(sed -n 's/.*bb_pubkey=\([^ ]*\).*/\1/p' /proc/cmdline | base64 -d 2>/dev/null)
|
||||
if [ -n "$KEY" ]; then
|
||||
|
||||
@@ -101,7 +101,7 @@ def _init_script(port: int) -> str:
|
||||
# Gateway data plane, multi-tenant against the local control plane.
|
||||
f"( cd /app && BOT_BOTTLE_GATEWAY_DAEMONS={_GATEWAY_DAEMONS} "
|
||||
f"BOT_BOTTLE_ORCHESTRATOR_URL=http://127.0.0.1:{port} "
|
||||
f"SUPERVISE_DB_PATH={_DB_PATH_IN_CONTAINER} python3 /app/gateway_init.py ) &\n"
|
||||
f"SUPERVISE_DB_PATH={_DB_PATH_IN_CONTAINER} python3 -m bot_bottle.gateway_init ) &\n"
|
||||
"while : ; do wait ; done\n"
|
||||
)
|
||||
|
||||
|
||||
@@ -38,6 +38,13 @@ COMMANDS = {
|
||||
"supervise": cmd_supervise,
|
||||
}
|
||||
|
||||
# Commands that manage host prerequisites (or are otherwise store-free) and
|
||||
# must run before — or without — a migrated DB. `backend` provisions/probes
|
||||
# the host (TAP pool, /dev/kvm, firecracker) and never opens the store, so
|
||||
# gating it on the schema breaks preflight on a fresh CI runner where stdin
|
||||
# isn't a TTY and the migration prompt can't be answered.
|
||||
NO_MIGRATION_COMMANDS = frozenset({"backend"})
|
||||
|
||||
|
||||
def usage() -> None:
|
||||
sys.stderr.write(f"usage: {PROG} <command> [args...]\n\n")
|
||||
@@ -80,7 +87,7 @@ def main(argv: list[str] | None = None) -> int:
|
||||
usage()
|
||||
die(f"unknown command: {command}")
|
||||
mgr = StoreManager.instance()
|
||||
if not mgr.is_migrated():
|
||||
if command not in NO_MIGRATION_COMMANDS and not mgr.is_migrated():
|
||||
sys.stderr.write("bot-bottle: database schema is out of date\n")
|
||||
sys.stderr.write("Migrate now? [y/N] ")
|
||||
sys.stderr.flush()
|
||||
|
||||
+12
-2
@@ -3,6 +3,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlite3
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
@@ -28,12 +29,21 @@ class DbStore:
|
||||
conn.row_factory = sqlite3.Row
|
||||
return conn
|
||||
|
||||
@contextmanager
|
||||
def _connection(self):
|
||||
conn = self._connect()
|
||||
try:
|
||||
with conn:
|
||||
yield conn
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
def is_migrated(self) -> bool:
|
||||
"""Return True if the DB is fully up-to-date, False if migration is needed."""
|
||||
if not self.db_path.exists():
|
||||
return False
|
||||
try:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT version FROM schema_versions WHERE module = ?",
|
||||
(self._migrations.schema_key,),
|
||||
@@ -45,7 +55,7 @@ class DbStore:
|
||||
|
||||
def migrate(self) -> None:
|
||||
"""Apply any pending migrations and set permissions on the DB file."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
self._migrations.apply(conn)
|
||||
self._chmod()
|
||||
|
||||
|
||||
@@ -419,18 +419,24 @@ PY
|
||||
while IFS=' ' read -r old new ref; do
|
||||
[ -z "$ref" ] && continue
|
||||
[ "$new" = "$zero" ] && continue
|
||||
if [ "$old" = "$zero" ]; then
|
||||
# New ref: scan only the commits this push introduces — those
|
||||
# reachable from $new but not from any ref the gate already has.
|
||||
# Everything already on the gate arrived via upstream mirror-fetch
|
||||
# or a previously gitleaks-scanned push, so it's already-upstream
|
||||
# or already-scanned; re-scanning it (the old `$new` full-ancestry
|
||||
# range) only resurfaces historical findings and blocks every new
|
||||
# branch. See PRD 0028 / issue #106.
|
||||
log_opts="$new --not --all"
|
||||
else
|
||||
log_opts="$old..$new"
|
||||
fi
|
||||
# Scan only the commits this push introduces — those reachable from
|
||||
# $new but not from any ref the gate already has. Everything already
|
||||
# on the gate arrived via upstream mirror-fetch or a previously
|
||||
# gitleaks-scanned push, so it's already-upstream or already-scanned;
|
||||
# re-scanning it only resurfaces historical fixture findings.
|
||||
#
|
||||
# Applies to both new refs and updates. The old existing-branch range
|
||||
# `$old..$new` walks commits reachable from the new tip but not the
|
||||
# *old branch tip*: on a rebase/force-push onto a freshly-advanced
|
||||
# main that pulls in all of main's new history (incl. the deliberate
|
||||
# sandbox-escape gitleaks fixtures), blocking the push. `--not --all`
|
||||
# excludes anything already on the gate regardless of ancestry, so it
|
||||
# is also correct for non-fast-forward pushes (a rebase can skip
|
||||
# commits off the direct path). Security-equivalent per PRD 0028's
|
||||
# analysis: the bare repo's refs come only from trusted upstream
|
||||
# mirror-fetch or gitleaks-gated pushes.
|
||||
# See PRD 0028 (open question) / issues #106, #346.
|
||||
log_opts="$new --not --all"
|
||||
echo "git-gate: gitleaks scanning $ref ($log_opts)" >&2
|
||||
if ! gitleaks git --log-opts="$log_opts" --no-banner --redact 1>&2; then
|
||||
echo "git-gate: gitleaks rejected push to $ref" >&2
|
||||
|
||||
@@ -167,7 +167,7 @@ class RegistryStore(DbStore):
|
||||
metadata=metadata,
|
||||
policy=policy,
|
||||
)
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"DELETE FROM orchestrator_bottles "
|
||||
"WHERE source_ip = ? AND state = 'active' AND bottle_id != ?",
|
||||
@@ -193,7 +193,7 @@ class RegistryStore(DbStore):
|
||||
def set_policy(self, bottle_id: str, policy: str) -> bool:
|
||||
"""Update a bottle's policy in place (live reload). Returns True if
|
||||
the bottle exists."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
cur = conn.execute(
|
||||
"UPDATE orchestrator_bottles SET policy = ? WHERE bottle_id = ?",
|
||||
(policy, bottle_id),
|
||||
@@ -203,7 +203,7 @@ class RegistryStore(DbStore):
|
||||
|
||||
def deregister(self, bottle_id: str) -> bool:
|
||||
"""Remove a bottle. Returns True if a row was deleted."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
cur = conn.execute(
|
||||
"DELETE FROM orchestrator_bottles WHERE bottle_id = ?", (bottle_id,)
|
||||
)
|
||||
@@ -211,7 +211,7 @@ class RegistryStore(DbStore):
|
||||
|
||||
def get(self, bottle_id: str) -> BottleRecord | None:
|
||||
"""Return the bottle by id, or None if absent."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT * FROM orchestrator_bottles WHERE bottle_id = ?", (bottle_id,)
|
||||
).fetchone()
|
||||
@@ -219,7 +219,7 @@ class RegistryStore(DbStore):
|
||||
|
||||
def all(self) -> list[BottleRecord]:
|
||||
"""Every registered bottle, oldest first."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT * FROM orchestrator_bottles ORDER BY created_at"
|
||||
).fetchall()
|
||||
@@ -232,7 +232,7 @@ class RegistryStore(DbStore):
|
||||
source IP is unspoofable (Firecracker `/31` + nft) and the control
|
||||
plane is reachable only by the trusted gateway; pair with the
|
||||
identity token (`attribute`) elsewhere."""
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT * FROM orchestrator_bottles "
|
||||
"WHERE source_ip = ? AND state = 'active'",
|
||||
|
||||
@@ -66,7 +66,7 @@ class QueueStore(DbStore):
|
||||
super().__init__(resolved, migrations)
|
||||
|
||||
def write_proposal(self, proposal: Proposal) -> Path:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT OR REPLACE INTO supervise_proposals (
|
||||
@@ -89,7 +89,7 @@ class QueueStore(DbStore):
|
||||
return self.db_path
|
||||
|
||||
def read_proposal(self, proposal_id: str) -> Proposal:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
row = conn.execute(
|
||||
"""
|
||||
SELECT * FROM supervise_proposals
|
||||
@@ -104,7 +104,7 @@ class QueueStore(DbStore):
|
||||
def list_pending_proposals(self) -> list[Proposal]:
|
||||
if not self.db_path.is_file():
|
||||
return []
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT p.* FROM supervise_proposals p
|
||||
@@ -125,7 +125,7 @@ class QueueStore(DbStore):
|
||||
def list_all_pending_proposals(self) -> list[Proposal]:
|
||||
if not self.db_path.is_file():
|
||||
return []
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT p.* FROM supervise_proposals p
|
||||
@@ -142,7 +142,7 @@ class QueueStore(DbStore):
|
||||
return [self._row_to_proposal(row) for row in rows]
|
||||
|
||||
def write_response(self, response: Response) -> Path:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT OR REPLACE INTO supervise_responses (
|
||||
@@ -161,7 +161,7 @@ class QueueStore(DbStore):
|
||||
return self.db_path
|
||||
|
||||
def read_response(self, proposal_id: str) -> Response:
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
row = conn.execute(
|
||||
"""
|
||||
SELECT * FROM supervise_responses
|
||||
@@ -176,7 +176,7 @@ class QueueStore(DbStore):
|
||||
def archive_proposal(self, proposal_id: str) -> None:
|
||||
if not self.db_path.is_file():
|
||||
return
|
||||
with self._connect() as conn:
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE supervise_proposals SET archived = 1
|
||||
|
||||
@@ -47,6 +47,7 @@ from .supervise_types import (
|
||||
STATUS_MODIFIED,
|
||||
STATUS_REJECTED,
|
||||
TOOLS,
|
||||
TOOL_CHECK_PROPOSAL,
|
||||
TOOL_EGRESS_ALLOW,
|
||||
TOOL_EGRESS_BLOCK,
|
||||
TOOL_EGRESS_TOKEN_ALLOW,
|
||||
@@ -263,6 +264,7 @@ __all__ = [
|
||||
"TOOLS",
|
||||
"EGRESS_FORWARD_PROXY",
|
||||
"EGRESS_INTROSPECT_URL",
|
||||
"TOOL_CHECK_PROPOSAL",
|
||||
"TOOL_EGRESS_ALLOW",
|
||||
"TOOL_EGRESS_BLOCK",
|
||||
"TOOL_GITLEAKS_ALLOW",
|
||||
|
||||
@@ -2,14 +2,24 @@
|
||||
|
||||
Per-bottle MCP server exposing tools the agent calls to propose egress
|
||||
config changes when stuck. The tools are `egress-allow`,
|
||||
`egress-block`, and `list-egress-routes`.
|
||||
`egress-block`, `list-egress-routes`, and `check-proposal`.
|
||||
|
||||
Each queued tool call:
|
||||
Each queued proposal tool call:
|
||||
|
||||
1. Validates the proposed file syntactically.
|
||||
2. Writes a Proposal to the host SQLite database.
|
||||
3. Blocks polling for a matching Response row.
|
||||
4. Returns the operator's `{status, notes}` to the agent.
|
||||
3. Blocks polling for a matching Response row, up to a short grace
|
||||
window (`SUPERVISE_RESPONSE_TIMEOUT_SECONDS`, default 30s).
|
||||
4. On a decision within the window, returns the operator's
|
||||
`{status, notes}`. On timeout, returns `status: pending` **with the
|
||||
proposal id** and leaves the proposal queued — the flow is
|
||||
non-blocking past the grace window (PRD prd-new / issue #412).
|
||||
|
||||
`check-proposal` is the non-blocking companion: given a `proposal_id`
|
||||
returned by a `pending` response, it reports the current decision
|
||||
(`pending` | `approved` | `modified` | `rejected`) without re-proposing,
|
||||
so an approval made out-of-band (e.g. a web review console) can be resumed
|
||||
without holding an HTTP request open.
|
||||
|
||||
One shared server fronts every bottle (PRD 0070) and attributes each
|
||||
proposal to the calling bottle by source IP, resolved from the orchestrator
|
||||
@@ -22,7 +32,9 @@ Speaks MCP over HTTP+JSON-RPC. Methods handled:
|
||||
* `initialize` — handshake; returns server info + caps.
|
||||
* `notifications/initialized` — ack-only.
|
||||
* `tools/list` — returns the tool definitions.
|
||||
* `tools/call` — validates, queues, blocks, returns.
|
||||
* `tools/call` — validates, queues, waits out the grace
|
||||
window, returns (pending past it); or, for
|
||||
`check-proposal`, a non-blocking status poll.
|
||||
|
||||
Everything else returns JSON-RPC error -32601 (method not found).
|
||||
|
||||
@@ -232,6 +244,31 @@ TOOL_DEFINITIONS: list[dict[str, object]] = [
|
||||
),
|
||||
"inputSchema": _proposal_input_schema(),
|
||||
},
|
||||
{
|
||||
"name": _sv.TOOL_CHECK_PROPOSAL,
|
||||
"description": (
|
||||
"Poll a previously queued proposal for the operator's decision "
|
||||
"WITHOUT blocking or re-proposing. Pass the `proposal_id` you "
|
||||
"got back when an `egress-allow`/`egress-block` call returned "
|
||||
"`status: pending`. Returns the current status: `pending` (no "
|
||||
"decision yet — poll again later), `approved`, `modified`, "
|
||||
"`rejected`, or `unknown` (no such queued proposal — wrong id, "
|
||||
"or it was already resolved and read)."
|
||||
),
|
||||
"inputSchema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"proposal_id": {
|
||||
"type": "string",
|
||||
"description": (
|
||||
"The proposal id from a `pending` response."
|
||||
),
|
||||
},
|
||||
},
|
||||
"required": ["proposal_id"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@@ -353,7 +390,7 @@ def handle_tools_call(
|
||||
deadline=deadline,
|
||||
)
|
||||
except TimeoutError:
|
||||
text = format_pending_response_text(config.response_timeout_seconds)
|
||||
text = format_pending_response_text(proposal.id, config.response_timeout_seconds)
|
||||
return {
|
||||
"content": [{"type": "text", "text": text}],
|
||||
"isError": False,
|
||||
@@ -370,6 +407,54 @@ def handle_tools_call(
|
||||
}
|
||||
|
||||
|
||||
def handle_check_proposal(
|
||||
params: dict[str, object],
|
||||
config: ServerConfig,
|
||||
) -> dict[str, object]:
|
||||
"""Non-blocking poll of a queued proposal's decision, by id.
|
||||
|
||||
Never creates a Proposal (so `check-proposal` isn't in `TOOLS`); it only
|
||||
reads the queue. Resolution order mirrors the synchronous path's terminal
|
||||
step — a decided proposal is archived here exactly as `handle_tools_call`
|
||||
archives it after `wait_for_response`, so `pending` proposals stay visible
|
||||
to the operator until they're both decided *and* polled."""
|
||||
args_raw = params.get("arguments", {})
|
||||
if not isinstance(args_raw, dict):
|
||||
raise _RpcClientError(ERR_INVALID_PARAMS, "tools/call 'arguments' must be an object")
|
||||
proposal_id = args_raw.get("proposal_id")
|
||||
if not isinstance(proposal_id, str) or not proposal_id.strip():
|
||||
raise _RpcClientError(
|
||||
ERR_INVALID_PARAMS,
|
||||
"check-proposal: 'proposal_id' is required and must be a non-empty string",
|
||||
)
|
||||
proposal_id = proposal_id.strip()
|
||||
|
||||
try:
|
||||
response = _sv.read_response(config.bottle_slug, proposal_id)
|
||||
except FileNotFoundError:
|
||||
# No decision yet — distinguish "still queued" from "unknown id".
|
||||
try:
|
||||
_sv.read_proposal(config.bottle_slug, proposal_id)
|
||||
except FileNotFoundError:
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_unknown_proposal_text(proposal_id)}],
|
||||
"isError": True,
|
||||
}
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_still_pending_text(proposal_id)}],
|
||||
"isError": False,
|
||||
}
|
||||
|
||||
try:
|
||||
_sv.archive_proposal(config.bottle_slug, proposal_id)
|
||||
except OSError as e:
|
||||
raise _RpcInternalError(f"failed to archive proposal: {e}") from e
|
||||
return {
|
||||
"content": [{"type": "text", "text": format_response_text(response)}],
|
||||
"isError": response.status == _sv.STATUS_REJECTED,
|
||||
}
|
||||
|
||||
|
||||
def format_response_text(response: "_sv.Response") -> str:
|
||||
"""Pretty-print a Response for the tool's text content. The agent
|
||||
reads the text and decides whether to retry / give up / surface."""
|
||||
@@ -382,12 +467,35 @@ def format_response_text(response: "_sv.Response") -> str:
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def format_pending_response_text(timeout_seconds: float) -> str:
|
||||
def format_pending_response_text(proposal_id: str, timeout_seconds: float) -> str:
|
||||
"""Grace-window timeout: the proposal stays queued, and the agent is
|
||||
told the id so it can `check-proposal` instead of re-proposing."""
|
||||
return "\n".join([
|
||||
"status: pending",
|
||||
f"proposal_id: {proposal_id}",
|
||||
(
|
||||
"notes: operator response timed out after "
|
||||
f"{timeout_seconds:g}s; proposal remains queued"
|
||||
f"notes: no operator decision within {timeout_seconds:g}s; the "
|
||||
"proposal remains queued. Poll it (do not re-propose) by calling "
|
||||
f"`check-proposal` with proposal_id={proposal_id!r}."
|
||||
),
|
||||
])
|
||||
|
||||
|
||||
def format_still_pending_text(proposal_id: str) -> str:
|
||||
return "\n".join([
|
||||
"status: pending",
|
||||
f"proposal_id: {proposal_id}",
|
||||
"notes: still queued; no operator decision yet. Call `check-proposal` again later.",
|
||||
])
|
||||
|
||||
|
||||
def format_unknown_proposal_text(proposal_id: str) -> str:
|
||||
return "\n".join([
|
||||
"status: unknown",
|
||||
f"proposal_id: {proposal_id}",
|
||||
(
|
||||
"notes: no queued proposal with this id for this bottle — the id "
|
||||
"may be wrong, or the proposal was already resolved and read."
|
||||
),
|
||||
])
|
||||
|
||||
@@ -482,6 +590,11 @@ class MCPHandler(http.server.BaseHTTPRequestHandler):
|
||||
# — silently dropping base routes like api.anthropic.com on approval.
|
||||
if req.params.get("name") == _sv.TOOL_LIST_EGRESS_ROUTES:
|
||||
return self._resolved_routes_payload()
|
||||
# `check-proposal` is a non-blocking read of the calling bottle's
|
||||
# own queue — attributed by source IP like a proposal, but it
|
||||
# never queues or blocks.
|
||||
if req.params.get("name") == _sv.TOOL_CHECK_PROPOSAL:
|
||||
return handle_check_proposal(req.params, self._attributed_config(config))
|
||||
# Attribute the proposal to the source-IP-resolved bottle, so the one
|
||||
# shared server queues each bottle's proposal under its own slug.
|
||||
return handle_tools_call(req.params, self._attributed_config(config))
|
||||
|
||||
@@ -20,6 +20,10 @@ TOOL_EGRESS_ALLOW = "egress-allow"
|
||||
TOOL_GITLEAKS_ALLOW = "gitleaks-allow"
|
||||
TOOL_EGRESS_TOKEN_ALLOW = "egress-token-allow"
|
||||
TOOL_LIST_EGRESS_ROUTES = "list-egress-routes"
|
||||
# Read-only agent tool: poll a queued proposal for the operator's decision
|
||||
# without blocking or re-proposing. It never becomes a `Proposal.tool` (no
|
||||
# queue record is created for it), so it is intentionally NOT in `TOOLS`.
|
||||
TOOL_CHECK_PROPOSAL = "check-proposal"
|
||||
TOOLS: tuple[str, ...] = (
|
||||
TOOL_EGRESS_ALLOW,
|
||||
TOOL_EGRESS_BLOCK,
|
||||
@@ -156,6 +160,7 @@ __all__ = [
|
||||
"TOOLS",
|
||||
"TOOL_EGRESS_ALLOW",
|
||||
"TOOL_EGRESS_BLOCK",
|
||||
"TOOL_CHECK_PROPOSAL",
|
||||
"TOOL_EGRESS_TOKEN_ALLOW",
|
||||
"TOOL_GITLEAKS_ALLOW",
|
||||
"TOOL_LIST_EGRESS_ROUTES",
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
# ADR 0005: Keep tracker metadata on issues
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-07-18
|
||||
- **Deciders:** didericis
|
||||
|
||||
## Context
|
||||
|
||||
Gitea exposes labels on both issues and pull requests. Applying the same labels
|
||||
to both copies planning metadata, creates a synchronization obligation, and
|
||||
makes disagreements between the two records possible. At the same time,
|
||||
unlabelled objects look accidental unless the repository states which object
|
||||
owns the metadata.
|
||||
|
||||
The repository already uses issues as work items and PRs as implementations of
|
||||
those work items. At this decision's cutoff, all open PRs reference issues, but
|
||||
121 of 219 historically merged PRs do not. Manufacturing retrospective issues
|
||||
for that history would create records that never participated in planning and
|
||||
would make the issue history less truthful.
|
||||
|
||||
## Decision
|
||||
|
||||
Issues are the canonical tracker records and own labels. Every issue has at
|
||||
least one label. An issue opened or left without labels receives
|
||||
`Status/Needs Triage` automatically until it is classified.
|
||||
|
||||
Pull requests carry no labels. Every new PR deliberately references at least
|
||||
one existing issue in its title or description with one of these forms:
|
||||
|
||||
- `Closes #123`, `Fixes #123`, or `Resolves #123` when merging completes it.
|
||||
- `Part of #123`, `Related to #123`, `Refs #123`, or `References #123` when it
|
||||
contributes without completing it.
|
||||
|
||||
Gitea Actions enforces both PR rules as a status check and repairs the empty
|
||||
issue-label state. Branch protection makes the PR policy check required.
|
||||
|
||||
The policy applies from 2026-07-18 onward. Existing issues may be labelled as
|
||||
they are encountered, but closed PRs are grandfathered: no retrospective
|
||||
issues or PR labels are created solely to make history conform.
|
||||
|
||||
## Consequences
|
||||
|
||||
- Classification, priority, and workflow metadata have one source of truth.
|
||||
- A PR's issue link is the navigation path to its planning metadata.
|
||||
- Multi-PR issues do not require copied or synchronized labels.
|
||||
- `Status/Needs Triage` is an intentional fallback, not a final
|
||||
classification.
|
||||
- Direct issue creation remains convenient; automation repairs a missing label
|
||||
immediately after creation because Gitea has no native required-label rule.
|
||||
- The required check must be configured in branch protection after this
|
||||
workflow lands.
|
||||
|
||||
## Links
|
||||
|
||||
- Issue #405.
|
||||
- `.gitea/workflows/tracker-policy.yml`.
|
||||
- `scripts/tracker_policy.py`.
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0023: smolmachines bottle backend
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0032: Decompose smolmachines launch and harden bringup sequencing
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis-claude
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0038: smolmachines Env Contract and Secret-Safe Injection
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis-codex
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0039: smolmachines Capability-Block Remediation
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis-codex
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0042: smolmachines Cross-Backend Parity Tests
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis-codex
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0057: Promote smolmachines to default backend; convert Docker to example-only
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** didericis
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0068: smolmachines backend on Linux
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
- **Status:** Superseded (2026-07-11) — was Active
|
||||
- **Author:** Claude
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
# PRD prd-new: Non-blocking supervise (async approval + proposal polling)
|
||||
|
||||
- **Status:** Draft
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-07-18
|
||||
- **Issue:** #412
|
||||
|
||||
## Summary
|
||||
|
||||
The per-bottle supervise MCP server (`bot_bottle/supervise_server.py`)
|
||||
answers `tools/call` **synchronously**: it queues the agent's proposal and
|
||||
blocks the tool call polling for the operator's decision. On timeout it
|
||||
returns `status: pending` and leaves the proposal queued — but it hands the
|
||||
agent **no proposal id** and offers **no way to poll a specific pending
|
||||
proposal**, so the only way to learn the outcome is to re-propose (a
|
||||
duplicate).
|
||||
|
||||
This PRD makes the MCP flow non-blocking and pollable, so an approval can
|
||||
happen out-of-band (a human taking minutes-to-hours in a review console)
|
||||
without holding an HTTP request open or wedging the agent:
|
||||
|
||||
1. Include the `proposal_id` in the `pending` response.
|
||||
2. Add a `check-proposal` MCP tool: a non-blocking status lookup by
|
||||
proposal id.
|
||||
3. Keep the short synchronous grace window for the common "operator is
|
||||
right there" fast path.
|
||||
|
||||
## Problem
|
||||
|
||||
`handle_tools_call` → `_sv.wait_for_response(...)` blocks up to
|
||||
`SUPERVISE_RESPONSE_TIMEOUT_SECONDS` (default 30s). Two problems follow:
|
||||
|
||||
- **Human latency ≠ tool-call latency.** A real review — rendered diff,
|
||||
RBAC routing to an approver, someone tapping approve on their phone — is
|
||||
minutes-to-hours. Holding the MCP request open that long is fragile
|
||||
(proxy/keepalive timeouts, the mitmproxy egress hop, and the agent
|
||||
harness's own tool-call timeout, which a long block can trip and stall
|
||||
the whole turn).
|
||||
- **No resume path.** The pending fallback already exists, but without a
|
||||
proposal id and a poll tool the agent can't reconnect to that specific
|
||||
decision — it re-proposes, duplicating the queue entry.
|
||||
|
||||
This is also the precondition for the planned web-console human-review
|
||||
flow (RBAC, audit retention, mobile) — see issue #412.
|
||||
|
||||
**Safety note:** the MCP tools only *propose* policy changes; enforcement
|
||||
stays at the egress proxy and the git-gate. Returning early on `pending`
|
||||
therefore opens no hole — the agent still cannot egress or push anything
|
||||
unapproved.
|
||||
|
||||
## Goals / success criteria
|
||||
|
||||
- A `pending` MCP response carries the `proposal_id`.
|
||||
- An agent can call `check-proposal(proposal_id)` and get the current
|
||||
state (`pending` | `approved` | `modified` | `rejected`) **without
|
||||
blocking** and **without creating a new proposal**.
|
||||
- The synchronous fast path (operator approves within the grace window) is
|
||||
unchanged: the first `tools/call` still returns the decision directly.
|
||||
- No change to enforcement, attribution (source-IP → bottle), or the
|
||||
operator-side queue/response schema.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- The git-gate `pre-receive` path (it is synchronous by nature and cannot
|
||||
poll — its async variant is reject-fast + re-push; tracked as a
|
||||
follow-up).
|
||||
- Backpressure / in-flight-proposal caps.
|
||||
- MCP server→client notifications (event-driven resume).
|
||||
- Any web-console UI (this PRD is the protocol groundwork it needs).
|
||||
|
||||
## Design
|
||||
|
||||
### `pending` response carries the id
|
||||
|
||||
`handle_tools_call`'s timeout branch formats the pending text with the
|
||||
`proposal.id` and a pointer to `check-proposal`, so the agent knows what to
|
||||
poll.
|
||||
|
||||
### `check-proposal` tool
|
||||
|
||||
A new read-only MCP tool (`TOOL_CHECK_PROPOSAL = "check-proposal"`),
|
||||
attributed to the calling bottle by source IP exactly like the proposal
|
||||
tools. Input: `{ "proposal_id": string }`. Behavior:
|
||||
|
||||
1. `read_response(slug, id)` →
|
||||
- **found**: archive the proposal (same terminal step the synchronous
|
||||
path takes) and return the decision via `format_response_text`;
|
||||
`isError` iff rejected.
|
||||
2. **not found** → `read_proposal(slug, id)` →
|
||||
- **found**: still queued → return `status: pending`.
|
||||
- **not found**: unknown id, or already resolved-and-archived (e.g. a
|
||||
second poll) → return `status: unknown`, `isError: true`.
|
||||
|
||||
Both lookups already raise `FileNotFoundError` when absent
|
||||
(`queue_store.py`), so the handler needs no new store methods. `check-`
|
||||
`proposal` is the only path (besides the synchronous response) that
|
||||
archives, so a proposal that times out to `pending` stays visible to the
|
||||
operator until it is decided and then polled.
|
||||
|
||||
### Grace window
|
||||
|
||||
Left at the existing 30s default (`SUPERVISE_RESPONSE_TIMEOUT_SECONDS`),
|
||||
which doubles as the instant-approve fast path. Tuning it down is an
|
||||
operator setting, not a code change; noted for the console rollout.
|
||||
|
||||
## Implementation chunks
|
||||
|
||||
1. **(this PR)** `TOOL_CHECK_PROPOSAL` constant; `check-proposal` tool
|
||||
definition + `handle_check_proposal`; dispatch wiring; `proposal_id` in
|
||||
the pending text; unit tests. Files: `bot_bottle/supervise_types.py`,
|
||||
`bot_bottle/supervise.py` (re-export), `bot_bottle/supervise_server.py`,
|
||||
`tests/unit/test_supervise_server.py`.
|
||||
2. **(follow-up)** git-gate `pre-receive` reject-fast + re-push.
|
||||
3. **(follow-up)** per-bottle in-flight-proposal backpressure cap.
|
||||
4. **(follow-up)** MCP notifications for event-driven resume; web-console
|
||||
review flow (RBAC, audit retention) on top.
|
||||
|
||||
## Open questions
|
||||
|
||||
- Should a resolved-but-unpolled proposal auto-archive after some TTL, or
|
||||
only on poll? (Leaning: only on poll, so a decision is never lost to a
|
||||
reaper before the agent sees it.)
|
||||
- Does the agent harness need an explicit "you have a pending proposal"
|
||||
nudge, or is returning `pending` from the original call enough? (Deferred
|
||||
to the notifications chunk.)
|
||||
@@ -1,35 +1,60 @@
|
||||
# Landscape: AI-agent sandbox tools
|
||||
|
||||
A broader survey than [`landscape-containerized-claude.md`](landscape-containerized-claude.md),
|
||||
which focused on Claude-Code-specific containerizers. This one covers
|
||||
general AI-agent sandbox / containment projects — some Claude-specific,
|
||||
some agent-agnostic, some hosted SaaS — and contrasts them with
|
||||
bot-bottle's design.
|
||||
Survey of AI-agent sandbox and containment projects — including local
|
||||
coding-agent wrappers, agent-agnostic runtimes, hosted platforms, and
|
||||
governance layers — contrasted with bot-bottle's design. The original
|
||||
Claude-Code-specific containerizer survey was folded into this note on
|
||||
2026-07-20 so there is one landscape and one positioning verdict.
|
||||
|
||||
Research conducted 2026-05-11. CubeSandbox added 2026-07-18 (see its
|
||||
per-project note and the addendum at the end). Also updated 2026-07-18:
|
||||
bot-bottle no longer uses **pipelock** — outbound DLP is now bot-bottle's
|
||||
own (deliberately simple) egress scanner (a mitmproxy addon with custom
|
||||
detectors, PRD 0017 / 0053), and git-push secret scanning is handled by
|
||||
detectors, PRD 0017 / 0052), and git-push secret scanning is handled by
|
||||
**gitleaks** in the git-gate. "pipelock" below has been replaced with the
|
||||
current mechanism; it survives only in older PRDs as history.
|
||||
|
||||
Updated again 2026-07-18: six additional tools added (Cleanroom,
|
||||
container-use, Docker sbx, Anthropic srt, Microsoft AGT, Open Agent
|
||||
Passport); an **Agent-tailored policy** row added to the comparison table;
|
||||
a separate Governance layers section added for AGT and OAP. See the
|
||||
second addendum at the end.
|
||||
|
||||
Updated 2026-07-20: the borrowable-ideas status was reconciled with the
|
||||
current implementation. In-flight credential injection and the microVM
|
||||
backends have shipped, while per-use SSH confirmation was superseded by
|
||||
keeping git credentials out of the agent entirely.
|
||||
|
||||
Also updated 2026-07-20: **E2B and Daytona added as first-class entries.**
|
||||
Earlier revisions mentioned E2B only as the API and lifecycle model that
|
||||
CubeSandbox implements, and omitted Daytona entirely. That was a survey gap,
|
||||
not a principled scope exclusion: both are major hosted sandbox platforms and
|
||||
belong in this landscape even though they target platform builders rather than
|
||||
bot-bottle's local single-operator workflow.
|
||||
|
||||
## Summary
|
||||
|
||||
Nine projects surveyed. None duplicate bot-bottle's combination of
|
||||
local VM-per-bottle isolation (Firecracker microVM on KVM Linux, Apple
|
||||
Container on macOS — Docker is now only the legacy fallback), a
|
||||
declarative JSON manifest, per-agent egress allowlist + outbound-content
|
||||
DLP via bot-bottle's own egress scanner (plus gitleaks secret-scanning on
|
||||
git push), and bottle/agent split. Two clusters stand out:
|
||||
The main table compares bot-bottle against fifteen isolation/sandbox tools.
|
||||
Governance/pre-action authorization and credential-only layers are covered
|
||||
separately because they don't provide VM or container isolation. None
|
||||
duplicate bot-bottle's combination of local
|
||||
VM-per-bottle isolation, a declarative per-role manifest, per-agent
|
||||
egress allowlist + outbound-content DLP, bottle/agent split, and the
|
||||
composable `extends:` policy model. Three clusters stand out:
|
||||
|
||||
- **Closest neighbours** — agent-safehouse and litterbox: local,
|
||||
single-user, thin wrappers over an existing OS primitive
|
||||
(`sandbox-exec`, Podman + Landlock).
|
||||
- **Different category** — tilde.run (hosted SaaS), boxlite and
|
||||
microsandbox (microVM libraries for platform builders), CubeSandbox
|
||||
(self-hosted multi-tenant microVM service), endo-familiar
|
||||
- **Different category (isolation)** — tilde.run (hosted SaaS), boxlite
|
||||
and microsandbox (microVM libraries for platform builders), E2B and Daytona
|
||||
(hosted sandbox platforms), CubeSandbox (self-hosted multi-tenant microVM
|
||||
service), endo-familiar
|
||||
(capability-security paradigm, no OS isolation).
|
||||
- **New: governance/pre-action layers** — Microsoft AGT and Open Agent
|
||||
Passport (OAP): framework-embedded tool-call interceptors with
|
||||
per-agent declarative policy. Closest competitors on agent-tailored
|
||||
policy, but operate at the tool-call level rather than providing
|
||||
network/filesystem isolation; they complement rather than substitute.
|
||||
|
||||
The microVM cluster (matchlock, smolmachines, boxlite, microsandbox,
|
||||
CubeSandbox) is the most relevant for the v2 isolation discussion in
|
||||
@@ -40,15 +65,17 @@ ergonomic enough that microVMs are **now bot-bottle's default backend**
|
||||
only as a legacy fallback for CI / hosts without KVM or Apple Container.
|
||||
That discussion has since shipped, not just been theorized.
|
||||
|
||||
**The one that matters most for positioning is CubeSandbox** — it is the
|
||||
first surveyed project to ship bot-bottle's would-be wedge (default-deny
|
||||
egress allowlist + full audit logs + in-flight credential custody so keys
|
||||
never enter the sandbox) *combined with* per-sandbox microVM isolation,
|
||||
**The one that matters most for positioning is CubeSandbox** — it ships
|
||||
bot-bottle's bundle of default-deny egress allowlisting, full audit logs, and
|
||||
in-flight credential custody *combined with* per-sandbox microVM isolation,
|
||||
open-source under Apache 2.0, with Tencent Cloud behind it and 10.4k
|
||||
stars. It's a self-hosted multi-tenant service for platform builders, not
|
||||
a single-user declarative tool, so it doesn't collide head-on — but it
|
||||
narrows the "nobody else bundles egress custody + credential injection"
|
||||
claim that the monetization positioning leans on. See the addendum.
|
||||
claim that the monetization positioning leans on. Daytona now also offers
|
||||
domain/CIDR firewall policy plus in-flight header credential substitution and
|
||||
response scrubbing, although its higher tiers are not default-deny and its
|
||||
production platform is proprietary. See the addendum.
|
||||
|
||||
## Per-project notes
|
||||
|
||||
@@ -85,7 +112,8 @@ claim that the monetization positioning leans on. See the addendum.
|
||||
|
||||
### agent-safehouse
|
||||
- **Source**: https://agent-safehouse.dev/ ; https://github.com/eugene1g/agent-safehouse
|
||||
- **License**: Apache 2.0 (~1,400 stars)
|
||||
- **HN launch**: [#47301085](https://news.ycombinator.com/item?id=47301085) (March 12 2026) — 823 points
|
||||
- **License**: Apache 2.0 (~1,781 stars at launch)
|
||||
- **Isolation**: macOS `sandbox-exec` (Seatbelt) profiles — kernel-level
|
||||
syscall interception, no container.
|
||||
- **Locality**: Local, macOS only.
|
||||
@@ -95,6 +123,16 @@ claim that the monetization positioning leans on. See the addendum.
|
||||
- **Config**: Shell functions or custom `sandbox-exec` profile files;
|
||||
LLM-assisted profile generation supported.
|
||||
- **Network policy**: Not addressed.
|
||||
- **Notable from HN thread**: Creator acknowledged the project is "just a
|
||||
policy-generator for `sandbox-exec` — no dependencies, no daemons, no
|
||||
subscription; I did put in many hours to identify the minimum required
|
||||
permissions for agents to continue working." Simon Willison noted that
|
||||
evaluating whether a sandboxing tool actually works as intended is hard.
|
||||
Top community sentiment: *"I honestly think that sandboxing is currently
|
||||
THE major challenge that needs to be solved for the tech to fully realise
|
||||
its potential."* The macOS Docker gap (Docker for Mac runs inside a Linux
|
||||
VM, so `sandbox-exec` is the only native primitive for bare-metal macOS
|
||||
processes) was the stated motivation.
|
||||
- **Maturity**: Active through March 2026.
|
||||
|
||||
### matchlock
|
||||
@@ -177,6 +215,80 @@ claim that the monetization positioning leans on. See the addendum.
|
||||
also supported.
|
||||
- **Maturity**: Active through April 2026.
|
||||
|
||||
### E2B *(added 2026-07-20)*
|
||||
|
||||
- **Source**: https://github.com/e2b-dev/e2b ; https://e2b.dev/docs
|
||||
- **License**: Apache 2.0 (~12.4k stars); commercial hosted service with
|
||||
self-hosting/BYOC support.
|
||||
- **Isolation**: Firecracker microVM per sandbox.
|
||||
- **Locality**: Cloud-hosted by default; self-hosting uses Terraform on AWS or
|
||||
GCP (with other targets documented as works in progress).
|
||||
- **Agent integration**: LLM-agnostic Python and JavaScript/TypeScript SDKs;
|
||||
code-interpreter and desktop-sandbox products. Platform primitive rather
|
||||
than a coding-agent wrapper.
|
||||
- **Config**: Programmatic SDK/API plus templates. Network configuration
|
||||
supports internet on/off, outbound allow/deny rules, and a custom egress
|
||||
proxy.
|
||||
- **Network policy**: Configurable per sandbox, but not documented as
|
||||
default-deny and no built-in outbound-content DLP is documented.
|
||||
- **Credentials**: Environment variables passed to the sandbox are explicitly
|
||||
not private at the OS level. No built-in in-flight application-credential
|
||||
injection is documented.
|
||||
- **Persistence**: Full memory + filesystem pause/resume, snapshots, and
|
||||
auto-resume. Continuous runtime is tier-limited, while paused sandboxes are
|
||||
retained indefinitely.
|
||||
- **Maturity**: Established hosted platform and the API compatibility target
|
||||
used by CubeSandbox.
|
||||
|
||||
### Daytona *(added 2026-07-20)*
|
||||
|
||||
- **Source**: https://github.com/daytonaio/daytona ;
|
||||
https://www.daytona.io/docs/
|
||||
- **License**: Current production platform is proprietary. The former AGPL
|
||||
repository remains public but is no longer maintained after Daytona moved
|
||||
production development closed-source in June 2026.
|
||||
- **Isolation**: Hosted container sandboxes by default, with separate Linux
|
||||
and Windows VM sandbox classes for dedicated-OS workloads. Each sandbox has
|
||||
its own filesystem and network stack; VM-only features include memory
|
||||
pause/resume and forking.
|
||||
- **Locality**: Hosted multi-tenant service, with dedicated/custom regions and
|
||||
customer runners available.
|
||||
- **Agent integration**: LLM/framework-agnostic SDKs (Python, TypeScript, Go,
|
||||
Ruby, Java), API, and CLI; official agent-framework guides. Platform
|
||||
primitive rather than a local coding-agent wrapper.
|
||||
- **Config**: Programmatic per-sandbox image/snapshot, resources, lifecycle,
|
||||
firewall, and secrets.
|
||||
- **Network policy**: Per-sandbox IPv4/domain allowlists and block-all mode,
|
||||
subordinate to organization/tier policy. Full internet access is the
|
||||
default on higher tiers, so it is configurable rather than uniformly
|
||||
default-deny.
|
||||
- **Credentials**: First-class secret manager with the same phantom-token
|
||||
pattern as bot-bottle: the sandbox environment gets an opaque placeholder,
|
||||
an HTTPS proxy substitutes the real secret in headers only for allowed
|
||||
hosts, and responses are scrubbed back to the placeholder.
|
||||
- **Persistence**: Persistent filesystem for stopped container sandboxes;
|
||||
memory + filesystem pause/resume for VM sandboxes; snapshots and configurable
|
||||
auto-stop.
|
||||
- **Maturity**: Production commercial platform. Notable April 2026 credential
|
||||
exposure was patched; the June 2026 closed-source transition materially
|
||||
changes its transparency/self-hosting posture.
|
||||
|
||||
### Other hosted runtimes carried forward from the earlier survey
|
||||
|
||||
- **Northflank Sandboxes** — hosted or customer-cloud, microVM-backed
|
||||
containers with SDK-managed lifecycle, optional persistent volumes, and
|
||||
sub-second claimed boot. This is a platform primitive for untrusted code and
|
||||
agents, not a local agent wrapper or role-policy layer.
|
||||
- **Cloudflare Sandbox SDK** — Workers/Durable Objects API over VM-isolated
|
||||
Linux containers for command, file, process, and service execution. It is a
|
||||
hosted TypeScript platform primitive; application authentication,
|
||||
authorization, and credential-proxy patterns remain the integrator's job.
|
||||
|
||||
Both belong to the same “build your agent platform on this runtime” category as
|
||||
E2B and Daytona. They were named but not analyzed in depth by the original
|
||||
Claude-specific note, so they remain outside the main comparison table rather
|
||||
than being presented with false precision.
|
||||
|
||||
### CubeSandbox *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/TencentCloud/CubeSandbox ;
|
||||
HN launch https://news.ycombinator.com/item?id=47863430
|
||||
@@ -210,20 +322,243 @@ claim that the monetization positioning leans on. See the addendum.
|
||||
- **Maturity**: Open-sourced July 2026 off production Tencent Cloud use;
|
||||
most-starred project in this set (~10.4k).
|
||||
|
||||
### Cleanroom *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/buildkite/cleanroom
|
||||
- **License**: Apache 2.0
|
||||
- **Isolation**: MicroVM — Firecracker on Linux, Virtualization.framework
|
||||
on macOS. Digest-pinned OCI images.
|
||||
- **Locality**: Self-hosted server (CI-oriented).
|
||||
- **Agent integration**: Generic process sandbox; CI-first, not a
|
||||
Claude/agent wrapper.
|
||||
- **Config**: `cleanroom.yaml` in the repo being sandboxed defines egress
|
||||
rules, resources, and network policy. Cleanroom resolves this from the
|
||||
commit being run.
|
||||
- **Network policy**: Default-deny + per-repo hostname allowlist (resolved
|
||||
from DNS answers + destination IP:port). Co-hosted services on the same
|
||||
IP:port are not distinguished. OIDC-backed auth for remote servers.
|
||||
- **Credentials**: Host-side only; not injected in-flight but not present
|
||||
in the VM.
|
||||
- **Notable**: Policy lives in the *repo being sandboxed*, not in an
|
||||
agent-role definition — closer to per-repo scoping than per-role.
|
||||
Supports Docker-inside-sandbox (`services.docker.required: true`), OIDC
|
||||
authorization, suspend/resume lifecycle.
|
||||
- **Maturity**: Active Buildkite product.
|
||||
|
||||
### container-use *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/dagger/container-use
|
||||
- **License**: Apache 2.0
|
||||
- **Isolation**: Docker container per agent + git worktree per agent.
|
||||
Containers share the host kernel; stronger than bare host but weaker
|
||||
than microVM.
|
||||
- **Locality**: Local.
|
||||
- **Agent integration**: MCP stdio server — Claude Code, Cursor, Windsurf.
|
||||
`claude mcp add container-use -- container-use stdio`.
|
||||
- **Config**: None for security policy. Environments are provisioned on
|
||||
demand; no allowlist or credential config.
|
||||
- **Network policy**: Not addressed.
|
||||
- **Notable**: Per-agent git branches (`container-use/<env_name>`);
|
||||
parallel agents without filesystem conflict; real-time log visibility
|
||||
and terminal attach for intervention; git-based review workflow.
|
||||
Oriented toward parallel development safety, not security containment.
|
||||
- **Maturity**: Early development, active.
|
||||
|
||||
### Docker sbx *(added 2026-07-18)*
|
||||
- **Source**: Docker proprietary (`sbx` CLI, separate from `docker`).
|
||||
- **License**: Proprietary.
|
||||
- **Isolation**: MicroVM (Docker's own implementation) — each session gets
|
||||
its own kernel, Docker daemon inside the VM, and filesystem.
|
||||
- **Locality**: Local (macOS and Windows; does not require Docker Desktop).
|
||||
- **Agent integration**: Explicit wrapper — Claude Code, Codex, Gemini
|
||||
CLI, Copilot CLI, Kiro. Launches agent inside the VM with
|
||||
`--dangerously-skip-permissions` by default.
|
||||
- **Config**: Open / Balanced / Locked Down network presets at launch. No
|
||||
per-role manifest.
|
||||
- **Network policy**: Default-deny; preset levels control strictness. TUI
|
||||
dashboard shows a live log of every outbound connection (allowed and
|
||||
blocked) with point-and-click allow/block for hosts.
|
||||
- **Credentials**: OS keychain + host-side proxy injection — API keys
|
||||
never enter the VM.
|
||||
- **Notable**: Best DX among microVM tools (one command, works like native
|
||||
yolo Claude but inside a VM); branch mode creates a git worktree in
|
||||
`.sbx/`. Network policy is preset-based, not role-declarative.
|
||||
- **Maturity**: GA 2026.
|
||||
|
||||
### Anthropic srt *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/anthropic-experimental/sandbox-runtime
|
||||
(`@anthropic-ai/sandbox-runtime` on npm, `sandbox-runtime` on PyPI)
|
||||
- **License**: Apache 2.0 (experimental).
|
||||
- **Isolation**: OS-level only — Seatbelt (`sandbox-exec`) on macOS,
|
||||
bubblewrap on Linux, WFP (Windows Filtering Platform) account-fenced on
|
||||
Windows. **No container or VM.** Lowest overhead in the set.
|
||||
- **Locality**: Local.
|
||||
- **Agent integration**: Claude Code's sandboxed bash tool uses this
|
||||
internally. Can wrap any arbitrary process (`srt <command>`). Cloud
|
||||
Claude Code sessions use full microVMs instead.
|
||||
- **Config**: Programmatic per-invocation — allow/deny path lists for
|
||||
filesystem; allow/denylist for network (HTTP proxy + SOCKS5).
|
||||
- **Network policy**: Proxy-based filtering (HTTP + SOCKS5); domain
|
||||
allowlist/denylist enforced at proxy layer. Custom proxy supported
|
||||
(e.g. mitmproxy for inspection + audit). Processes that ignore proxy
|
||||
env vars may bypass filtering on some platforms.
|
||||
- **Notable**: Cross-platform (macOS/Linux/Windows); wraps any process,
|
||||
not just agents; no role/manifest concept. Annotated as a research
|
||||
preview — APIs may change.
|
||||
- **Maturity**: Early research preview.
|
||||
|
||||
## Claude-specific wrappers and developer environments
|
||||
|
||||
These projects were the focus of the original containerized-Claude survey.
|
||||
They remain useful comparisons for local developer experience, but most are
|
||||
templates or wrappers rather than policy-bearing sandbox platforms, so they
|
||||
are grouped here instead of widening the main table further.
|
||||
|
||||
### claudebox
|
||||
|
||||
- **Source**: https://github.com/RchGrav/claudebox
|
||||
- **Isolation**: Docker, with per-project images, authentication state, and
|
||||
configuration.
|
||||
- **Agent integration**: Claude Code wrapper with 15+ preconfigured language
|
||||
and task profiles.
|
||||
- **Network policy**: Per-project firewall allowlists.
|
||||
- **Closest overlap**: local one-command developer workflow and project-scoped
|
||||
network policy.
|
||||
- **Difference**: profiles describe development toolchains, not named agent
|
||||
roles. There is no bottle/agent split, composable role manifest, provider
|
||||
plugin layer, or outbound-content DLP.
|
||||
|
||||
### Spritz / claude-code-sandbox
|
||||
|
||||
- **Source**: https://github.com/textcortex/claude-code-sandbox (archived;
|
||||
points to its successor, Spritz).
|
||||
- **Isolation**: The original project ran Claude Code in local Docker with
|
||||
bypass permissions; Spritz moved toward Kubernetes-native multi-agent
|
||||
infrastructure.
|
||||
- **Difference**: the successor targets cluster orchestration rather than a
|
||||
low-dependency local launcher. It is architecturally closer to hosted or
|
||||
Kubernetes platform runtimes than to bot-bottle's single-operator CLI.
|
||||
|
||||
### Trail of Bits claude-code-devcontainer
|
||||
|
||||
- **Source**: https://github.com/trailofbits/claude-code-devcontainer
|
||||
- **Isolation**: A Docker devcontainer that exposes only project files and is
|
||||
designed to run Claude Code with `bypassPermissions` for security audits and
|
||||
untrusted-code review.
|
||||
- **Difference**: a hardened, reusable environment definition rather than an
|
||||
agent launcher or fleet. It has no named-role manifest, per-role credential
|
||||
custody, supervision plane, or multi-backend abstraction.
|
||||
|
||||
### Smaller wrappers and official templates
|
||||
|
||||
Projects such as `arezi/claude-sandbox`, `nkrefman/claude-sandbox`, and
|
||||
`VishalJ99/claude-docker`, plus Docker/Anthropic devcontainer templates, prove
|
||||
there is steady demand for “Claude in a container.” They are deliberately
|
||||
small launch/build configurations. They compete on setup simplicity, not on
|
||||
role-aware policy, credential custody, persistent supervision, or a fleet
|
||||
model, and are better treated as a product category than as individual rows.
|
||||
|
||||
### SuperHQ
|
||||
|
||||
- **Source**: https://superhq.ai/
|
||||
- **Isolation**: Apple-Silicon desktop application using local microVMs via
|
||||
Virtualization.framework/libkrun-era components.
|
||||
- **Agent integration**: Claude Code, Codex, and Pi in a GUI, with mobile
|
||||
remote access.
|
||||
- **Credentials and review**: host-side auth gateway injects credentials on
|
||||
the wire; a temporary overlay stages writes for diff-and-accept review.
|
||||
- **Closest overlap**: local microVM isolation, multi-provider launching, and
|
||||
credential custody for security-minded individual developers.
|
||||
- **Difference**: GUI desktop product on Apple Silicon rather than a
|
||||
cross-platform declarative CLI/fleet layer. The July 2026 snapshot in the
|
||||
original survey recorded a user request for per-run tool-call and network
|
||||
audit logging; treat that as point-in-time rather than a permanent gap.
|
||||
|
||||
## Credential gateway without isolation
|
||||
|
||||
### OneCLI
|
||||
|
||||
[OneCLI](https://onecli.sh/) is a framework-agnostic identity gateway rather
|
||||
than a sandbox. Its phantom-token design gives the agent a placeholder and
|
||||
substitutes the encrypted real credential at the network layer. It therefore
|
||||
matches bot-bottle closely on secret custody, and is more portable because it
|
||||
can sit in front of agents launched by anything, but it supplies no container
|
||||
or VM boundary, filesystem isolation, role manifest, or egress-content DLP.
|
||||
|
||||
The positioning consequence from the earlier survey still holds: secret
|
||||
custody alone is not unique. bot-bottle's relevant combination is local
|
||||
isolation + default-deny egress + payload DLP + declarative roles + credential
|
||||
custody. OneCLI's managed tier also places custody with a third party, whereas
|
||||
bot-bottle keeps it within operator-controlled infrastructure. See
|
||||
[`agent-credential-proxy-landscape.md`](agent-credential-proxy-landscape.md)
|
||||
for the detailed build-versus-adopt analysis.
|
||||
|
||||
## Governance / pre-action authorization layers
|
||||
|
||||
These two tools don't provide VM or filesystem isolation; they intercept
|
||||
tool calls before execution and evaluate them against a per-agent
|
||||
declarative policy. They are the closest competitors on **agent-tailored
|
||||
policy** and complement isolation sandboxes rather than substituting for
|
||||
them.
|
||||
|
||||
### Microsoft Agent Governance Toolkit (AGT) *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/microsoft/agent-governance-toolkit
|
||||
- **License**: MIT (~3.3k stars, open-sourced April 2, 2026).
|
||||
- **Isolation**: None (OS/VM). Execution rings (0–3, inspired by CPU
|
||||
privilege levels) control what an agent can do at the framework layer.
|
||||
MCP security gateway treats MCP traffic as an untrusted boundary.
|
||||
- **Locality**: Embedded in the agent framework (Python, TypeScript, .NET,
|
||||
Rust, Go; 20+ framework adapters).
|
||||
- **Agent integration**: Framework-agnostic. Plugs into Semantic Kernel,
|
||||
AutoGen, and others as a middleware layer.
|
||||
- **Config**: YAML policy per agent — tools can be `allowed`, `denied`,
|
||||
`sandboxed`, or routed through an `approval` step. Every action passes
|
||||
through a governance gate checking: agent DID, trust score, risk tier,
|
||||
requested tool, action type, and policy rules.
|
||||
- **Network policy**: Not directly — operates at tool-call level.
|
||||
- **Credentials**: Per-agent DID (Ed25519 decentralized identifier); agent
|
||||
does not borrow a human's credentials.
|
||||
- **Notable**: Dynamic trust score (0–1,000, behavioral decay) —
|
||||
privilege follows observed behaviour, not just provisioning. Covers all
|
||||
10 OWASP Agentic Top 10 risks. Kill switch + SLO monitoring. Sub-ms
|
||||
policy enforcement.
|
||||
- **Maturity**: MIT, ~3.3k ⭐, v3.7.0 May 2026.
|
||||
|
||||
### Open Agent Passport (OAP) *(added 2026-07-18)*
|
||||
- **Source**: https://github.com/aporthq/aport-spec ; spec at
|
||||
https://api.aport.io/spec/spec/oap/oap-spec.md/ ; arXiv 2603.20953
|
||||
- **License**: Open specification.
|
||||
- **Isolation**: None. Pre-action hook only — intercepts tool calls
|
||||
synchronously before execution, evaluates against a cloud-registry
|
||||
declarative policy, fails closed.
|
||||
- **Locality**: Local hook + cloud policy registry.
|
||||
- **Agent integration**: Framework-agnostic; hook pattern.
|
||||
- **Config**: Declarative policy rules in a cloud registry (evaluated in
|
||||
order; first failing rule denies). Ed25519-signed, hash-chained audit
|
||||
records per decision.
|
||||
- **Network policy**: Not directly.
|
||||
- **Notable**: 53ms median authorization decision (N=1,000). In an
|
||||
adversarial testbed ($5,000 bounty, 1,151 sessions), social engineering
|
||||
succeeded 74.6% of the time under a permissive policy; under a
|
||||
restrictive OAP policy, 0% success across 879 attempts. Assumes
|
||||
framework runtime is not compromised.
|
||||
- **Maturity**: Specification + reference implementation, 2026.
|
||||
|
||||
## Comparison table
|
||||
|
||||
| Axis | bot-bottle | endo-familiar | litterbox | agent-safehouse | matchlock | tilde.run | boxlite | microsandbox | smolmachines | CubeSandbox |
|
||||
|---|---|---|---|---|---|---|---|---|---|---|
|
||||
| Isolation | MicroVM per bottle default (Firecracker/KVM on Linux, Apple Container on macOS) + own egress DLP scanner; Docker legacy fallback, gVisor there if present | Object-capability (no OS isolation) | Podman + opt. Landlock | macOS `sandbox-exec` | MicroVM (Firecracker / Virt.fw) | Hosted container (unverified) | MicroVM (KVM / Hypervisor.fw) | MicroVM (libkrun) | MicroVM (libkrun / KVM) | MicroVM (RustVMM / KVM) |
|
||||
| Local vs hosted | Local | Local | Local (Linux) | Local (macOS) | Local | Hosted SaaS | Local | Local | Local | Self-hosted (server/cluster) |
|
||||
| Open source | Apache 2.0 | Apache 2.0 | Apache 2.0 | Apache 2.0 | MIT | No | Apache 2.0 | Apache 2.0 | Apache 2.0 | Apache 2.0 |
|
||||
| Agent target | Claude Code | Generic (demo) | Generic | Multi-agent wrapper | Generic (+ Claude/OpenAI SDKs) | Claude focus | Generic | Claude + Cursor (MCP/Skills) | Generic (AGENTS.md) | E2B-compatible (platform builders) |
|
||||
| Network policy | Default-deny via own egress scanner + per-bottle allowlist + content DLP + gitleaks on git push | Capability model only | Limited | Not addressed | Default-deny + allowlist + secret-injecting proxy | Default-deny + logging | Per-VM net (unverified) | Not documented | Off by default + allowlist | Default-deny allowlist + instant egress block + audit logs + per-sandbox tokens (eBPF) + credential vault |
|
||||
| Parallel agents | Yes (one bottle per agent) | n/a | Not addressed | One at a time | Multiple VMs | Yes (dashboard) | SDK-level | SDK-level | Architectural | Yes (2,000+/host claimed) |
|
||||
| Long-running posture | Persistent by default (named, supervised) | n/a (demo) | Session (up while in use) | Per-invocation | Ephemeral VM per run | Per-run (versioned) | Ephemeral + snapshot/fork | Ephemeral / on-demand | Named persistent by default | Ephemeral + auto pause/resume |
|
||||
| DX: run Claude yolo-style | One command → interactive yolo Claude (`start <agent>`, `--dangerously-skip-permissions` default) | n/a (lib demo) | Wizard + build, then run claude inside (Linux only) | One-command wrapper (`safehouse claude --dangerously-skip-permissions`) | CLI: run a cmd in a VM (not a Claude wrapper) | Hosted (`tilde exec`), not local-native | SDK code required (build the run yourself) | CLI/MCP: sandbox-as-a-tool for the agent, not a wrapper around it | SSH into a named machine, run claude there | Stand up a cluster + drive via E2B SDK |
|
||||
| Config | JSON manifest (bottles + agents) | Programmatic refs | CLI wizard | Profile files / shell fns | CLI / SDK | DSL + CLI + SDK | SDK | CLI / SDK / MCP | TOML Smolfile | E2B-compatible SDK |
|
||||
| Maturity | Active May 2026 | Research (2022+) | Early (~66 ⭐) | Active (~1.4k ⭐) | Experimental (~574 ⭐) | Private preview | YC, ~4.7k ⭐ | YC, ~6k ⭐, beta | ~3.1k ⭐ | Tencent, prod, ~10.4k ⭐ |
|
||||
*Isolation/sandbox tools only. AGT and OAP are governance layers — see their per-project notes above.*
|
||||
|
||||
| Axis | bot-bottle | endo-familiar | litterbox | agent-safehouse | matchlock | tilde.run | boxlite | microsandbox | smolmachines | E2B | Daytona | CubeSandbox | Cleanroom | container-use | Docker sbx | Anthropic srt |
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
| Isolation | MicroVM per bottle default (Firecracker/KVM on Linux, Apple Container on macOS) + own egress DLP scanner; Docker legacy fallback, gVisor there if present | Object-capability (no OS isolation) | Podman + opt. Landlock | macOS `sandbox-exec` | MicroVM (Firecracker / Virt.fw) | Hosted container (unverified) | MicroVM (KVM / Hypervisor.fw) | MicroVM (libkrun) | MicroVM (libkrun / KVM) | Firecracker microVM | Container or Linux/Windows VM class | MicroVM (RustVMM / KVM) | MicroVM (Firecracker / Virt.fw) | Docker container + git worktree | MicroVM (proprietary) | OS-level (Seatbelt / bubblewrap / WFP) — no container |
|
||||
| Local vs hosted | Local | Local | Local (Linux) | Local (macOS) | Local | Hosted SaaS | Local | Local | Local | Hosted; self-host/BYOC available | Hosted; dedicated/custom regions | Self-hosted (server/cluster) | Self-hosted server | Local | Local | Local |
|
||||
| Open source | Apache 2.0 | Apache 2.0 | Apache 2.0 | Apache 2.0 | MIT | No | Apache 2.0 | Apache 2.0 | Apache 2.0 | Apache 2.0 | Production closed-source; legacy AGPL repo unmaintained | Apache 2.0 | Apache 2.0 | Apache 2.0 | Proprietary | Apache 2.0 (experimental) |
|
||||
| Agent target | Claude Code, Codex, Pi, and provider plugins | Generic (demo) | Generic | Multi-agent wrapper | Generic (+ Claude/OpenAI SDKs) | Claude focus | Generic | Claude + Cursor (MCP/Skills) | Generic (AGENTS.md) | LLM-agnostic platform builders | LLM-agnostic platform builders | E2B-compatible (platform builders) | CI / generic process | Claude Code, Cursor, Windsurf (MCP) | Claude Code, Codex, Gemini CLI, Copilot, Kiro | Claude Code (and any process) |
|
||||
| Network policy | Default-deny via own egress scanner + per-bottle allowlist + content DLP + gitleaks on git push | Capability model only | Limited | Not addressed | Default-deny + allowlist + secret-injecting proxy | Default-deny + logging | Per-VM net (unverified) | Not documented | Off by default + allowlist | Per-sandbox allow/deny rules and custom egress proxy; internet configurable | Per-sandbox CIDR/domain allowlist or block-all; tier policy; secret-injecting proxy | Default-deny allowlist + instant egress block + audit logs + per-sandbox tokens (eBPF) + credential vault | Default-deny + per-repo host allowlist (cleanroom.yaml) | Not addressed | Default-deny; Open / Balanced / Locked Down presets; live TUI network panel | Proxy-based allowlist/denylist (HTTP + SOCKS5); custom proxy supported |
|
||||
| Parallel agents | Yes (one bottle per agent) | n/a | Not addressed | One at a time | Multiple VMs | Yes (dashboard) | SDK-level | SDK-level | Architectural | Yes (platform service) | Yes (platform service) | Yes (2,000+/host claimed) | Yes (server model) | Yes (per-agent containers + worktrees) | Yes | Yes |
|
||||
| Long-running posture | Persistent by default (named, supervised) | n/a (demo) | Session (up while in use) | Per-invocation | Ephemeral VM per run | Per-run (versioned) | Ephemeral + snapshot/fork | Ephemeral / on-demand | Named persistent by default | Runtime tier limits + indefinite pause/resume | Persistent filesystem; VM pause/resume; configurable auto-stop | Ephemeral + auto pause/resume | Per-run + suspend/resume | Per-agent container (ephemeral) | Per-session; branch mode creates git worktree in .sbx/ | Per-invocation |
|
||||
| DX: run Claude yolo-style | One command → interactive yolo Claude (`start <agent>`, `--dangerously-skip-permissions` default) | n/a (lib demo) | Wizard + build, then run claude inside (Linux only) | One-command wrapper (`safehouse claude --dangerously-skip-permissions`) | CLI: run a cmd in a VM (not a Claude wrapper) | Hosted (`tilde exec`), not local-native | SDK code required (build the run yourself) | CLI/MCP: sandbox-as-a-tool for the agent, not a wrapper around it | SSH into a named machine, run claude there | SDK/CLI sandbox; wire the agent yourself | SDK/CLI sandbox; wire the agent yourself | Stand up a cluster + drive via E2B SDK | CI-oriented, not a Claude wrapper | MCP server: `claude mcp add container-use -- container-use stdio` | One command: `sbx` wraps claude with `--dangerously-skip-permissions` default | Library/wrapper, not a standalone CLI |
|
||||
| Config | YAML-in-Markdown manifests (bottles + agents) | Programmatic refs | CLI wizard | Profile files / shell fns | CLI / SDK | DSL + CLI + SDK | SDK | CLI / SDK / MCP | TOML Smolfile | SDK/API + templates | SDK/API/CLI + images/snapshots | E2B-compatible SDK | cleanroom.yaml in repo | None (no policy config) | Preset levels at launch | Programmatic per-invocation (allow/deny lists) |
|
||||
| Agent-tailored policy | Yes — bottle/agent split; declarative per-role egress + credentials; composable via `extends:` | Partial — capability model scopes per-agent, but no declarative role manifest | No | Partial — per-agent profile files (Seatbelt); no egress | No | Yes — per-agent DSL RBAC (allow/deny/approve per action/repo/agent) | No | No | No | No — per-sandbox SDK config | No — per-sandbox SDK config | No — per-sandbox SDK config, not role-scoped | Partial — per-repo cleanroom.yaml, not per-role | No | No — network presets only | No |
|
||||
| Maturity | Active July 2026 | Research (2022+) | Early (~66 ⭐) | Active (~1.8k ⭐) | Experimental (~574 ⭐) | Private preview | YC, ~4.7k ⭐ | YC, ~6k ⭐, beta | ~3.1k ⭐ | Established hosted platform, ~12.4k ⭐ | Production commercial; closed-source since June 2026 | Tencent, prod, ~10.4k ⭐ | Active (Buildkite product) | Early development | GA 2026 | Early research preview |
|
||||
|
||||
## What's closest, what's different
|
||||
|
||||
@@ -233,55 +568,121 @@ existing OS primitive, low-dep. The split is the isolation primitive —
|
||||
bot-bottle now defaults to a VM per bottle (Firecracker microVM on KVM
|
||||
Linux, Apple Container on macOS) with its own DLP-scanning egress proxy,
|
||||
keeping Docker only as a legacy fallback; agent-safehouse uses
|
||||
`sandbox-exec`; litterbox
|
||||
uses Podman + Landlock. matchlock and smolmachines are close on *both* the
|
||||
policy side (default-deny net, per-host allowlist) and — now that
|
||||
bot-bottle has moved off containers-by-default — the microVM isolation
|
||||
primitive.
|
||||
`sandbox-exec`; litterbox uses Podman + Landlock. matchlock and
|
||||
smolmachines are close on *both* the policy side (default-deny net,
|
||||
per-host allowlist) and — now that bot-bottle has moved off
|
||||
containers-by-default — the microVM isolation primitive. Note: Apple
|
||||
Container 1.0 stable shipped June 9 2026 (frozen CLI and APIs), which
|
||||
makes the macOS backend stable surface area rather than a moving target.
|
||||
|
||||
**New closest on agent-tailored policy.** Two governance tools are the
|
||||
direct competitors on the "coarse-grained sandbox" axis. **tilde.run**
|
||||
has had per-agent DSL RBAC since its launch (though it's hosted SaaS).
|
||||
**Microsoft AGT** is the most serious new entrant: per-agent DID
|
||||
identity, YAML policy that can allow/deny/sandbox/approve individual tool
|
||||
calls per agent, and a dynamic behavioural trust score. It operates at
|
||||
the framework tool-call layer, not the network layer — so it's
|
||||
complementary to bot-bottle's network/filesystem isolation rather than a
|
||||
direct substitute, but on the "does this sandbox know what this agent is
|
||||
for?" question it is the most complete answer in the field. OAP's
|
||||
pre-action hook pattern achieves similar goals with cryptographic audit
|
||||
and a 0% adversarial-attack success rate under a restrictive policy.
|
||||
|
||||
**New closest on DX.** **Docker sbx** is the first tool in this set that
|
||||
matches bot-bottle on the "one command, dangerously-skip-permissions safe
|
||||
by default" DX bar, at microVM isolation strength, with host-side
|
||||
credential injection. It is proprietary, preset-based (not role-
|
||||
declarative), and cloud-agent-specific, but it directly competes on the
|
||||
UX proposition. agent-safehouse was the previous DX peer; Docker sbx
|
||||
materially raises the bar.
|
||||
|
||||
**New closest on repo-scoped policy.** **Cleanroom** (Buildkite) is the
|
||||
first tool to combine microVM isolation with a declarative egress policy
|
||||
file — though the policy lives in the repo being sandboxed
|
||||
(`cleanroom.yaml`), not in an agent-role manifest. That makes it per-
|
||||
repo rather than per-role: the same Cleanroom config applies to any
|
||||
agent running in that repo. The distinction matters for bot-bottle's
|
||||
use case (one developer running multiple agent *roles* with different
|
||||
egress footprints), but for CI/CD use cases Cleanroom is a direct
|
||||
alternative.
|
||||
|
||||
**Solving a different problem.** tilde.run is hosted SaaS for team /
|
||||
production agent pipelines with data-versioned rollback — explicitly
|
||||
opposite to bot-bottle's "infrastructure I control" goal. boxlite,
|
||||
microsandbox, and CubeSandbox are infrastructure libraries/services aimed
|
||||
at platform builders embedding sandboxes into agent frameworks; they
|
||||
opposite to bot-bottle's "infrastructure I control" goal. E2B and Daytona
|
||||
are hosted sandbox platforms, while boxlite, microsandbox, and CubeSandbox
|
||||
are infrastructure libraries/services aimed at platform builders embedding
|
||||
sandboxes into agent frameworks; they
|
||||
would be a *backend* bot-bottle could call, not a competitor to its
|
||||
manifest layer. endo-familiar is in a different paradigm entirely:
|
||||
capability passing rather than kernel boundaries.
|
||||
|
||||
## Borrowable ideas
|
||||
|
||||
What bot-bottle already has that the survey suggested as
|
||||
differentiators:
|
||||
### Already shipped or otherwise addressed
|
||||
|
||||
- Default-deny egress with a per-agent allowlist (own egress scanner).
|
||||
- DLP scanning of outbound traffic.
|
||||
- Bottle / agent split (manifest layer above the isolation primitive).
|
||||
- gVisor auto-detection on Linux.
|
||||
- **In-flight secret injection** (suggested by matchlock) — **shipped.**
|
||||
Real provider and git-host tokens are held outside the agent and injected
|
||||
by the egress gateway on matching routes. The agent receives only proxy
|
||||
URLs and, where a client requires a credential-shaped value, a placeholder;
|
||||
`GITEA_TOKEN` and equivalent real tokens do not appear in the agent's
|
||||
environment.
|
||||
- **MicroVM backend** — **shipped.** MicroVMs are now the default:
|
||||
Firecracker on KVM Linux and Apple Container on macOS. Docker is the legacy
|
||||
fallback.
|
||||
- **Per-use SSH key confirmation** (suggested by litterbox) — **addressed by
|
||||
stronger credential custody instead.** The agent does not hold the upstream
|
||||
git SSH key or an SSH-agent socket: git-gate holds the credential and gates
|
||||
git operations. A confirmation wrapper inside the agent would therefore
|
||||
protect a credential that is no longer there. Operator approval at the gate
|
||||
remains the appropriate control point for any future per-use confirmation.
|
||||
|
||||
Ideas worth considering, without abandoning the Python-stdlib-first /
|
||||
local, single-operator stance:
|
||||
### Still worth considering
|
||||
|
||||
1. **Per-use SSH key confirmation** (from litterbox). Even with
|
||||
KnownHostKey pinning and the egress DLP scanner, a wrapper SSH agent that
|
||||
prompts on each key use (e.g. via `osascript` / `notify-send`) would
|
||||
catch an agent doing something off-policy with a key it legitimately
|
||||
holds. Pure-stdlib, no new deps.
|
||||
2. **In-flight secret injection** (from matchlock). The egress scanner
|
||||
already does allowlisting and DLP; teaching it to *inject* tokens at
|
||||
proxy time so e.g. `GITEA_TOKEN` never appears in the container's
|
||||
env would close the "agent reads its own env and exfiltrates" path.
|
||||
Fits the existing egress-proxy architecture.
|
||||
3. **MicroVM backend** — ~~on the radar~~ **shipped since this survey.**
|
||||
microVMs are now bot-bottle's default (Firecracker on KVM Linux, Apple
|
||||
Container on macOS); Docker is the legacy fallback. The libkrun / Apple
|
||||
Virtualization.framework ergonomics that microsandbox, smolmachines,
|
||||
and matchlock demonstrated turned out to be enough to make it the
|
||||
default rather than an opt-in.
|
||||
- **Live network activity in the supervisor TUI** (from Docker sbx): show
|
||||
allowed and blocked connections and let the operator propose policy changes
|
||||
from the existing supervision surface.
|
||||
- **Tamper-evident audit records** (from OAP): sign and hash-chain egress and
|
||||
supervision decisions for compliance-sensitive deployments.
|
||||
- **Behaviour-informed policy downgrade** (from Microsoft AGT): use repeated
|
||||
DLP alerts or supervision holds as a signal to narrow policy or request
|
||||
closer review. This needs a carefully specified trust model before it can be
|
||||
more than a heuristic.
|
||||
|
||||
Not worth borrowing: the SDK-first programmatic API style of boxlite /
|
||||
microsandbox (cuts against the declarative-manifest stance), and the
|
||||
hosted-SaaS dashboard model of tilde.run (cuts against the
|
||||
"infrastructure I control" goal).
|
||||
|
||||
## Publishing and positioning verdict
|
||||
|
||||
Publishing remains worthwhile, but the defensible claim is the combination,
|
||||
not any single primitive. Credential custody is matched by OneCLI, matchlock,
|
||||
Daytona, Docker sbx, and CubeSandbox; local one-command isolation is matched by
|
||||
agent-safehouse and Docker sbx; hosted microVM execution is a crowded platform
|
||||
category.
|
||||
|
||||
bot-bottle remains unusual in combining:
|
||||
|
||||
- local, operator-controlled execution with persistent named bottles;
|
||||
- one declarative role layer across Claude Code, Codex, Pi, and provider
|
||||
plugins;
|
||||
- composable agent/bottle manifests, skills, and system prompts;
|
||||
- Firecracker/Apple Container isolation with a Docker fallback;
|
||||
- default-deny per-role egress, payload DLP, and git-push secret scanning;
|
||||
- credentials injected outside the agent process; and
|
||||
- supervision and audit state suited to long-running parallel agents.
|
||||
|
||||
The practical wedge is “as easy as native yolo, with declarative role policy
|
||||
and self-hosted custody,” including scoped access to private LAN/Tailnet
|
||||
services that cloud-first runtimes cannot provide without additional network
|
||||
plumbing. The main competitive risks are a local wrapper such as claudebox or
|
||||
Docker sbx growing a role-manifest layer, and GUI products such as SuperHQ
|
||||
adding equivalent policy and audit depth.
|
||||
|
||||
## Caveats
|
||||
|
||||
- Star counts and last-commit dates are point-in-time snapshots.
|
||||
@@ -298,7 +699,8 @@ hosted-SaaS dashboard model of tilde.run (cuts against the
|
||||
|
||||
CubeSandbox (Tencent Cloud, Apache 2.0, ~10.4k stars, HN launch
|
||||
[#47863430](https://news.ycombinator.com/item?id=47863430)) is the first
|
||||
project in this survey to combine, in one open-source stack, everything
|
||||
open-source, self-hostable project in this survey to combine, in one stack,
|
||||
the main primitives
|
||||
bot-bottle treated as its differentiator:
|
||||
|
||||
- **Egress custody (connection level)** — default-deny domain allowlist
|
||||
@@ -366,7 +768,7 @@ instead of per-action prompts. On this axis the field splits cleanly:
|
||||
network egress; bot-bottle adds VM-grade isolation, egress DLP, and
|
||||
persistent/parallel bottles across macOS + Linux.
|
||||
- **Libraries / services** (you build the run yourself): boxlite,
|
||||
microsandbox, CubeSandbox, E2B. These hand you an SDK or a cluster and
|
||||
microsandbox, CubeSandbox, E2B, Daytona. These hand you an SDK or a cluster and
|
||||
expect you to wire the agent in — powerful for platform builders,
|
||||
heavyweight for "just run Claude on my laptop." microsandbox's MCP/Skills
|
||||
angle is *sandbox-as-a-tool the agent calls*, which is the inverse of
|
||||
@@ -374,10 +776,11 @@ instead of per-action prompts. On this axis the field splits cleanly:
|
||||
- **In between:** litterbox (wizard + build, Linux only), smolmachines
|
||||
(SSH into a named machine), matchlock (run a command in a VM).
|
||||
|
||||
So DX is a genuine bot-bottle differentiator, and the only project that
|
||||
matches it (agent-safehouse) does so with materially weaker isolation and
|
||||
no egress story. "As easy as native yolo, but actually sandboxed" is a
|
||||
defensible one-liner.
|
||||
So DX is a genuine bot-bottle differentiator. agent-safehouse matches the
|
||||
one-command wrapper with weaker isolation and no egress story; Docker sbx now
|
||||
matches it at microVM strength but remains proprietary and preset-based. "As
|
||||
easy as native yolo, with declarative role policy" is the narrower defensible
|
||||
one-liner.
|
||||
|
||||
Why it still doesn't collide head-on:
|
||||
|
||||
@@ -385,7 +788,8 @@ Why it still doesn't collide head-on:
|
||||
builders* (drop-in E2B replacement, SDK-driven, 2,000 sandboxes on a
|
||||
box). bot-bottle is a *single-operator, declarative-manifest tool for
|
||||
the infrastructure I run*. Different buyer, different ergonomics — no
|
||||
JSON manifest, no bottle/agent split, no "one command on my laptop."
|
||||
declarative role manifest, no bottle/agent split, no "one command on my
|
||||
laptop."
|
||||
2. **Backend, not competitor.** Like boxlite/microsandbox, CubeSandbox is
|
||||
something bot-bottle could sit *on top of* — a `"runtime": "microvm"`
|
||||
or `"runtime": "cubesandbox"` backend under the manifest layer — while
|
||||
@@ -396,7 +800,8 @@ Why it matters anyway:
|
||||
|
||||
- The "nobody else bundles connection-level egress allowlist + audit +
|
||||
in-flight credential custody" line is **no longer true for the
|
||||
primitive** — a well-funded, 10k-star open-source project now ships it.
|
||||
primitive** — CubeSandbox ships the open-source/self-hosted combination,
|
||||
and Daytona ships a proprietary firewall + credential-substitution variant.
|
||||
But **content DLP on authorized channels is still not matched** (see
|
||||
above), and neither is the *layer above* the primitive (declarative
|
||||
manifest, cross-vendor orchestration, operator UX, the
|
||||
@@ -407,5 +812,59 @@ Why it matters anyway:
|
||||
that in mind.
|
||||
- Worth a closer look at **how** CubeSandbox does credential injection
|
||||
and per-sandbox egress tokens (eBPF virtual switch vs. bot-bottle's
|
||||
mitmproxy egress proxy) before the next iteration of bot-bottle's
|
||||
in-flight-secret feature — see borrowable idea #2 above.
|
||||
mitmproxy egress proxy) when hardening bot-bottle's now-shipped
|
||||
credential-custody implementation.
|
||||
|
||||
## Addendum 2026-07-18 (second pass) — agent-tailored policy landscape
|
||||
|
||||
The second-pass question was: how novel is bot-bottle's per-agent,
|
||||
role-tailored sandbox relative to the expanded field?
|
||||
|
||||
**The short answer:** on the isolation + network + role-tailoring
|
||||
combination, bot-bottle remains the only tool in this set. On
|
||||
role-tailored *policy at the tool-call level*, Microsoft AGT and OAP are
|
||||
the most complete answers, but they don't provide isolation; they
|
||||
complement rather than substitute.
|
||||
|
||||
**The competitive picture by axis:**
|
||||
|
||||
- *Agent-tailored egress (declarative, per-role)* — bot-bottle and
|
||||
tilde.run. Cleanroom is per-repo, not per-role. Everyone else is
|
||||
per-session or not addressed.
|
||||
- *Agent-tailored tool-call policy (declarative, per-agent identity)* —
|
||||
Microsoft AGT (YAML policy + DID identity + trust score), OAP
|
||||
(declarative policy rules + cryptographic audit). Neither provides
|
||||
network/filesystem isolation.
|
||||
- *Composable policy (role overlays)* — bot-bottle (`extends:`). No
|
||||
other tool surveyed supports composable role-policy inheritance.
|
||||
- *Isolation + DX (one-command safe yolo)* — bot-bottle and Docker sbx.
|
||||
Docker sbx is proprietary, preset-based, and cloud-agent-specific;
|
||||
it's the first DX-class competitor at microVM isolation strength.
|
||||
|
||||
**What the HN "coarse-grained" complaint maps to:** The complaint is
|
||||
that a VM isolates the filesystem but doesn't know if the agent
|
||||
*should* be sending an email. bot-bottle's bottle/agent split is a
|
||||
structural answer to this: the bottle manifest declares exactly what
|
||||
the role can reach, and the sandbox enforces it at the network layer.
|
||||
Microsoft AGT is the most complete answer at the semantic/tool-call
|
||||
layer. The gap both leave open is *intent classification* — knowing
|
||||
whether a permitted action is consistent with the agent's actual task.
|
||||
See `hn-agent-safety-discourse-july-2026.md` for the blast-radius
|
||||
analysis.
|
||||
|
||||
**Open ideas from new tools (also summarized above):**
|
||||
|
||||
- **Microsoft AGT's trust-score decay** — privilege that reflects
|
||||
observed behaviour rather than static provisioning. Applied to
|
||||
bot-bottle: a bottle that has triggered DLP alerts or supervise holds
|
||||
could auto-downgrade its network preset, or flag the session for
|
||||
closer review. Fits the existing supervise-server architecture.
|
||||
- **Docker sbx's live network TUI** — real-time per-session view of
|
||||
allowed and blocked outbound connections with point-and-click
|
||||
allow/block. `cli.py supervise` is the right surface; adding a
|
||||
live-connections panel would directly address the "I can't see what
|
||||
the agent is doing" gap without any backend changes.
|
||||
- **OAP's cryptographic audit chain** — Ed25519-signed, hash-chained
|
||||
audit records. Currently bot-bottle logs egress decisions but doesn't
|
||||
chain them. A tamper-evident audit record per session would be useful
|
||||
for the compliance use case the CubeSandbox positioning targets.
|
||||
|
||||
@@ -0,0 +1,357 @@
|
||||
# HN discourse on agent sandbox safety — June/July 2026
|
||||
|
||||
A survey of community opinion and notable security disclosures on Hacker
|
||||
News and adjacent sources over June–July 2026. The question: what does
|
||||
the current discourse say about whether sandboxes are sufficient for
|
||||
agentic AI safety, and where does bot-bottle land against the issues
|
||||
being raised?
|
||||
|
||||
Research conducted 2026-07-18.
|
||||
|
||||
## Summary
|
||||
|
||||
The past month marks a turning point in community opinion. Earlier in
|
||||
2026, the debate was mostly "which sandbox tool is best?" By June–July,
|
||||
a cascade of critical CVEs and novel attack classes has shifted the
|
||||
framing to "sandboxes are not enough — what else do you need?" The
|
||||
attacks that drove this shift are structurally distinct: most route
|
||||
through legitimate, trusted channels (Sentry issues, MCP descriptions,
|
||||
README files) rather than exploiting the isolation boundary directly.
|
||||
|
||||
bot-bottle's architecture holds up well against the direct-escape class
|
||||
(Firecracker/Apple Container default backends, credentials never in the
|
||||
agent's env, harness entirely on the host). The remaining gap is prompt
|
||||
injection — attacker-controlled data interpreted as model instructions.
|
||||
Egress controls and prompt injection defenses are orthogonal: egress
|
||||
limits what the agent can *send out*; injection is about what it is
|
||||
*told to do*. The two don't substitute for each other. Inside a tightly-
|
||||
egressed sandbox a successful injection can't exfiltrate to unknown
|
||||
hosts, but it can still corrupt the work product, push malicious commits
|
||||
past a secret scanner, or use allowlisted channels for exfiltration.
|
||||
Those residual risks are addressed below.
|
||||
|
||||
## The sandboxing boom sets the stage
|
||||
|
||||
The preceding months generated a wave of sandbox tooling. A March 28
|
||||
Ask HN thread
|
||||
([#47444917](https://news.ycombinator.com/item?id=47444917)) catalogued
|
||||
the explosion: E2B, AIO Sandbox, AgentSphere, Yolobox, Exe.dev,
|
||||
AgentFence, DenoSandbox, Capsule (WASM), ERA, Vibekit, Daytona, Modal,
|
||||
Nono, and more — all launched within roughly 12 months. A parallel March
|
||||
9 thread ([#47185250](https://news.ycombinator.com/item?id=47185250))
|
||||
surveyed what developers were actually deploying: "containers or YOLO"
|
||||
dominated. The honest community mood was that most teams hadn't solved
|
||||
this and were shipping anyway.
|
||||
|
||||
The March 12 launch of **Agent Safehouse**
|
||||
([#47301085](https://news.ycombinator.com/item?id=47301085), 823 points)
|
||||
crystallised the community framing: a zero-dep `sandbox-exec` wrapper for
|
||||
macOS that attracted the top comment *"I honestly think that sandboxing is
|
||||
currently THE major challenge that needs to be solved for the tech to fully
|
||||
realise its potential."* The creator's own framing — "no dependencies, no
|
||||
daemons, no subscription; the simplicity is the feature" — and Simon
|
||||
Willison's observation that evaluating whether a sandboxing tool works as
|
||||
intended is itself hard, both prefigure the June–July shift in tone. See
|
||||
[`agent-sandbox-landscape.md`](agent-sandbox-landscape.md) for a full
|
||||
per-project breakdown.
|
||||
|
||||
## The June–July attack cascade
|
||||
|
||||
Six attack patterns broke in quick succession. Together they form the
|
||||
argument that the community's framing was wrong: the threat model for
|
||||
agents isn't just "code that escapes its container" — it's also prompt
|
||||
injection, where attacker-controlled data is interpreted as model
|
||||
instructions regardless of whether any isolation boundary was crossed.
|
||||
Sections 2–4 below are all the same attack class; the "trusted channel"
|
||||
label describes the delivery vector, not a different threat.
|
||||
|
||||
### 1. Sandbox escape CVEs (DuneSlide, CVE-2026-39861)
|
||||
|
||||
Cato AI Labs disclosed **DuneSlide** (CVE-2026-50548/50549, CVSS 9.8),
|
||||
a pair of flaws in Cursor 2.x. CVE-2026-50548 abuses the sandbox's
|
||||
`working_directory` parameter to point writes at system files; CVE-26-50549
|
||||
exploits a symlink-resolution fallback that fails open. Both start with
|
||||
a prompt injection and end in sandbox escape — and Cato's framing was
|
||||
blunt: "each CVE defeats a different guardrail; the problem is
|
||||
structural, not a string of one-offs."
|
||||
|
||||
Claude Code's own sandbox had a similar escape this year:
|
||||
**CVE-2026-39861** (symlink flaw). The CurXecute/MCPoison/CVE-2026-26268
|
||||
chain from Cursor added a poisoned Slack message, a swap-after-approval
|
||||
MCP config, and a Git hook as three more entry points in the same
|
||||
attack class.
|
||||
|
||||
All patched, but the pattern holds: any application-level sandbox that
|
||||
takes attacker-influenced values as path parameters is reachable from a
|
||||
prompt injection.
|
||||
|
||||
### 2. Prompt injection via MCP data (Agentjacking)
|
||||
|
||||
Tenet's "Agentjacking" technique planted a fake bug report in Sentry's
|
||||
MCP output. When an agent queries Sentry to fix open issues, the
|
||||
malicious event is rendered as structured content visually
|
||||
indistinguishable from a real Sentry event, and the agent executes the
|
||||
embedded instructions with the developer's full privileges. Hit rate
|
||||
across Claude Code and Cursor: **85%**. The route is entirely through a
|
||||
legitimately-authorized MCP channel — no isolation boundary is crossed;
|
||||
the injection arrives inbound through a channel the sandbox explicitly
|
||||
trusts.
|
||||
|
||||
The Cloud Security Alliance's summary: treat observability, bug-report,
|
||||
and integration data as **untrusted agent input**, not neutral
|
||||
development metadata.
|
||||
|
||||
### 3. README-embedded prompt injection
|
||||
|
||||
A July disclosure showed malicious instructions hidden in `README.md`
|
||||
— a file that receives no trust prompt and requires no elevated access.
|
||||
When asked point-blank whether the repo held hidden instructions, both
|
||||
Claude Sonnet 4.6 and GPT-5.5 said no. A payload written for Sonnet
|
||||
4.6 transferred unchanged to Sonnet 5, Opus 4.8, and GPT-5.5. The
|
||||
attack surface is every repo an agent is asked to work in.
|
||||
|
||||
### 4. Prompt injection via MCP tool descriptions
|
||||
|
||||
Microsoft research (June 30) showed that attacker-controlled MCP tool
|
||||
description fields can silently redirect agent behavior. The injection
|
||||
is embedded in metadata the model reads during tool selection — before
|
||||
any sandbox enforcement or egress check runs, and entirely on the
|
||||
inbound path that egress controls cannot touch.
|
||||
|
||||
### 5. MCP STDIO command injection (10 CVEs)
|
||||
|
||||
OX Security disclosed a systemic command injection class in Anthropic's
|
||||
MCP protocol, covering 10 CVEs across multiple coding agents. The
|
||||
Windsurf case (CVE-2026-30615): processing attacker-controlled HTML
|
||||
causes the agent to auto-register a malicious MCP STDIO server and
|
||||
execute arbitrary commands with no further user interaction.
|
||||
|
||||
### 6. LiteLLM gateway compromise (CVE-2026-40217, CVE-2026-42271)
|
||||
|
||||
CVE-2026-40217 exposes LiteLLM's guardrail sandbox via `exec()` with no
|
||||
source filtering. CVE-2026-42271 (exploited in the wild, added to CISA's
|
||||
KEV catalog) lets callers spawn subprocesses through MCP preview
|
||||
endpoints. The threat extends to any agent routed through a compromised
|
||||
LiteLLM proxy: the proxy can swap model responses for forged tool calls
|
||||
in transit, giving the attacker a reverse shell from the developer's
|
||||
machine.
|
||||
|
||||
## HN community opinion clusters
|
||||
|
||||
**"Move enforcement to the kernel, not the app"** — the Nono Show HN
|
||||
([#46849615](https://news.ycombinator.com/item?id=46849615)) and a
|
||||
kernel-sandbox thread
|
||||
([#47066574](https://news.ycombinator.com/item?id=47066574)) both argued
|
||||
that application-layer sandboxes are inherently bypassable by the code
|
||||
they're sandboxing. The academic framing, from *Red-Teaming the Agentic
|
||||
Red-Team* ([arXiv 2606.24496](https://arxiv.org/pdf/2606.24496)):
|
||||
"enforcement should occur at the OS level via the kernel refusing system
|
||||
calls that violate policy at runtime — not pre-execution argument
|
||||
validation in tool calls."
|
||||
|
||||
**"The harness belongs outside the sandbox"** — a May thread
|
||||
([#47990675](https://news.ycombinator.com/item?id=47990675)) converged
|
||||
on clean architectural separation: harness in one VM, tool execution in
|
||||
another. Top comment: "having the harness in one VM, and tool use applied
|
||||
to user data in another, is about as safe as you can be at present."
|
||||
Several replies described a hypervisor-like policy layer — sitting outside
|
||||
both VMs — as the right long-term model.
|
||||
|
||||
**"Sandboxes are too coarse-grained"** — a Feb thread
|
||||
([#47006445](https://news.ycombinator.com/item?id=47006445)) argued
|
||||
that VMs don't answer the real question: knowing whether an agent
|
||||
*should* be sending an email or making a transaction. "Everything's just
|
||||
in the same big box." This framing picked up traction through June–July
|
||||
as the trusted-channel attacks dominated.
|
||||
|
||||
**"MCP's trust model is the real problem"** — the month's recurring
|
||||
theme. MCP by design gives agents access to authorized external services.
|
||||
Once a trusted channel delivers a malicious payload, filesystem sandboxing
|
||||
is irrelevant. The community call: treat all MCP tool metadata and return
|
||||
values as untrusted input subject to policy validation before ingestion,
|
||||
and disable automatic MCP server loading from untrusted repositories.
|
||||
|
||||
## How bot-bottle addresses these issues
|
||||
|
||||
### What it covers well
|
||||
|
||||
**Direct sandbox escape (CVEs, container breakout)**
|
||||
|
||||
bot-bottle's default backends are Firecracker microVM (KVM Linux) and
|
||||
Apple Container (macOS). Both run the agent in a separate VM with a
|
||||
dedicated kernel — the container-escape CVE class (Dirty Pipe, runc
|
||||
escapes, DuneSlide's path-parameter abuse) requires escaping a real
|
||||
hypervisor boundary, not just a namespace. On the legacy Docker backend,
|
||||
gVisor auto-detection provides a userspace syscall barrier for hosts where
|
||||
neither KVM nor Apple Container is available.
|
||||
|
||||
The bot-bottle process itself runs entirely on the host, outside the VM.
|
||||
This is the "harness outside the sandbox" architecture the HN thread
|
||||
converged on as best practice. The bottle manifest, egress rules, and
|
||||
secrets never enter the agent VM.
|
||||
|
||||
**Credential theft on sandbox escape**
|
||||
|
||||
Even on a successful VM/container escape, the agent has nothing useful
|
||||
to steal. Credentials are injected in-flight by the gateway proxy
|
||||
(`auth.scheme` / `auth.token_ref` in the egress route config) — `printenv`
|
||||
inside the agent shows proxy URLs only. The git-gate similarly holds the
|
||||
upstream SSH credential on the host; the agent pushes through a
|
||||
gitleaks-scanned daemon that forwards clean refs upstream. An escaped
|
||||
agent gets the host filesystem, not the keys.
|
||||
|
||||
**Orphaned-agent credential risk**
|
||||
|
||||
bot-bottle is explicitly ephemeral: when the agent exits, `cli.py` tears
|
||||
down every gateway and both networks — nothing persists between runs. The
|
||||
agent never holds credentials, so there is nothing to orphan.
|
||||
|
||||
**MCP config redirection / STDIO auto-registration**
|
||||
|
||||
The trust boundary at `$HOME` means bottles live only under
|
||||
`~/.bot-bottle/bottles/` — a cloned repo cannot add egress routes or
|
||||
redirect env vars to attacker hosts (the design rationale is in
|
||||
`docs/prds/0011-per-file-md-manifest.md`). Auto-registering a malicious
|
||||
MCP STDIO server from within the agent is still sandboxed by the VM, and
|
||||
any outbound calls from that server must pass the egress allowlist and
|
||||
outbound DLP scanner.
|
||||
|
||||
**Per-agent role tailoring (the "coarse-grained sandbox" complaint)**
|
||||
|
||||
The Feb 2026 HN thread that argued "sandboxes are too coarse-grained"
|
||||
was pointing at a real gap: a VM isolates the filesystem but doesn't
|
||||
know whether an agent *should* be sending email or calling an external
|
||||
API. bot-bottle's bottle/agent split is a structural answer at the
|
||||
network layer — the bottle manifest declares exactly what each role can
|
||||
reach (which hosts, which paths, which HTTP methods), and the egress
|
||||
scanner enforces it. A `gitea-dev` bottle that only lists
|
||||
`gitea.dideric.is` and `api.anthropic.com` structurally cannot send
|
||||
email or reach AWS, not because the model was told not to, but because
|
||||
those routes don't exist.
|
||||
|
||||
The `extends:` composition model means provider-level policy (the Claude
|
||||
auth route) lives in one base bottle and role-specific overlays are
|
||||
stacked on top — no duplication, and changing the base propagates to all
|
||||
derived roles.
|
||||
|
||||
Competitive position on this axis (from `agent-sandbox-landscape.md`):
|
||||
|
||||
| Tool | Agent-tailored policy |
|
||||
|---|---|
|
||||
| **bot-bottle** | Yes — declarative per-role manifest; `extends:` composition; egress + credentials scoped to role |
|
||||
| **tilde.run** | Yes — per-agent DSL RBAC (allow/deny/approve per action/repo/agent), but hosted SaaS |
|
||||
| **Microsoft AGT** | Yes — YAML policy + per-agent DID + trust score, but tool-call level only (no network isolation) |
|
||||
| **OAP** | Yes — declarative pre-action policy + cryptographic audit, but no isolation |
|
||||
| **Cleanroom** | Partial — per-repo `cleanroom.yaml`, not per-role |
|
||||
| **Docker sbx** | No — network presets only |
|
||||
| **Anthropic srt** | No — programmatic per-invocation |
|
||||
| **matchlock / smolmachines / microsandbox** | No |
|
||||
| **agent-safehouse** | Partial — per-agent Seatbelt profiles; no egress |
|
||||
|
||||
Two takeaways: bot-bottle and tilde.run are the only isolation tools
|
||||
with declarative role-tailored policy; Microsoft AGT and OAP are the
|
||||
closest competitors on role-tailoring but operate at the tool-call layer
|
||||
without network/filesystem isolation — complementary, not substitutes.
|
||||
|
||||
**Outbound exfiltration (any injection class)**
|
||||
|
||||
Whatever triggers the agent — README injection, Agentjacking, MCP
|
||||
description poisoning — the final step in most attacks is exfiltration.
|
||||
bot-bottle's egress allowlist is default-deny with a per-bottle host
|
||||
allowlist; unknown hosts get a hard 403. Outbound DLP scanning
|
||||
(`outbound_detectors: [token_patterns, known_secrets]`) catches tokens
|
||||
and secrets in outbound bodies; the `supervise` policy (default for
|
||||
manifest routes) holds the request for operator approval rather than
|
||||
silently blocking it. Together these limit what a successful injection
|
||||
can *do* even if it succeeds at the model layer.
|
||||
|
||||
**LiteLLM / compromised-proxy attacks**
|
||||
|
||||
bot-bottle does not use LiteLLM. The model API route (e.g.
|
||||
`api.anthropic.com`) is an auto-injected provider route on the egress
|
||||
allowlist; the agent dials the gateway, not the model API directly.
|
||||
A compromised third-party proxy is not in the architecture.
|
||||
|
||||
### Where it is weaker
|
||||
|
||||
**Prompt injection**
|
||||
|
||||
Egress controls and prompt injection defenses are orthogonal. Egress
|
||||
limits what the agent can *send out* (outbound leg); prompt injection
|
||||
is about what attacker-controlled data *tells the agent to do* (inbound
|
||||
leg). The two don't substitute for each other and must be treated
|
||||
separately.
|
||||
|
||||
The inbound DLP scanner (`inbound_detectors: [naive_injection_detection]`)
|
||||
is the only runtime defense against injection arriving through allowlisted
|
||||
channels — Sentry MCP responses, MCP tool descriptions, README content.
|
||||
It is explicitly pattern-matching and will not catch a sufficiently
|
||||
crafted payload. There is no semantic / intent-level gate between what
|
||||
the model decides and what the agent executes.
|
||||
|
||||
**Blast radius within the permitted scope**
|
||||
|
||||
Inside a tightly-egressed sandbox a successful injection can't
|
||||
exfiltrate to unknown hosts, but it still has real options:
|
||||
|
||||
- *Work product corruption.* The agent can modify, delete, or backdoor
|
||||
files in the working directory. This is within its permitted scope;
|
||||
egress controls have nothing to say about it.
|
||||
|
||||
- *Malicious commits past the git-gate.* The git-gate scans outbound
|
||||
refs for secrets (gitleaks), not for semantic code intent. A prompt-
|
||||
injected agent can commit subtly malicious code — logic bombs,
|
||||
backdoored auth paths, code that exfiltrates data through the
|
||||
application's own HTTP clients at runtime — that looks clean to a
|
||||
secret scanner.
|
||||
|
||||
- *Exfiltration through allowlisted channels.* If an attacker knows or
|
||||
can predict what hosts are in the egress allowlist, those channels are
|
||||
available for exfiltration. A GitHub remote being allowlisted means
|
||||
"push to an attacker-controlled fork" is viable. A logging endpoint
|
||||
being allowlisted means structured data can leave through it. The
|
||||
outbound DLP scanner catches credential tokens and known secrets but
|
||||
not arbitrary business data.
|
||||
|
||||
- *Dependency installation within the sandbox.* An agent that runs
|
||||
`npm install` or `pip install` on attacker-specified packages executes
|
||||
code inside the sandbox with the same capabilities the agent has:
|
||||
filesystem access, tool calls, calls to allowlisted hosts. Supply chain
|
||||
injection via package names is in the same injection family, triggered
|
||||
by the same prompt-injection path.
|
||||
|
||||
### What would close the remaining gaps
|
||||
|
||||
The blast-radius risks above point at two distinct mitigations that
|
||||
don't yet exist in bot-bottle:
|
||||
|
||||
- *Outbound intent classification.* The egress addon today scans
|
||||
outbound request content for token patterns. What it lacks is
|
||||
awareness of context — it can't distinguish "agent is pushing a
|
||||
legitimate commit" from "agent was injected and is pushing a backdoor."
|
||||
The `supervise` policy is already the right shape for human-in-the-loop
|
||||
review on sensitive outbound actions; extending it with context from
|
||||
the agent's recent tool calls (what files were touched, what was the
|
||||
triggering task) would narrow the gap.
|
||||
|
||||
- *Semantic code review on git push.* gitleaks is the wrong tool for
|
||||
catching injected logic. A review step on outbound commits — even a
|
||||
simple diff summary surfaced in `cli.py supervise` before the push is
|
||||
forwarded — would close the malicious-commit path without requiring
|
||||
the agent to be fully trusted.
|
||||
|
||||
## Sources
|
||||
|
||||
- [Ask HN: The new wave of AI agent sandboxes? (Mar 2026)](https://news.ycombinator.com/item?id=47444917)
|
||||
- [OK, let's survey how everybody is sandboxing AI coding agents (Mar 2026)](https://news.ycombinator.com/item?id=47185250)
|
||||
- [The agent harness belongs outside the sandbox (May 2026)](https://news.ycombinator.com/item?id=47990675)
|
||||
- [Show HN: Nono – Kernel-enforced sandboxing for AI agents (Feb 2026)](https://news.ycombinator.com/item?id=46849615)
|
||||
- [Kernel-enforced sandbox for AI agents, MCP and LLM workloads (Feb 2026)](https://news.ycombinator.com/item?id=47066574)
|
||||
- [Sandboxes will be left in 2026 (Feb 2026)](https://news.ycombinator.com/item?id=47006445)
|
||||
- [Critical Cursor Flaws / DuneSlide – The Hacker News](https://thehackernews.com/2026/07/critical-cursor-flaws-could-let-prompt.html)
|
||||
- [Agentjacking Attack – The Hacker News](https://thehackernews.com/2026/06/agentjacking-attack-tricks-ai-coding.html)
|
||||
- [Friendly Fire: AI Agents Built to Catch Malicious Code – The Hacker News](https://thehackernews.com/2026/07/friendly-fire-ai-agents-built-to-catch.html)
|
||||
- [Microsoft Warns Poisoned MCP Tool Descriptions – The Hacker News](https://thehackernews.com/2026/06/microsoft-warns-poisoned-mcp-tool.html)
|
||||
- [MCP STDIO Command Injection Advisory – OX Security](https://www.ox.security/blog/mcp-supply-chain-advisory-rce-vulnerabilities-across-the-ai-ecosystem/)
|
||||
- [LiteLLM Vulnerability Chain – The Hacker News](https://thehackernews.com/2026/06/litellm-vulnerability-chain-lets-low.html)
|
||||
- [Red-Teaming the Agentic Red-Team (arXiv 2606.24496)](https://arxiv.org/pdf/2606.24496)
|
||||
@@ -1,182 +0,0 @@
|
||||
# Landscape: containerized AI coding agent tools
|
||||
|
||||
Research into whether bot-bottle is redundant with existing projects, and
|
||||
whether it's worth publishing.
|
||||
|
||||
## Summary
|
||||
|
||||
The "AI coding agents in isolated sandboxes" space is active but not saturated.
|
||||
bot-bottle occupies a distinct position: no surveyed project combines all five
|
||||
of its defining features. Publishing is likely worthwhile, with the main risk
|
||||
being claudebox expanding to absorb the same niche.
|
||||
|
||||
**Updated 2026-07-09:** bot-bottle now supports three isolation backends
|
||||
(Docker, Apple `container`, smolmachines/libkrun microVMs) and three built-in
|
||||
agent providers (Claude Code, OpenAI Codex, Pi) with an open plugin system for
|
||||
arbitrary providers. This meaningfully strengthens the differentiation against
|
||||
all surveyed competitors.
|
||||
|
||||
## Closest competitor: claudebox
|
||||
|
||||
[RchGrav/claudebox](https://github.com/RchGrav/claudebox) is the most
|
||||
feature-complete analog. It runs Claude Code in Docker with per-project
|
||||
isolated images, 15+ pre-configured dev-language profiles, and per-project
|
||||
network firewall allowlists. Actively maintained with multiple forks.
|
||||
|
||||
What it lacks: manifest-driven named agents, per-agent env resolution modes
|
||||
(prompt / host-forward / literal), skill directory injection, per-agent system
|
||||
prompts, SSH-agent forwarding without copying private keys, home+project
|
||||
manifest merge.
|
||||
|
||||
## Other surveyed projects
|
||||
|
||||
- **textcortex/claude-code-sandbox → spritz** — evolved toward
|
||||
Kubernetes-native multi-agent infra; not stdlib-first or local-Docker.
|
||||
Original sandbox repo is archived.
|
||||
- **trailofbits/claude-code-devcontainer** — devcontainer config for security
|
||||
audits; not a general agent launcher.
|
||||
- **Several small solo repos** (arezi/claude-sandbox, nkrefman/claude-sandbox,
|
||||
VishalJ99/claude-docker) — lightweight Docker wrappers with no multi-agent
|
||||
config layer.
|
||||
- **Docker's official sandbox templates** — launch-and-run Dockerfiles plus an
|
||||
npm-based runtime; not a manifest-driven fleet manager.
|
||||
|
||||
## Adjacent (different model)
|
||||
|
||||
- **dagger/container-use** (mid-2025) — exposes an MCP server so the *agent*
|
||||
spins up its own containers with Git worktrees. Inverted model vs. bot-bottle
|
||||
(agent controls container rather than being launched into one by a manifest).
|
||||
Still marked early-development.
|
||||
- **E2B, Northflank, Cloudflare Sandbox SDK** — cloud-hosted SaaS sandbox
|
||||
runtimes; fundamentally different architecture.
|
||||
- **superhq.ai / SuperHQ** (v0.4.4, April 2026) — macOS desktop app (Rust/GPUI)
|
||||
that runs Claude Code, Codex, and Pi inside microVMs via Apple's
|
||||
Virtualization.framework (their own shuru-sdk / libkrun). Auth gateway
|
||||
injects API keys on the wire so the sandbox never sees them; tmpfs overlay
|
||||
stages agent writes for diff-and-accept review; mobile remote access via
|
||||
remote.superhq.ai. Early alpha, free on launch, Apple Silicon only.
|
||||
|
||||
Overlap: both projects cover agent isolation, credential proxying, and
|
||||
multi-provider support (Claude Code / Codex / Pi). Differences: SuperHQ is a
|
||||
GUI desktop app with no manifest layer; bot-bottle is a CLI fleet manager with
|
||||
named agents, skills injection, per-agent system prompts, and cross-platform
|
||||
backends (Docker, Apple `container`, smolmachines). SuperHQ's microVM
|
||||
isolation story is now partially matched by bot-bottle's `macos_container` and
|
||||
smolmachines backends. Worth watching — it targets the same security-minded
|
||||
power-user audience and moves fast.
|
||||
|
||||
**Known gap in SuperHQ (user-requested, as of 2026-07-09):** A named user
|
||||
(Brian Cheong, Founder, Dunialabs.io) explicitly called out the absence of
|
||||
per-run audit logging: tool calls and network egress. Bot-bottle covers both:
|
||||
network egress is logged by pipelock/mitmproxy, and per-run op-log/audit state
|
||||
is persisted to SQLite.
|
||||
|
||||
- **OneCLI** ([onecli.sh](https://onecli.sh/)) — YC-backed, GA, open-source
|
||||
(Apache-2.0, Rust) "identity gateway for AI agents": a credential/secret
|
||||
broker that holds API keys and OAuth tokens out of the agent's reach and
|
||||
injects them at the network layer (phantom-token — the agent sees a
|
||||
placeholder, the gateway swaps in the real, AES-256-GCM-encrypted credential
|
||||
at request time). Framework-agnostic and drop-in for any HTTP-calling agent,
|
||||
50+ app integrations, plus a hosted cloud tier with a per-agent dashboard and
|
||||
audit logs. Full technical breakdown in
|
||||
[`agent-credential-proxy-landscape.md`](agent-credential-proxy-landscape.md).
|
||||
|
||||
**How close a competitor:** near-exact on the *single axis of agent secret
|
||||
custody* — the exact thing bot-bottle sells as "the agent never sees real
|
||||
credentials, even via `printenv`." OneCLI does that one job well, is mature
|
||||
and funded, and is *more portable* (it sits in front of anything; bot-bottle
|
||||
only helps agents launched through bot-bottle). Takeaway: bot-bottle should
|
||||
stop treating secret custody as a *unique* differentiator. But OneCLI is
|
||||
**not** a competitor to bot-bottle's actual product — it does no agent
|
||||
sandboxing (containers/microVMs), no fleet/manifest layer, no named agents /
|
||||
skills / per-agent system prompts, no multi-provider launching, no egress
|
||||
firewall.
|
||||
|
||||
**Our edge:** (1) *Isolation is the product, not a proxy.* OneCLI keeps the
|
||||
key out of reach at the network layer, but the agent itself still runs
|
||||
unsandboxed — a hijacked agent behind OneCLI has full run of its host and can
|
||||
exfil captured data through any allowed host. bot-bottle runs the agent inside
|
||||
a kernel/VM-enforced sandbox, injects credentials across that same
|
||||
out-of-process boundary, *and* clamps egress with pipelock — defense in depth
|
||||
vs. a single network layer. (2) *Fleet + manifest model* with named agents,
|
||||
skills, per-agent system prompts, multi-provider and multi-backend — OneCLI
|
||||
has no equivalent. (3) *Trust posture:* OneCLI's managed tier reintroduces a
|
||||
third-party credential custodian, whereas bot-bottle's OSS-runtime +
|
||||
paid-control-plane split keeps custody inside the operator's own boundary —
|
||||
the stronger story for the security-minded self-hoster. (4) *Runs inside your
|
||||
network boundary — local/internal reach.* Because bot-bottle executes the
|
||||
agent on your own host (homelab, corporate LAN, a Tailnet) and egress is a
|
||||
manifest field, giving an agent *scoped* access to **internal** resources — a
|
||||
private Gitea, a LAN database, a Tailscale node — is just another egress-route
|
||||
line, not a networking project (the same move an operator already makes to
|
||||
reach their Tailscale services). OneCLI's OSS core can self-host too, but it's
|
||||
a credential *broker* for outbound API calls, not an agent runtime, and its
|
||||
managed tier + 50+ integrations are oriented at public SaaS — it doesn't put
|
||||
the agent behind your firewall for you. This is a reach advantage, distinct
|
||||
from the isolation ones above, and it's a wedge cloud-first agent products
|
||||
(Devin, Copilot Workspace, OneCLI Cloud) structurally can't match. **Tactical
|
||||
read:**
|
||||
adopt OneCLI's OSS core for the credential slice if building is undesirable
|
||||
(it's mature now); don't build atop its managed tier (competitor, not
|
||||
dependency); re-position bot-bottle on isolation + fleet + self-hosted custody
|
||||
rather than "we hide your secrets."
|
||||
|
||||
## What no found project does
|
||||
|
||||
None combine:
|
||||
1. Named-agent manifest with per-agent env resolution (prompt / host-forward / literal), supporting multiple providers (Claude Code, Codex, Pi, arbitrary plugins)
|
||||
2. Skills directory injection
|
||||
3. Per-agent system prompts
|
||||
4. SSH-agent key forwarding without copying private keys into the container
|
||||
5. Home + project manifest merge
|
||||
6. Pluggable isolation backends: Docker (Linux/macOS), Apple `container` (macOS microVMs), smolmachines/libkrun microVMs
|
||||
7. Per-run audit log: network egress via pipelock/mitmproxy + op-log persisted to SQLite
|
||||
|
||||
**In-flight directions (not yet shipped):**
|
||||
|
||||
- **Forge-native dispatch (issue #317):** Gitea webhook → orchestrator spins up a bottle
|
||||
with the issue body as prompt → agent works → bottle freezes awaiting review comment →
|
||||
rehydrates on comment → tears down on PR close. The issue-to-PR lifecycle concept is not
|
||||
novel (Devin, Copilot Workspace, SWE-agent all do this as cloud services); what's
|
||||
distinct is doing it self-hosted, manifest-driven, inside bot-bottle's isolation
|
||||
primitives.
|
||||
- **Paid web control plane (issue #327):** Browser-based multi-host agent launch and
|
||||
monitoring; account-scoped bottle and agent definitions; secret custody (encrypted at
|
||||
rest, injected into the sidecar at launch, never exposed to the agent or returned by any
|
||||
read API). Monetization model: OSS runtime free, control plane paid — a standard split
|
||||
(HashiCorp, Grafana) applied to a self-hosted agent sandbox. The principled secret
|
||||
custody model (agent never sees real credentials, even via printenv) is more rigorous
|
||||
than most surveyed tools but not unprecedented.
|
||||
|
||||
## Publishing verdict
|
||||
|
||||
Worth publishing. Differentiators that matter to the target audience (power
|
||||
users running parallel AI coding agent sessions with distinct personas/tooling):
|
||||
|
||||
- The Python-stdlib-first, low-dependency design — competitors are npm-based,
|
||||
Rust/GUI, or Kubernetes-native.
|
||||
- Named agents with distinct skills and system prompts, not just language profiles.
|
||||
- Multi-backend isolation: Docker, Apple `container` microVMs, and
|
||||
smolmachines/libkrun — single manifest works across all three.
|
||||
- Multi-provider: Claude Code, Codex, Pi, plus an open plugin system for
|
||||
arbitrary providers.
|
||||
- SSH forwarding without key copying.
|
||||
- Per-run audit log (tool calls + network egress) — an explicitly requested gap
|
||||
in SuperHQ as of 2026-07-09.
|
||||
- Forge-native dispatch and a paid control plane (in flight) bring bot-bottle
|
||||
into the same product category as cloud services like Devin and Copilot
|
||||
Workspace — but self-hosted, with stronger isolation guarantees and a
|
||||
manifest-driven fleet model those services don't have.
|
||||
|
||||
Main risk: claudebox adds manifest/agent config; SuperHQ is moving fast on the
|
||||
GUI / microVM side. The space is moving fast enough that publishing sooner is
|
||||
better if establishing prior art matters.
|
||||
|
||||
Discovery will be slow without active promotion; an Anthropic Discord post or
|
||||
HN "Show HN" would do most of the work.
|
||||
|
||||
## Caveats
|
||||
|
||||
- GitHub search cannot surface private or very new repos comprehensively.
|
||||
- Counts (stars, forks) were not confirmed for every project.
|
||||
- Initial research conducted 2026-05-07; SuperHQ entry added 2026-07-09; the space moves fast.
|
||||
@@ -1,6 +1,6 @@
|
||||
# smolmachines as a VM backend for bot-bottle
|
||||
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/landscape-containerized-claude.md`.
|
||||
> **Superseded (2026-07-11).** The smolmachines backend was removed — Linux now uses the Firecracker backend, macOS uses macos-container. Kept as a historical record; see the removal commit `c07ebca` and `docs/research/agent-sandbox-landscape.md`.
|
||||
|
||||
Evaluation of whether [smolmachines](https://smolmachines.com/) would
|
||||
simplify the macOS agent-VM-isolation work spelled out in
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[build-system]
|
||||
requires = ["setuptools>=68"]
|
||||
build-backend = "setuptools.backends.legacy:build"
|
||||
build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "bot-bottle"
|
||||
|
||||
+4
-2
@@ -26,8 +26,10 @@ rm -f .coverage
|
||||
echo "== unit ==" >&2
|
||||
"$PY" -m coverage run -m unittest discover -t . -s tests/unit
|
||||
|
||||
echo "== integration (skips without Docker) ==" >&2
|
||||
"$PY" -m coverage run --append -m unittest discover -t . -s tests/integration
|
||||
echo "== integration (firecracker; skips docker tests) ==" >&2
|
||||
BOT_BOTTLE_BACKEND=firecracker SKIP_DOCKER_TESTS=1 \
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_DIR="${BOT_BOTTLE_CI_INFRA_ARTIFACT_DIR:-}" \
|
||||
"$PY" -m coverage run --append -m unittest discover -t . -s tests/integration
|
||||
|
||||
echo "== combined report ==" >&2
|
||||
"$PY" -m coverage report -m
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Enforce the repository's issue/PR metadata policy in Gitea Actions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
ISSUE_REFERENCE = re.compile(
|
||||
r"(?im)\b(?:close[sd]?|fix(?:e[sd])?|resolve[sd]?|part\s+of|"
|
||||
r"related\s+to|refs?|references)\s+#(\d+)\b"
|
||||
)
|
||||
TRIAGE_LABEL = "Status/Needs Triage"
|
||||
|
||||
|
||||
def deliberate_issue_numbers(title: str, body: str) -> set[int]:
|
||||
"""Return same-repository issue numbers referenced intentionally."""
|
||||
return {int(match) for match in ISSUE_REFERENCE.findall(f"{title}\n{body}")}
|
||||
|
||||
|
||||
class GiteaApi:
|
||||
"""Small API client using the Actions-provided repository token."""
|
||||
|
||||
def __init__(self, api_url: str, repository: str, token: str) -> None:
|
||||
self.base = f"{api_url.rstrip('/')}/repos/{repository}"
|
||||
self.token = token
|
||||
|
||||
def request(self, method: str, path: str, payload: object | None = None) -> Any:
|
||||
data = None if payload is None else json.dumps(payload).encode()
|
||||
request = urllib.request.Request(
|
||||
f"{self.base}{path}",
|
||||
data=data,
|
||||
method=method,
|
||||
headers={
|
||||
"Authorization": f"token {self.token}",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=15) as response:
|
||||
if response.status == 204:
|
||||
return None
|
||||
return json.load(response)
|
||||
|
||||
|
||||
def check_pull_request(event: dict[str, Any], api: GiteaApi) -> list[str]:
|
||||
"""Return policy violations for a pull_request event."""
|
||||
pull = event["pull_request"]
|
||||
errors: list[str] = []
|
||||
labels = pull.get("labels") or []
|
||||
if labels:
|
||||
errors.append(
|
||||
"PRs must be unlabeled; put tracker metadata on the linked issue "
|
||||
f"(found: {', '.join(label['name'] for label in labels)})."
|
||||
)
|
||||
|
||||
numbers = deliberate_issue_numbers(pull.get("title", ""), pull.get("body", ""))
|
||||
if not numbers:
|
||||
errors.append(
|
||||
"PR must reference an issue with Closes/Fixes/Resolves #N, "
|
||||
"Part of #N, Related to #N, Refs #N, or References #N."
|
||||
)
|
||||
return errors
|
||||
|
||||
real_issues = 0
|
||||
for number in sorted(numbers):
|
||||
try:
|
||||
item = api.request("GET", f"/issues/{number}")
|
||||
except urllib.error.HTTPError as error:
|
||||
if error.code == 404:
|
||||
errors.append(f"Referenced issue #{number} does not exist.")
|
||||
continue
|
||||
raise
|
||||
if item.get("pull_request") is not None:
|
||||
errors.append(f"#{number} is a pull request, not an issue.")
|
||||
else:
|
||||
real_issues += 1
|
||||
|
||||
if not real_issues and not errors:
|
||||
errors.append("PR must reference at least one real issue.")
|
||||
return errors
|
||||
|
||||
|
||||
def ensure_issue_label(event: dict[str, Any], api: GiteaApi) -> bool:
|
||||
"""Apply the triage label if an issue event leaves the issue unlabeled."""
|
||||
issue = event["issue"]
|
||||
if issue.get("pull_request") is not None or issue.get("labels"):
|
||||
return False
|
||||
labels = api.request("GET", "/labels?limit=100")
|
||||
triage = next((label for label in labels if label["name"] == TRIAGE_LABEL), None)
|
||||
if triage is None:
|
||||
raise RuntimeError(f"repository label {TRIAGE_LABEL!r} does not exist")
|
||||
api.request("POST", f"/issues/{issue['number']}/labels", {"labels": [triage["id"]]})
|
||||
return True
|
||||
|
||||
|
||||
def _load_event(path: str) -> dict[str, Any]:
|
||||
return json.loads(Path(path).read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("command", choices=("check-pr", "label-issue"))
|
||||
parser.add_argument("--event", default=os.environ.get("GITHUB_EVENT_PATH"))
|
||||
args = parser.parse_args()
|
||||
if not args.event:
|
||||
parser.error("--event or GITHUB_EVENT_PATH is required")
|
||||
|
||||
api = GiteaApi(
|
||||
os.environ["GITHUB_API_URL"],
|
||||
os.environ["GITHUB_REPOSITORY"],
|
||||
os.environ["GITHUB_TOKEN"],
|
||||
)
|
||||
event = _load_event(args.event)
|
||||
if args.command == "check-pr":
|
||||
errors = check_pull_request(event, api)
|
||||
if errors:
|
||||
print("\n".join(f"::error::{error}" for error in errors))
|
||||
return 1
|
||||
print("PR tracker policy passed.")
|
||||
return 0
|
||||
|
||||
changed = ensure_issue_label(event, api)
|
||||
print(f"Applied {TRIAGE_LABEL}." if changed else "Issue already has a label.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Resolved paths to system binaries used by subprocess-based tests.
|
||||
|
||||
NixOS and other non-FHS hosts don't populate ``/bin`` (there is no
|
||||
``/bin/sleep``), so tests that spawn real short-lived helper processes
|
||||
must resolve the binary from ``PATH`` rather than hardcoding an FHS path.
|
||||
Import the resolved constant (e.g. ``SLEEP``) instead of writing
|
||||
``/bin/sleep`` inline.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import shutil
|
||||
|
||||
|
||||
def resolve(name: str, fallback: str) -> str:
|
||||
"""Absolute path to ``name`` from ``PATH``; ``fallback`` on FHS hosts
|
||||
where the binary isn't on ``PATH`` but lives at a known ``/bin`` path."""
|
||||
return shutil.which(name) or fallback
|
||||
|
||||
|
||||
# Real ``sleep`` binary. ``/bin/sleep`` is absent on NixOS; resolve from
|
||||
# PATH so subprocess tests run instead of erroring with FileNotFoundError.
|
||||
SLEEP = resolve("sleep", "/bin/sleep")
|
||||
+29
-9
@@ -2,24 +2,44 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import unittest
|
||||
|
||||
|
||||
def docker_available() -> bool:
|
||||
if os.environ.get("SKIP_DOCKER_TESTS"):
|
||||
return False
|
||||
if shutil.which("docker") is None:
|
||||
return False
|
||||
return (
|
||||
subprocess.run(
|
||||
["docker", "info"],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
check=False,
|
||||
).returncode
|
||||
== 0
|
||||
)
|
||||
try:
|
||||
return (
|
||||
subprocess.run(
|
||||
["docker", "info"],
|
||||
stdout=subprocess.DEVNULL,
|
||||
stderr=subprocess.DEVNULL,
|
||||
check=False,
|
||||
timeout=5,
|
||||
).returncode
|
||||
== 0
|
||||
)
|
||||
except subprocess.TimeoutExpired:
|
||||
return False
|
||||
|
||||
|
||||
def skip_unless_docker(reason: str = "docker unreachable"):
|
||||
return unittest.skipUnless(docker_available(), reason)
|
||||
|
||||
|
||||
def skip_unless_docker_or_firecracker(
|
||||
reason: str = "neither Docker nor Firecracker selected",
|
||||
):
|
||||
"""Skip a backend-agnostic test unless one supported backend can run.
|
||||
|
||||
Firecracker does not require the host Docker daemon. The KVM coverage job
|
||||
deliberately sets ``SKIP_DOCKER_TESTS`` to exclude Docker-only integration
|
||||
classes while still exercising this path.
|
||||
"""
|
||||
firecracker_selected = os.environ.get("BOT_BOTTLE_BACKEND") == "firecracker"
|
||||
return unittest.skipUnless(firecracker_selected or docker_available(), reason)
|
||||
|
||||
@@ -86,13 +86,12 @@ class TestGatewayImage(unittest.TestCase):
|
||||
self.assertIn("Mitmproxy", out)
|
||||
|
||||
def test_python_imports_supervise_module(self):
|
||||
# The bundle's supervise daemon imports `supervise` as a
|
||||
# same-directory sibling of `supervise_server`. Probe the
|
||||
# import resolves with `python3 -c` from /app (the
|
||||
# Dockerfile's WORKDIR).
|
||||
# The supervise daemon is installed as part of the bot_bottle
|
||||
# package (5ad3449), not as flat sibling modules under /app.
|
||||
# Probe that the package imports resolve inside the image.
|
||||
rc, out = self._run_in_image(
|
||||
"python3", "-c",
|
||||
"import supervise; import supervise_server; print('ok')",
|
||||
"from bot_bottle import supervise, supervise_server; print('ok')",
|
||||
)
|
||||
self.assertEqual(0, rc, msg=out)
|
||||
self.assertIn("ok", out)
|
||||
|
||||
@@ -31,7 +31,7 @@ from pathlib import Path
|
||||
from bot_bottle.backend import BottleSpec, get_bottle_backend
|
||||
from bot_bottle.bottle_state import cleanup_state
|
||||
from bot_bottle.manifest import ManifestIndex
|
||||
from tests._docker import skip_unless_docker
|
||||
from tests._docker import skip_unless_docker_or_firecracker
|
||||
|
||||
|
||||
# Secrets planted in the bottle env as literals (agents substitute via
|
||||
@@ -67,7 +67,7 @@ _DUMMY_HOST_KEY = (
|
||||
)
|
||||
|
||||
|
||||
@skip_unless_docker()
|
||||
@skip_unless_docker_or_firecracker()
|
||||
@unittest.skipIf(
|
||||
os.environ.get("GITEA_ACTIONS") == "true"
|
||||
and os.environ.get("BOT_BOTTLE_BACKEND") != "firecracker",
|
||||
@@ -91,11 +91,9 @@ class TestSandboxEscape(unittest.TestCase):
|
||||
|
||||
@classmethod
|
||||
def setUpClass(cls) -> None:
|
||||
# Docker is always required (the agent + companion containers run under it,
|
||||
# and VM backends still use it for the gateway); the
|
||||
# class-level @skip_unless_docker already covers that. Pin
|
||||
# Docker when BOT_BOTTLE_BACKEND is unset to preserve the
|
||||
# Docker-backed CI path.
|
||||
# Pin Docker when BOT_BOTTLE_BACKEND is unset to preserve the
|
||||
# Docker-backed CI path. Firecracker uses its persistent infra VM for
|
||||
# the shared gateway and therefore does not require host Docker.
|
||||
cls._backend_name = os.environ.get("BOT_BOTTLE_BACKEND", "docker")
|
||||
|
||||
# Throwaway static key for the git-gate fixture. It need not
|
||||
|
||||
@@ -215,6 +215,18 @@ class TestDockerSetupStatus(unittest.TestCase):
|
||||
with patch.object(dk.shutil, "which", return_value=None):
|
||||
self.assertFalse(dk._daemon_reachable())
|
||||
|
||||
def test_daemon_reachable_true_when_daemon_responds(self):
|
||||
with patch.object(dk.shutil, "which", return_value="/usr/bin/docker"), \
|
||||
patch.object(dk.subprocess, "run",
|
||||
return_value=subprocess.CompletedProcess([], 0)):
|
||||
self.assertTrue(dk._daemon_reachable())
|
||||
|
||||
def test_daemon_reachable_false_on_timeout(self):
|
||||
with patch.object(dk.shutil, "which", return_value="/usr/bin/docker"), \
|
||||
patch.object(dk.subprocess, "run",
|
||||
side_effect=subprocess.TimeoutExpired(["docker", "info"], 5)):
|
||||
self.assertFalse(dk._daemon_reachable())
|
||||
|
||||
def test_status_reports_missing_docker(self):
|
||||
with patch.object(dk.shutil, "which", return_value=None):
|
||||
rc, out = _cap(dk.status)
|
||||
|
||||
@@ -95,5 +95,60 @@ class TestMainDispatch(unittest.TestCase):
|
||||
self.assertEqual(130, main(["x"]))
|
||||
|
||||
|
||||
class TestMigrationGate(unittest.TestCase):
|
||||
"""The dispatcher's schema-migration gate (cli/__init__.py)."""
|
||||
|
||||
def setUp(self) -> None:
|
||||
# Force the "schema out of date" branch for every test here.
|
||||
patcher = patch.object(StoreManager, "is_migrated", return_value=False)
|
||||
patcher.start()
|
||||
self.addCleanup(patcher.stop)
|
||||
self.addCleanup(StoreManager.reset)
|
||||
|
||||
def test_store_command_blocks_when_stdin_cannot_confirm(self) -> None:
|
||||
# Non-TTY stdin at EOF (as in CI): the [y/N] prompt reads "" and the
|
||||
# command is refused rather than migrating silently.
|
||||
ran: list[bool] = []
|
||||
|
||||
def handler(_rest: list[str]) -> int:
|
||||
ran.append(True)
|
||||
return 0
|
||||
|
||||
with patch.dict(climod.COMMANDS, {"list": handler}), \
|
||||
patch("sys.stdin", io.StringIO("")), \
|
||||
patch("sys.stderr", io.StringIO()):
|
||||
self.assertEqual(1, main(["list"]))
|
||||
self.assertEqual([], ran, "gated command must not dispatch")
|
||||
|
||||
def test_backend_command_skips_gate(self) -> None:
|
||||
# `backend` provisions/probes the host and never opens the store, so
|
||||
# it must run even on an unmigrated DB with unanswerable stdin.
|
||||
ran: list[bool] = []
|
||||
|
||||
def handler(_rest: list[str]) -> int:
|
||||
ran.append(True)
|
||||
return 0
|
||||
|
||||
with patch.dict(climod.COMMANDS, {"backend": handler}), \
|
||||
patch("sys.stdin", io.StringIO("")), \
|
||||
patch("sys.stderr", io.StringIO()):
|
||||
self.assertEqual(0, main(["backend", "status"]))
|
||||
self.assertEqual([True], ran, "exempt command must dispatch")
|
||||
|
||||
def test_store_command_migrates_on_confirmation(self) -> None:
|
||||
migrated: list[bool] = []
|
||||
|
||||
def handler(_rest: list[str]) -> int:
|
||||
return 0
|
||||
|
||||
with patch.dict(climod.COMMANDS, {"list": handler}), \
|
||||
patch.object(StoreManager, "migrate",
|
||||
side_effect=lambda: migrated.append(True)), \
|
||||
patch("sys.stdin", io.StringIO("y\n")), \
|
||||
patch("sys.stderr", io.StringIO()):
|
||||
self.assertEqual(0, main(["list"]))
|
||||
self.assertEqual([True], migrated, "confirmed gate must migrate")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -57,6 +57,14 @@ class TestCmdStartSelector(unittest.TestCase):
|
||||
self._bottle_picker_mock = self._bottle_picker_patch.start()
|
||||
self._bottle_picker_mock.return_value = ["claude"] # default: one bottle selected
|
||||
|
||||
# name_color_modal opens /dev/tty and blocks on keyboard input on
|
||||
# self-hosted runners that have a real controlling terminal. Stub it
|
||||
# out like the other tui pickers so tests don't wait for a keypress.
|
||||
self._modal_patch = patch.object(
|
||||
tui_mod, "name_color_modal", return_value=("researcher", ""),
|
||||
)
|
||||
self._modal_patch.start()
|
||||
|
||||
self._env_patch = patch.dict(os.environ, {}, clear=False)
|
||||
self._env_patch.start()
|
||||
os.environ.pop("BOT_BOTTLE_BACKEND", None)
|
||||
@@ -66,6 +74,7 @@ class TestCmdStartSelector(unittest.TestCase):
|
||||
self._launch_patch.stop()
|
||||
self._agent_picker_patch.stop()
|
||||
self._bottle_picker_patch.stop()
|
||||
self._modal_patch.stop()
|
||||
self._env_patch.stop()
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Unit: DbStore._connection() context manager and is_migrated()."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlite3
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from bot_bottle.db_store import DbStore
|
||||
from bot_bottle.migrations import TableMigrations
|
||||
|
||||
|
||||
def _store(tmp: Path) -> DbStore:
|
||||
migrations = TableMigrations("test", ["CREATE TABLE items (id INTEGER PRIMARY KEY)"])
|
||||
return DbStore(tmp / "test.db", migrations)
|
||||
|
||||
|
||||
class TestDbStoreIsMigrated(unittest.TestCase):
|
||||
def test_returns_false_when_db_absent(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
store = _store(Path(d))
|
||||
self.assertFalse(store.is_migrated())
|
||||
|
||||
def test_returns_false_when_schema_versions_missing(self):
|
||||
# DB file exists but has no schema_versions table → OperationalError → False.
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
store = _store(Path(d))
|
||||
conn = sqlite3.connect(store.db_path)
|
||||
conn.close()
|
||||
self.assertFalse(store.is_migrated())
|
||||
|
||||
def test_returns_true_after_migrate(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
store = _store(Path(d))
|
||||
store.migrate()
|
||||
self.assertTrue(store.is_migrated())
|
||||
|
||||
def test_returns_false_when_behind(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
migrations = TableMigrations(
|
||||
"test",
|
||||
[
|
||||
"CREATE TABLE items (id INTEGER PRIMARY KEY)",
|
||||
"ALTER TABLE items ADD COLUMN name TEXT",
|
||||
],
|
||||
)
|
||||
store = DbStore(Path(d) / "test.db", migrations)
|
||||
# Apply only the first migration manually.
|
||||
conn = sqlite3.connect(store.db_path)
|
||||
with conn:
|
||||
TableMigrations("test", [migrations.migrations[0]]).apply(conn)
|
||||
conn.close()
|
||||
self.assertFalse(store.is_migrated())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,35 @@
|
||||
"""Tests for integration-test backend selection helpers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
from tests._docker import skip_unless_docker_or_firecracker
|
||||
|
||||
|
||||
class TestSkipUnlessDockerOrFirecracker(unittest.TestCase):
|
||||
def test_firecracker_runs_when_docker_tests_are_disabled(self):
|
||||
with patch.dict(
|
||||
os.environ,
|
||||
{"BOT_BOTTLE_BACKEND": "firecracker", "SKIP_DOCKER_TESTS": "1"},
|
||||
clear=True,
|
||||
):
|
||||
decorated = skip_unless_docker_or_firecracker()(type("Case", (), {}))
|
||||
|
||||
self.assertFalse(getattr(decorated, "__unittest_skip__", False))
|
||||
|
||||
def test_non_firecracker_still_skips_when_docker_tests_are_disabled(self):
|
||||
with patch.dict(
|
||||
os.environ,
|
||||
{"BOT_BOTTLE_BACKEND": "docker", "SKIP_DOCKER_TESTS": "1"},
|
||||
clear=True,
|
||||
):
|
||||
decorated = skip_unless_docker_or_firecracker()(type("Case", (), {}))
|
||||
|
||||
self.assertTrue(getattr(decorated, "__unittest_skip__", False))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -35,9 +35,12 @@ class TestNetpoolSlots(unittest.TestCase):
|
||||
def test_slot_ip_math_31_pairs(self):
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_FC_IP_BASE": "100.64.0.0"}):
|
||||
s0, s1 = netpool.slot(0), netpool.slot(1)
|
||||
self.assertEqual(("bbfc0", "100.64.0.0", "100.64.0.1"),
|
||||
# Iface names track netpool's (env-driven) prefix — the KVM CI runner
|
||||
# overrides it for its isolated pool, so don't hardcode "bbfc".
|
||||
pfx = netpool.IFACE_PREFIX
|
||||
self.assertEqual((f"{pfx}0", "100.64.0.0", "100.64.0.1"),
|
||||
(s0.iface, s0.host_ip, s0.guest_ip))
|
||||
self.assertEqual(("bbfc1", "100.64.0.2", "100.64.0.3"),
|
||||
self.assertEqual((f"{pfx}1", "100.64.0.2", "100.64.0.3"),
|
||||
(s1.iface, s1.host_ip, s1.guest_ip))
|
||||
|
||||
def test_guest_cidr_is_31(self):
|
||||
@@ -180,11 +183,14 @@ class TestNetpoolOverlap(unittest.TestCase):
|
||||
self.assertEqual("tailscale0", conflicts[0].dev)
|
||||
|
||||
def test_ignores_own_taps_and_default(self):
|
||||
# The "own tap" route uses netpool's (env-driven) iface name, so the
|
||||
# test still exercises the self-ignore path on the KVM CI runner, whose
|
||||
# BOT_BOTTLE_FC_IFACE_PREFIX differs from the default.
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_FC_IP_BASE": "10.243.0.0",
|
||||
"BOT_BOTTLE_FC_POOL_SIZE": "8"}), \
|
||||
self._routes([
|
||||
{"dst": "default", "dev": "enp4s0"},
|
||||
{"dst": "10.243.0.0/31", "dev": "bbfc0"},
|
||||
{"dst": "10.243.0.0/31", "dev": netpool.slot(0).iface},
|
||||
{"dst": "192.168.1.0/24", "dev": "enp4s0"},
|
||||
]):
|
||||
self.assertEqual([], netpool.overlapping_routes())
|
||||
@@ -203,7 +209,7 @@ class TestNetpoolAllocation(unittest.TestCase):
|
||||
# over (and here, exhaust the pool).
|
||||
slot, lock = netpool.allocate("first")
|
||||
self.addCleanup(lock.close)
|
||||
self.assertEqual("bbfc0", slot.iface)
|
||||
self.assertEqual(netpool.slot(0).iface, slot.iface)
|
||||
with patch.object(netpool, "die",
|
||||
side_effect=SystemExit("exhausted")):
|
||||
with self.assertRaises(SystemExit):
|
||||
@@ -323,9 +329,14 @@ class TestNetpoolDefaultsSingleSource(unittest.TestCase):
|
||||
with patch.dict(os.environ, {}, clear=True):
|
||||
self.assertEqual(int(d["BOT_BOTTLE_FC_POOL_SIZE"]), netpool.pool_size())
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_IP_BASE"], netpool.ip_base())
|
||||
# Module constants resolve through the same shared file.
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_IFACE_PREFIX"], netpool.IFACE_PREFIX)
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_NFT_TABLE"], netpool.NFT_TABLE)
|
||||
# The module's *defaults* come from the same shared file. Assert the
|
||||
# parsed defaults (not IFACE_PREFIX/NFT_TABLE, which layer a live env
|
||||
# override on top — the KVM CI runner sets those for its isolated pool,
|
||||
# which would otherwise mask this single-source check).
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_IFACE_PREFIX"],
|
||||
netpool._DEFAULTS["BOT_BOTTLE_FC_IFACE_PREFIX"])
|
||||
self.assertEqual(d["BOT_BOTTLE_FC_NFT_TABLE"],
|
||||
netpool._DEFAULTS["BOT_BOTTLE_FC_NFT_TABLE"])
|
||||
|
||||
def test_env_var_overrides_the_shared_default(self):
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_FC_IP_BASE": "10.99.0.0"}):
|
||||
|
||||
@@ -63,9 +63,12 @@ class TestNetpoolProbes(unittest.TestCase):
|
||||
self.assertEqual(2, ok.call_count)
|
||||
|
||||
def test_missing_taps(self):
|
||||
# Derive the expected iface from netpool's (env-driven) config rather
|
||||
# than hardcoding "bbfc1": the KVM CI runner sets BOT_BOTTLE_FC_* for
|
||||
# its isolated pool, so the prefix there is not the default.
|
||||
with patch.dict("os.environ", {"BOT_BOTTLE_FC_POOL_SIZE": "2"}), \
|
||||
patch.object(netpool, "tap_present", side_effect=[True, False]):
|
||||
self.assertEqual(["bbfc1"], netpool.missing_taps())
|
||||
self.assertEqual([netpool.slot(1).iface], netpool.missing_taps())
|
||||
|
||||
def test_orch_slot_is_top_of_ip_base_16(self):
|
||||
# Dedicated orchestrator link: /31 at the top of the IP_BASE /16,
|
||||
|
||||
@@ -24,7 +24,7 @@ class TestBuildAgentRootfsDir(unittest.TestCase):
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
|
||||
def test_cache_hit_skips_rebuild(self):
|
||||
digest = image_builder._dockerfile_hash(self.dockerfile)
|
||||
digest = image_builder._rootfs_digest(self.dockerfile)
|
||||
base = self.cache / "rootfs" / f"agent-{digest}"
|
||||
base.mkdir(parents=True)
|
||||
(base / ".bb-ready").write_text("ok\n")
|
||||
@@ -55,6 +55,17 @@ class TestBuildAgentRootfsDir(unittest.TestCase):
|
||||
image_builder._dockerfile_hash(other),
|
||||
)
|
||||
|
||||
def test_rootfs_digest_tracks_dockerfile_and_init(self):
|
||||
# Same Dockerfile, different injected init -> different rootfs key, so
|
||||
# an init fix (e.g. /tmp perms) rebuilds instead of reusing a stale
|
||||
# rootfs; different Dockerfiles also differ.
|
||||
base = image_builder._rootfs_digest(self.dockerfile)
|
||||
with patch.object(image_builder.util, "_GUEST_INIT", "#!/bin/sh\n# changed\n"):
|
||||
self.assertNotEqual(base, image_builder._rootfs_digest(self.dockerfile))
|
||||
other = self.cache / "Dockerfile2"
|
||||
other.write_text("FROM python:3.12-slim\n")
|
||||
self.assertNotEqual(base, image_builder._rootfs_digest(other))
|
||||
|
||||
|
||||
class TestSmokeTest(unittest.TestCase):
|
||||
def test_empty_argv_is_noop(self):
|
||||
|
||||
@@ -38,7 +38,9 @@ class TestBuildInfraRootfs(unittest.TestCase):
|
||||
# and exports PATH so gateway_init's subprocess daemons find python3.
|
||||
init = build.call_args.kwargs["init_script"]
|
||||
self.assertIn("bot_bottle.orchestrator", init)
|
||||
self.assertIn("gateway_init.py", init)
|
||||
# Gateway launches via the installed package (there is no
|
||||
# /app/gateway_init.py file since the daemons moved into bot_bottle).
|
||||
self.assertIn("bot_bottle.gateway_init", init)
|
||||
self.assertIn("export PATH=", init)
|
||||
# Persistent registry volume mounted at the DB dir before the CP starts.
|
||||
self.assertIn("/dev/vdb", init)
|
||||
@@ -98,7 +100,11 @@ class TestRegistryVolume(unittest.TestCase):
|
||||
class TestEnsureBuilt(unittest.TestCase):
|
||||
def test_default_pulls_artifact_without_docker(self):
|
||||
# PRD 0069 Stage 2: the launch host pulls the prebuilt rootfs; no Docker.
|
||||
with patch.object(infra_vm.docker_mod, "build_image") as build, \
|
||||
# Pin BOT_BOTTLE_INFRA_BUILD off: the coverage CI job exports it =local
|
||||
# for the integration suite, and that ambient value would otherwise send
|
||||
# this default-path test down the local Docker-build branch.
|
||||
with patch.dict(os.environ, {"BOT_BOTTLE_INFRA_BUILD": ""}), \
|
||||
patch.object(infra_vm.docker_mod, "build_image") as build, \
|
||||
patch.object(infra_vm.infra_artifact, "ensure_artifact_gz") as pull:
|
||||
infra_vm.ensure_built()
|
||||
build.assert_not_called()
|
||||
@@ -138,27 +144,58 @@ class TestWaitForHealth(unittest.TestCase):
|
||||
|
||||
|
||||
class TestEnsureRunningSingleton(unittest.TestCase):
|
||||
def test_adopts_when_healthy(self):
|
||||
# A healthy control plane + existing key -> adopt (no boot), vm=None.
|
||||
with patch.object(infra_vm, "_health_ok", return_value=True), \
|
||||
patch.object(infra_vm, "_infra_dir") as d, \
|
||||
patch.object(infra_vm, "boot") as boot:
|
||||
keydir = MagicMock()
|
||||
(keydir / "id_ed25519").exists.return_value = True
|
||||
d.return_value = keydir
|
||||
infra = infra_vm.ensure_running()
|
||||
def test_adopts_when_healthy_and_version_matches(self):
|
||||
# Healthy control plane + existing key + matching version marker
|
||||
# -> adopt (no boot), vm=None.
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = Path(td)
|
||||
(d / "id_ed25519").write_text("k")
|
||||
(d / "booted-version").write_text("v-current\n")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d), \
|
||||
patch.object(infra_vm, "_expected_version", return_value="v-current"), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=True), \
|
||||
patch.object(infra_vm, "boot") as boot:
|
||||
infra = infra_vm.ensure_running()
|
||||
boot.assert_not_called()
|
||||
self.assertIsNone(infra.vm)
|
||||
|
||||
def test_reboots_when_version_stale(self):
|
||||
# Healthy control plane but the running VM booted an OLDER image
|
||||
# (marker mismatch) -> reboot rather than adopt stale code.
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = Path(td)
|
||||
(d / "id_ed25519").write_text("k")
|
||||
(d / "booted-version").write_text("v-old\n")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d), \
|
||||
patch.object(infra_vm, "_expected_version", return_value="v-current"), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=True), \
|
||||
patch.object(infra_vm, "stop") as stop, \
|
||||
patch.object(infra_vm, "ensure_built"), \
|
||||
patch.object(infra_vm, "wait_for_health"), \
|
||||
patch.object(infra_vm, "boot") as boot:
|
||||
boot.return_value = infra_vm.InfraVm(
|
||||
guest_ip="10.243.255.1", private_key=Path("/k"), vm=MagicMock())
|
||||
infra_vm.ensure_running()
|
||||
stop.assert_called_once() # dislodge the outdated VM
|
||||
boot.assert_called_once()
|
||||
# The fresh boot records the current version for the next launcher.
|
||||
self.assertEqual("v-current\n", (d / "booted-version").read_text())
|
||||
|
||||
def test_boots_when_unhealthy(self):
|
||||
with patch.object(infra_vm, "_health_ok", return_value=False), \
|
||||
patch.object(infra_vm, "stop") as stop, \
|
||||
patch.object(infra_vm, "ensure_built") as built, \
|
||||
patch.object(infra_vm, "boot") as boot, \
|
||||
patch.object(infra_vm, "wait_for_health") as wait:
|
||||
boot.return_value = infra_vm.InfraVm(
|
||||
guest_ip="10.243.255.1", private_key=Path("/k"), vm=MagicMock())
|
||||
infra_vm.ensure_running()
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=Path(td)), \
|
||||
patch.object(infra_vm, "_expected_version", return_value="v-current"), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=False), \
|
||||
patch.object(infra_vm, "stop") as stop, \
|
||||
patch.object(infra_vm, "ensure_built") as built, \
|
||||
patch.object(infra_vm, "boot") as boot, \
|
||||
patch.object(infra_vm, "wait_for_health") as wait:
|
||||
boot.return_value = infra_vm.InfraVm(
|
||||
guest_ip="10.243.255.1", private_key=Path("/k"), vm=MagicMock())
|
||||
infra_vm.ensure_running()
|
||||
stop.assert_called_once() # clear a stale VM first
|
||||
built.assert_called_once()
|
||||
boot.assert_called_once()
|
||||
@@ -185,5 +222,74 @@ class TestKillPidfile(unittest.TestCase):
|
||||
kill.assert_not_called()
|
||||
|
||||
|
||||
class TestAdoptable(unittest.TestCase):
|
||||
def _dir(self, td: str, *, key: bool = True, version: str | None = None) -> Path:
|
||||
d = Path(td)
|
||||
if key:
|
||||
(d / "id_ed25519").write_text("k")
|
||||
if version is not None:
|
||||
(d / "booted-version").write_text(version + "\n")
|
||||
return d
|
||||
|
||||
def test_true_when_key_version_and_health(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = self._dir(td, version="v1")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=True):
|
||||
self.assertTrue(infra_vm._adoptable(d / "id_ed25519", "u", "v1"))
|
||||
|
||||
def test_false_when_key_missing(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = self._dir(td, key=False, version="v1")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d):
|
||||
self.assertFalse(infra_vm._adoptable(d / "id_ed25519", "u", "v1"))
|
||||
|
||||
def test_false_when_no_version_marker(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = self._dir(td) # key present, no booted-version
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d):
|
||||
self.assertFalse(infra_vm._adoptable(d / "id_ed25519", "u", "v1"))
|
||||
|
||||
def test_false_when_version_mismatch(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
d = self._dir(td, version="v-old")
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=d), \
|
||||
patch.object(infra_vm, "_health_ok", return_value=True):
|
||||
self.assertFalse(infra_vm._adoptable(d / "id_ed25519", "u", "v1"))
|
||||
|
||||
|
||||
class TestKillInfraFirecrackers(unittest.TestCase):
|
||||
def _fake_proc(self, root: Path, pid: int, comm: str, cmdline: list[str]) -> None:
|
||||
p = root / str(pid)
|
||||
p.mkdir()
|
||||
(p / "comm").write_text(comm + "\n")
|
||||
(p / "cmdline").write_bytes(b"\0".join(a.encode() for a in cmdline) + b"\0")
|
||||
|
||||
def test_kills_only_matching_infra_firecracker(self):
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as td, \
|
||||
tempfile.TemporaryDirectory() as proc:
|
||||
infra_dir = Path(td)
|
||||
cfg = str(infra_dir / "config.json")
|
||||
root = Path(proc)
|
||||
# target: firecracker bound to the infra config -> killed
|
||||
self._fake_proc(root, 111, "firecracker",
|
||||
["firecracker", "--no-api", "--config-file", cfg])
|
||||
# a firecracker for a different (interactive) VM -> spared
|
||||
self._fake_proc(root, 222, "firecracker",
|
||||
["firecracker", "--config-file", "/home/u/other.json"])
|
||||
# a non-firecracker process on the same config path -> spared
|
||||
self._fake_proc(root, 333, "python3", ["python3", cfg])
|
||||
(root / "not-a-pid").mkdir()
|
||||
with patch.object(infra_vm, "_infra_dir", return_value=infra_dir), \
|
||||
patch.object(infra_vm.os, "kill") as kill:
|
||||
infra_vm._kill_infra_firecrackers(proc_root=root)
|
||||
kill.assert_called_once_with(111, infra_vm.signal.SIGKILL)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -2,9 +2,9 @@
|
||||
|
||||
Tests both the helper functions in `bot_bottle.gateway_init`
|
||||
and the supervisor's end-to-end signal / exit-code behavior. The
|
||||
end-to-end tests use real subprocesses (`/bin/sleep`,
|
||||
`/bin/sh -c '...'`) — short-lived, no docker required — so they
|
||||
run under `tests/unit/` rather than `tests/integration/`."""
|
||||
end-to-end tests use real subprocesses (`sleep`, `/bin/sh -c '...'`) —
|
||||
short-lived, no docker required — so they run under `tests/unit/`
|
||||
rather than `tests/integration/`."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -25,6 +25,7 @@ from bot_bottle.gateway_init import (
|
||||
_env_for_daemon,
|
||||
_selected_daemons,
|
||||
)
|
||||
from tests._bin import SLEEP
|
||||
|
||||
|
||||
class TestEnvForDaemon(unittest.TestCase):
|
||||
@@ -182,7 +183,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
# up and the supervisor never set shutdown_at.
|
||||
specs = [
|
||||
_DaemonSpec("crasher", ("/bin/sh", "-c", "exit 1")),
|
||||
_DaemonSpec("longrun", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("longrun", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -214,7 +215,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
# signal-killed longrun's negative returncode.
|
||||
specs = [
|
||||
_DaemonSpec("crasher", ("/bin/sh", "-c", "exit 1")),
|
||||
_DaemonSpec("longrun", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("longrun", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -259,7 +260,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
)
|
||||
specs = [
|
||||
_DaemonSpec("egress", sighup_marker),
|
||||
_DaemonSpec("other", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("other", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -282,7 +283,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_forward_signal_unknown_daemon_no_op(self):
|
||||
specs = [_DaemonSpec("a", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("a", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
delivered = sup.forward_signal(signal.SIGHUP, "ghost")
|
||||
@@ -294,8 +295,8 @@ class TestSupervisor(unittest.TestCase):
|
||||
# Restart one daemon; the other (supervise, the MCP server
|
||||
# in production) must remain untouched.
|
||||
specs = [
|
||||
_DaemonSpec("git-gate", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("supervise", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("git-gate", (SLEEP, "30")),
|
||||
_DaemonSpec("supervise", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -319,8 +320,8 @@ class TestSupervisor(unittest.TestCase):
|
||||
|
||||
def test_request_restart_is_drained_by_tick(self):
|
||||
specs = [
|
||||
_DaemonSpec("git-gate", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("supervise", ("/bin/sleep", "30")),
|
||||
_DaemonSpec("git-gate", (SLEEP, "30")),
|
||||
_DaemonSpec("supervise", (SLEEP, "30")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -343,7 +344,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_repeated_restart_requests_coalesce(self):
|
||||
specs = [_DaemonSpec("git-gate", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("git-gate", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
time.sleep(0.1)
|
||||
@@ -366,7 +367,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_request_restart_unknown_daemon_no_op(self):
|
||||
specs = [_DaemonSpec("a", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("a", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
ok = sup.request_restart("ghost")
|
||||
@@ -376,7 +377,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_restart_unknown_daemon_no_op(self):
|
||||
specs = [_DaemonSpec("a", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("a", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
ok = sup.restart_daemon("ghost")
|
||||
@@ -385,7 +386,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_restart_during_shutdown_is_no_op(self):
|
||||
specs = [_DaemonSpec("git-gate", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("git-gate", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
sup.request_shutdown(reason="test")
|
||||
@@ -395,7 +396,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self._drive(sup)
|
||||
|
||||
def test_pending_restart_dropped_during_shutdown(self):
|
||||
specs = [_DaemonSpec("git-gate", ("/bin/sleep", "30"))]
|
||||
specs = [_DaemonSpec("git-gate", (SLEEP, "30"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
time.sleep(0.1)
|
||||
@@ -413,8 +414,8 @@ class TestSupervisor(unittest.TestCase):
|
||||
# both should receive SIGTERM and exit. Signal-only
|
||||
# shutdown clamps to a zero supervisor exit code.
|
||||
specs = [
|
||||
_DaemonSpec("a", ("/bin/sleep", "60")),
|
||||
_DaemonSpec("b", ("/bin/sleep", "60")),
|
||||
_DaemonSpec("a", (SLEEP, "60")),
|
||||
_DaemonSpec("b", (SLEEP, "60")),
|
||||
]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
@@ -449,7 +450,7 @@ class TestSupervisor(unittest.TestCase):
|
||||
self.assertEqual(0, rc)
|
||||
|
||||
def test_idempotent_shutdown_requests(self):
|
||||
specs = [_DaemonSpec("a", ("/bin/sleep", "60"))]
|
||||
specs = [_DaemonSpec("a", (SLEEP, "60"))]
|
||||
sup = _Supervisor(specs)
|
||||
sup.start_all()
|
||||
time.sleep(0.1)
|
||||
@@ -470,7 +471,7 @@ class TestMainEndToEnd(unittest.TestCase):
|
||||
|
||||
@classmethod
|
||||
def setUpClass(cls):
|
||||
for p in ("/bin/sh", "/bin/sleep"):
|
||||
for p in ("/bin/sh", SLEEP):
|
||||
if not Path(p).exists():
|
||||
raise unittest.SkipTest(f"missing {p}")
|
||||
|
||||
@@ -485,8 +486,8 @@ class TestMainEndToEnd(unittest.TestCase):
|
||||
"import os, runpy, sys\n"
|
||||
"from bot_bottle import gateway_init as si\n"
|
||||
"si._DAEMONS = (\n"
|
||||
" si._DaemonSpec('alpha', ('/bin/sleep','30')),\n"
|
||||
" si._DaemonSpec('beta', ('/bin/sleep','30')),\n"
|
||||
f" si._DaemonSpec('alpha', ({SLEEP!r},'30')),\n"
|
||||
f" si._DaemonSpec('beta', ({SLEEP!r},'30')),\n"
|
||||
")\n"
|
||||
"sys.exit(si.main([]))\n"
|
||||
)
|
||||
|
||||
@@ -168,16 +168,18 @@ class TestHookRender(unittest.TestCase):
|
||||
# Stdin is buffered to a tempfile so both phases can re-read.
|
||||
self.assertIn("refs_file=$(mktemp)", hook)
|
||||
|
||||
def test_new_ref_scan_scoped_to_incoming_commits(self):
|
||||
# A new branch (old=all-zeros) must scan only commits new to the
|
||||
# gate, not the full ancestry — otherwise historical findings
|
||||
# block every new-branch push (PRD 0028 / issue #106).
|
||||
def test_scan_scoped_to_incoming_commits(self):
|
||||
# Every non-delete push scans only commits new to the gate, not
|
||||
# the full ancestry and not the `$old..$new` delta — otherwise
|
||||
# historical fixtures block new-branch pushes (PRD 0028 / #106)
|
||||
# and a rebase/force-push onto an advanced main drags in main's
|
||||
# history incl. the sandbox-escape fixtures (#346).
|
||||
hook = git_gate_render_hook()
|
||||
self.assertIn('log_opts="$new --not --all"', hook)
|
||||
# The old over-broad full-ancestry range must be gone.
|
||||
# Neither the full-ancestry range nor the ancestry-blind delta
|
||||
# range may survive.
|
||||
self.assertNotIn('log_opts="$new"', hook)
|
||||
# Existing-branch delta scan is unchanged.
|
||||
self.assertIn('log_opts="$old..$new"', hook)
|
||||
self.assertNotIn('log_opts="$old..$new"', hook)
|
||||
|
||||
def test_forward_ssh_is_non_interactive_and_bounded(self):
|
||||
# No prompt (BatchMode) and a connect timeout, so an unreachable
|
||||
|
||||
@@ -50,7 +50,12 @@ class _CacheMixin(unittest.TestCase):
|
||||
self._env = mock.patch.dict(
|
||||
os.environ,
|
||||
{"BOT_BOTTLE_FC_CACHE": self._tmp.name,
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_TOKEN": ""},
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_TOKEN": "",
|
||||
# Pin the candidate-dir override off: the coverage CI job exports a
|
||||
# candidate dir for the integration suite, and an ambient value
|
||||
# would send these registry-pull tests down the local-bundle path.
|
||||
# Cases that exercise the candidate path set it explicitly.
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": ""},
|
||||
clear=False,
|
||||
)
|
||||
self._env.start()
|
||||
@@ -93,6 +98,31 @@ class TestVersionInputs(unittest.TestCase):
|
||||
(pkg / "netpool.defaults.env").write_text("FOO=1\n")
|
||||
for name in ("Dockerfile.orchestrator", "Dockerfile.gateway", "Dockerfile.infra"):
|
||||
(root / name).write_text(f"FROM scratch # {name}\n")
|
||||
(root / "pyproject.toml").write_text("[project]\nname = 'bot-bottle'\n")
|
||||
|
||||
def test_pyproject_toml_change_bumps_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._fake_repo(root)
|
||||
before = ia.infra_artifact_version("init", repo_root=root)
|
||||
(root / "pyproject.toml").write_text(
|
||||
"[project]\nname = 'bot-bottle'\ndependencies = ['httpx']\n")
|
||||
after = ia.infra_artifact_version("init", repo_root=root)
|
||||
self.assertNotEqual(before, after)
|
||||
|
||||
def test_dropbear_change_bumps_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._fake_repo(root)
|
||||
dropbear = root / "dropbear"
|
||||
dropbear.write_bytes(b"dropbear-v1")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_FC_DROPBEAR": str(dropbear),
|
||||
}):
|
||||
before = ia.infra_artifact_version("init", repo_root=root)
|
||||
dropbear.write_bytes(b"dropbear-v2")
|
||||
after = ia.infra_artifact_version("init", repo_root=root)
|
||||
self.assertNotEqual(before, after)
|
||||
|
||||
def test_non_python_file_change_bumps_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
@@ -118,6 +148,58 @@ class TestVersionInputs(unittest.TestCase):
|
||||
|
||||
|
||||
class TestEnsureArtifact(_CacheMixin):
|
||||
def test_uses_verified_ci_candidate_without_network(self) -> None:
|
||||
version = "deadbeef00000000"
|
||||
gz = _gz(b"candidate ext4")
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
(root / "version.txt").write_text(version + "\n")
|
||||
(root / "rootfs.ext4.gz").write_bytes(gz)
|
||||
digest = hashlib.sha256(gz).hexdigest()
|
||||
(root / "rootfs.ext4.gz.sha256").write_text(
|
||||
f"{digest} rootfs.ext4.gz\n")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": str(root),
|
||||
}), mock.patch.object(ia.urllib.request, "urlopen") as net:
|
||||
path = ia.ensure_artifact_gz(version)
|
||||
self.assertEqual(root / "rootfs.ext4.gz", path)
|
||||
net.assert_not_called()
|
||||
|
||||
def test_rejects_candidate_for_another_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
(root / "version.txt").write_text("wrong\n")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": str(root),
|
||||
}):
|
||||
with self.assertRaises(Die) as ctx:
|
||||
ia.ensure_artifact_gz("expected")
|
||||
self.assertIn("version mismatch", str(ctx.exception.message))
|
||||
|
||||
def test_rejects_incomplete_candidate(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
(root / "version.txt").write_text("v1\n")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": str(root),
|
||||
}):
|
||||
with self.assertRaises(Die) as ctx:
|
||||
ia.ensure_artifact_gz("v1")
|
||||
self.assertIn("incomplete", str(ctx.exception.message))
|
||||
|
||||
def test_rejects_candidate_checksum_mismatch(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
(root / "version.txt").write_text("v1\n")
|
||||
(root / "rootfs.ext4.gz").write_bytes(b"bad")
|
||||
(root / "rootfs.ext4.gz.sha256").write_text("0" * 64 + " rootfs.ext4.gz\n")
|
||||
with mock.patch.dict(os.environ, {
|
||||
"BOT_BOTTLE_INFRA_ARTIFACT_DIR": str(root),
|
||||
}):
|
||||
with self.assertRaises(Die) as ctx:
|
||||
ia.ensure_artifact_gz("v1")
|
||||
self.assertIn("checksum mismatch", str(ctx.exception.message))
|
||||
|
||||
def test_downloads_verifies_and_caches(self) -> None:
|
||||
version = "deadbeef00000000"
|
||||
gz = _gz(b"fake ext4 bytes")
|
||||
|
||||
@@ -46,7 +46,9 @@ class TestInfraRun(unittest.TestCase):
|
||||
argv = self._run_container(MacosInfraService(repo_root=Path("/r")))
|
||||
script = argv[-1]
|
||||
self.assertIn("bot_bottle.orchestrator", script)
|
||||
self.assertIn("gateway_init.py", script)
|
||||
# Gateway launches via the installed package (there is no
|
||||
# /app/gateway_init.py file since the daemons moved into bot_bottle).
|
||||
self.assertIn("bot_bottle.gateway_init", script)
|
||||
self.assertIn("127.0.0.1", script) # they reach each other on loopback
|
||||
|
||||
def test_db_is_a_container_only_volume(self) -> None:
|
||||
|
||||
@@ -6,8 +6,11 @@ read it into memory. Network is mocked; no Docker, no real build.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
from email.message import Message
|
||||
import tempfile
|
||||
import unittest
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
@@ -58,5 +61,99 @@ class TestPut(unittest.TestCase):
|
||||
self.assertEqual(b"abc123 rootfs\n", captured[0].data)
|
||||
|
||||
|
||||
class TestPublishBundle(unittest.TestCase):
|
||||
def _bundle(self, root: Path, version: str) -> None:
|
||||
payload = b"candidate"
|
||||
(root / "version.txt").write_text(version + "\n")
|
||||
(root / "rootfs.ext4.gz").write_bytes(payload)
|
||||
digest = hashlib.sha256(payload).hexdigest()
|
||||
(root / "rootfs.ext4.gz.sha256").write_text(
|
||||
f"{digest} rootfs.ext4.gz\n")
|
||||
|
||||
def test_existing_identical_artifact_is_success(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "v1")
|
||||
sha = (root / "rootfs.ext4.gz.sha256").read_bytes()
|
||||
response = mock.MagicMock()
|
||||
response.__enter__.return_value.read.return_value = sha
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="v1"
|
||||
), mock.patch.object(
|
||||
pub.urllib.request, "urlopen", return_value=response
|
||||
), mock.patch.object(pub, "_put") as put:
|
||||
self.assertEqual("v1", pub._publish_bundle(root, "token"))
|
||||
put.assert_not_called()
|
||||
|
||||
def test_partial_artifact_is_replaced(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "v1")
|
||||
missing = urllib.error.HTTPError("u", 404, "missing", Message(), None)
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="v1"
|
||||
), mock.patch.object(
|
||||
pub.urllib.request, "urlopen", side_effect=missing
|
||||
), mock.patch.object(pub, "_delete") as delete, \
|
||||
mock.patch.object(pub, "_put") as put:
|
||||
pub._publish_bundle(root, "token")
|
||||
self.assertEqual(3, delete.call_count)
|
||||
self.assertEqual(3, put.call_count)
|
||||
|
||||
def test_rejects_bundle_for_different_checkout(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "old")
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="new"
|
||||
):
|
||||
with self.assertRaises(SystemExit) as ctx:
|
||||
pub._publish_bundle(root, "token")
|
||||
self.assertIn("does not match checkout", str(ctx.exception))
|
||||
|
||||
def test_rejects_bad_bundle_checksum(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "v1")
|
||||
(root / "rootfs.ext4.gz").write_bytes(b"tampered")
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="v1"
|
||||
):
|
||||
with self.assertRaises(SystemExit) as ctx:
|
||||
pub._publish_bundle(root, "token")
|
||||
self.assertIn("checksum mismatch", str(ctx.exception))
|
||||
|
||||
def test_registry_lookup_failure_is_reported(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d)
|
||||
self._bundle(root, "v1")
|
||||
failure = urllib.error.URLError("offline")
|
||||
with mock.patch.object(
|
||||
pub.infra_artifact, "infra_artifact_version", return_value="v1"
|
||||
), mock.patch.object(pub.urllib.request, "urlopen", side_effect=failure):
|
||||
with self.assertRaises(SystemExit) as ctx:
|
||||
pub._publish_bundle(root, "token")
|
||||
self.assertIn("registry unreachable", str(ctx.exception))
|
||||
|
||||
|
||||
class TestMain(unittest.TestCase):
|
||||
def test_output_builds_candidate_and_records_version(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
root = Path(d) / "candidate"
|
||||
with mock.patch.object(
|
||||
pub, "build_artifact", return_value=("v1", root / "g", root / "s")
|
||||
) as build:
|
||||
self.assertEqual(0, pub.main(["--output", str(root)]))
|
||||
build.assert_called_once_with(root)
|
||||
self.assertEqual("v1\n", (root / "version.txt").read_text())
|
||||
|
||||
def test_publish_dir_publishes_existing_candidate(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as d, \
|
||||
mock.patch.object(pub.infra_artifact, "_config", return_value=("", "", "t")), \
|
||||
mock.patch.object(pub, "_publish_bundle", return_value="v1") as publish:
|
||||
self.assertEqual(0, pub.main(["--publish-dir", d]))
|
||||
publish.assert_called_once_with(Path(d), "t")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -32,7 +32,9 @@ from bot_bottle.supervise_server import (
|
||||
_RpcError,
|
||||
_RpcInternalError,
|
||||
_response_timeout_from_env,
|
||||
format_pending_response_text,
|
||||
format_response_text,
|
||||
handle_check_proposal,
|
||||
handle_initialize,
|
||||
handle_tools_call,
|
||||
handle_tools_list,
|
||||
@@ -218,6 +220,7 @@ class TestHandleToolsList(unittest.TestCase):
|
||||
_sv.TOOL_EGRESS_ALLOW,
|
||||
_sv.TOOL_EGRESS_BLOCK,
|
||||
_sv.TOOL_LIST_EGRESS_ROUTES,
|
||||
_sv.TOOL_CHECK_PROPOSAL,
|
||||
]),
|
||||
sorted(names),
|
||||
)
|
||||
@@ -484,9 +487,10 @@ class TestFormatResponseText(unittest.TestCase):
|
||||
|
||||
class TestFormatPendingResponseText(unittest.TestCase):
|
||||
def test_formats_timeout_message(self):
|
||||
text = supervise_server.format_pending_response_text(12.5)
|
||||
text = supervise_server.format_pending_response_text("prop-9", 12.5)
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn("12.5s", text)
|
||||
self.assertIn("proposal_id: prop-9", text)
|
||||
|
||||
|
||||
# --- End-to-end HTTP sanity ------------------------------------------------
|
||||
@@ -685,5 +689,129 @@ class TestResolvedRoutesPayload(unittest.TestCase):
|
||||
_handler(None)._resolved_routes_payload()
|
||||
|
||||
|
||||
class TestNonBlockingSupervise(unittest.TestCase):
|
||||
"""PRD prd-new / issue #412: pending responses carry the proposal id, and
|
||||
`check-proposal` polls a queued proposal without blocking or re-proposing."""
|
||||
|
||||
_ROUTES = "routes:\n - host: example.com\n"
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory(prefix="supervise-nonblock-test.")
|
||||
self._home_patch = use_bottle_root(Path(self._tmp.name) / ".bot-bottle")
|
||||
self.config = ServerConfig(bottle_slug="dev")
|
||||
_qs.QueueStore("dev").migrate()
|
||||
_as.AuditStore().migrate()
|
||||
|
||||
def tearDown(self):
|
||||
self._home_patch()
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _seed_proposal(self) -> "_sv.Proposal":
|
||||
p = _sv.Proposal.new(
|
||||
bottle_slug="dev",
|
||||
tool=_sv.TOOL_EGRESS_ALLOW,
|
||||
proposed_file=self._ROUTES,
|
||||
justification="need example.com",
|
||||
current_file_hash=_sv.sha256_hex(self._ROUTES),
|
||||
)
|
||||
_sv.write_proposal(p)
|
||||
return p
|
||||
|
||||
def _check(self, proposal_id: str) -> dict[str, object]:
|
||||
return handle_check_proposal({"arguments": {"proposal_id": proposal_id}}, self.config)
|
||||
|
||||
# --- pending response carries the id ---
|
||||
|
||||
def test_pending_text_includes_id_and_pointer(self):
|
||||
text = format_pending_response_text("abc-123", 30.0)
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn("proposal_id: abc-123", text)
|
||||
self.assertIn("check-proposal", text)
|
||||
|
||||
def test_tools_call_timeout_returns_pending_with_id_and_stays_queued(self):
|
||||
# No responder → the grace window expires → pending, not blocked forever.
|
||||
result = handle_tools_call(
|
||||
{
|
||||
"name": _sv.TOOL_EGRESS_ALLOW,
|
||||
"arguments": {"routes_yaml": self._ROUTES, "justification": "x"},
|
||||
},
|
||||
ServerConfig(bottle_slug="dev", response_timeout_seconds=0.05),
|
||||
)
|
||||
self.assertFalse(result["isError"]) # type: ignore[index]
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: pending", text)
|
||||
pending = _sv.list_pending_proposals("dev")
|
||||
self.assertEqual(1, len(pending)) # still queued, not archived
|
||||
self.assertIn(pending[0].id, text) # agent got the id to poll
|
||||
|
||||
# --- check-proposal poll ---
|
||||
|
||||
def test_check_returns_approved_and_archives(self):
|
||||
p = self._seed_proposal()
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=p.id, status=_sv.STATUS_APPROVED, notes="ok"))
|
||||
result = self._check(p.id)
|
||||
self.assertFalse(result["isError"])
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: approved", text)
|
||||
self.assertIn("notes: ok", text)
|
||||
with self.assertRaises(FileNotFoundError): # archived on read
|
||||
_sv.read_proposal("dev", p.id)
|
||||
|
||||
def test_check_rejected_sets_isError(self):
|
||||
p = self._seed_proposal()
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=p.id, status=_sv.STATUS_REJECTED, notes="no"))
|
||||
result = self._check(p.id)
|
||||
self.assertTrue(result["isError"])
|
||||
self.assertIn("status: rejected", result["content"][0]["text"]) # type: ignore[index]
|
||||
|
||||
def test_check_pending_when_no_decision_yet(self):
|
||||
p = self._seed_proposal()
|
||||
result = self._check(p.id)
|
||||
self.assertFalse(result["isError"])
|
||||
text = result["content"][0]["text"] # type: ignore[index]
|
||||
self.assertIn("status: pending", text)
|
||||
self.assertIn(p.id, text)
|
||||
self.assertEqual(1, len(_sv.list_pending_proposals("dev"))) # not archived
|
||||
|
||||
def test_check_unknown_id_is_error(self):
|
||||
result = self._check("no-such-proposal")
|
||||
self.assertTrue(result["isError"])
|
||||
self.assertIn("status: unknown", result["content"][0]["text"]) # type: ignore[index]
|
||||
|
||||
def test_check_missing_id_raises(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": {}}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_check_empty_id_raises(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": {"proposal_id": " "}}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_check_arguments_must_be_object(self):
|
||||
with self.assertRaises(_RpcClientError) as cm:
|
||||
handle_check_proposal({"arguments": []}, self.config)
|
||||
self.assertEqual(ERR_INVALID_PARAMS, cm.exception.code)
|
||||
|
||||
def test_full_nonblocking_round_trip(self):
|
||||
# 1. tools/call times out → pending with id
|
||||
result = handle_tools_call(
|
||||
{
|
||||
"name": _sv.TOOL_EGRESS_ALLOW,
|
||||
"arguments": {"routes_yaml": self._ROUTES, "justification": "x"},
|
||||
},
|
||||
ServerConfig(bottle_slug="dev", response_timeout_seconds=0.05),
|
||||
)
|
||||
pid = _sv.list_pending_proposals("dev")[0].id
|
||||
self.assertIn(pid, result["content"][0]["text"]) # type: ignore[index]
|
||||
# 2. operator decides out-of-band
|
||||
_sv.write_response("dev", _sv.Response(proposal_id=pid, status=_sv.STATUS_APPROVED, notes="ok"))
|
||||
# 3. agent resumes by polling — no re-proposing
|
||||
poll = self._check(pid)
|
||||
self.assertFalse(poll["isError"])
|
||||
self.assertIn("status: approved", poll["content"][0]["text"]) # type: ignore[index]
|
||||
self.assertEqual([], _sv.list_pending_proposals("dev")) # resolved + archived
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
import unittest
|
||||
from unittest.mock import Mock
|
||||
|
||||
from scripts.tracker_policy import (
|
||||
TRIAGE_LABEL,
|
||||
check_pull_request,
|
||||
deliberate_issue_numbers,
|
||||
ensure_issue_label,
|
||||
)
|
||||
|
||||
|
||||
class TestDeliberateIssueNumbers(unittest.TestCase):
|
||||
def test_accepts_completing_and_noncompleting_forms(self):
|
||||
self.assertEqual(
|
||||
deliberate_issue_numbers("Fixes #12", "Part of #14; refs #15"),
|
||||
{12, 14, 15},
|
||||
)
|
||||
|
||||
def test_does_not_treat_incidental_number_as_link(self):
|
||||
self.assertEqual(deliberate_issue_numbers("Audit #12", "See PR #14"), set())
|
||||
|
||||
|
||||
class TestCheckPullRequest(unittest.TestCase):
|
||||
def test_accepts_unlabelled_pr_linked_to_real_issue(self):
|
||||
api = Mock()
|
||||
api.request.return_value = {"number": 12, "pull_request": None}
|
||||
event = {"pull_request": {"title": "Change", "body": "Part of #12", "labels": []}}
|
||||
self.assertEqual(check_pull_request(event, api), [])
|
||||
|
||||
def test_rejects_labels_and_pr_reference(self):
|
||||
api = Mock()
|
||||
api.request.return_value = {"number": 12, "pull_request": {}}
|
||||
event = {
|
||||
"pull_request": {
|
||||
"title": "Change",
|
||||
"body": "Closes #12",
|
||||
"labels": [{"name": "Kind/Bug"}],
|
||||
}
|
||||
}
|
||||
errors = check_pull_request(event, api)
|
||||
self.assertEqual(len(errors), 2)
|
||||
self.assertIn("unlabeled", errors[0])
|
||||
self.assertIn("not an issue", errors[1])
|
||||
|
||||
|
||||
class TestEnsureIssueLabel(unittest.TestCase):
|
||||
def test_adds_triage_label_to_unlabelled_issue(self):
|
||||
api = Mock()
|
||||
api.request.side_effect = [[{"id": 55, "name": TRIAGE_LABEL}], None]
|
||||
event = {"issue": {"number": 405, "labels": [], "pull_request": None}}
|
||||
self.assertTrue(ensure_issue_label(event, api))
|
||||
api.request.assert_any_call("POST", "/issues/405/labels", {"labels": [55]})
|
||||
|
||||
def test_leaves_labelled_issue_unchanged(self):
|
||||
api = Mock()
|
||||
event = {"issue": {"number": 405, "labels": [{"name": "Kind/Documentation"}]}}
|
||||
self.assertFalse(ensure_issue_label(event, api))
|
||||
api.request.assert_not_called()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user