Compare commits

..

1 Commits

Author SHA1 Message Date
didericis-claude 01cee056be feat(secrets): encrypt egress tokens at rest with per-bottle ENV_VAR_SECRET
test / integration-docker (push) Successful in 45s
test / unit (push) Successful in 48s
Update Quality Badges / update-badges (push) Failing after 52s
test / integration-firecracker (push) Successful in 5m27s
test / coverage (push) Failing after 29s
test / publish-infra (push) Has been skipped
lint / lint (push) Successful in 2m31s
Implements the interim secret-provider design (PRD prd-new-secret-provider):
each agent receives a random ENV_VAR_SECRET injected into its container env
at launch. The host uses this key to encrypt each egress auth token value
(HMAC-SHA256 CTR mode, stdlib-only) and store it in a new
bottled_agent_secrets table (one row per env var, key column plaintext for
auditing). The key never touches the DB.

On infra container restart the in-memory token map is lost. launch_consolidated
now calls _reprovision_running_bottles after ensure_running: for each
registered bottle still alive on the gateway network it execs
`printenv ENV_VAR_SECRET` into the agent container and posts the result to the
new POST /bottles/<id>/reprovision_gateway control-plane endpoint, which
decrypts the stored rows and restores _tokens — no manual intervention needed.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-07-22 00:12:34 +00:00
10 changed files with 50 additions and 611 deletions
+11 -24
View File
@@ -1,5 +1,4 @@
# Run the project's test suite when package or runtime inputs change on a PR # Run the project's test suite on every PR push and on push to main.
# or on push to main.
# #
# The suite uses stdlib `unittest` discovery — no external Python # The suite uses stdlib `unittest` discovery — no external Python
# dependencies are required to execute it. Tests are split by directory: # dependencies are required to execute it. Tests are split by directory:
@@ -24,34 +23,22 @@ on:
branches: branches:
- main - main
paths: paths:
- 'bot_bottle/**' - '**.py'
- 'tests/**/*.py' - '.gitea/workflows/**.yml'
- 'cli.py' - 'scripts/**'
- 'scripts/coverage.sh' - 'README.md'
- 'scripts/critical-modules.txt' # Dockerfiles and pyproject.toml are baked into the infra rootfs; a
- 'scripts/diff_coverage.py' # change here alters what the integration/coverage jobs build locally.
- 'scripts/tracker_policy.py'
- 'scripts/firecracker-netpool.sh'
- 'Dockerfile*' - 'Dockerfile*'
- 'pyproject.toml' - 'pyproject.toml'
- 'requirements-dev.txt'
- '.coveragerc'
- '.dockerignore'
pull_request: pull_request:
paths: paths:
- 'bot_bottle/**' - '**.py'
- 'tests/**/*.py' - '.gitea/workflows/**.yml'
- 'cli.py' - 'scripts/**'
- 'scripts/coverage.sh' - 'README.md'
- 'scripts/critical-modules.txt'
- 'scripts/diff_coverage.py'
- 'scripts/tracker_policy.py'
- 'scripts/firecracker-netpool.sh'
- 'Dockerfile*' - 'Dockerfile*'
- 'pyproject.toml' - 'pyproject.toml'
- 'requirements-dev.txt'
- '.coveragerc'
- '.dockerignore'
workflow_dispatch: workflow_dispatch:
jobs: jobs:
+1 -1
View File
@@ -21,7 +21,7 @@ jobs:
- uses: actions/checkout@v3 - uses: actions/checkout@v3
with: with:
fetch-depth: 0 fetch-depth: 0
token: ${{ secrets.BADGE_PUSH_TOKEN }} token: ${{ secrets.GITHUB_TOKEN }}
# No actions/setup-python: the runner image ships Python 3.12 and older # No actions/setup-python: the runner image ships Python 3.12 and older
# act_runner engines mishandle setup-python's PATH. Install into the # act_runner engines mishandle setup-python's PATH. Install into the
+1 -1
View File
@@ -5,7 +5,7 @@
# bot-bottle # bot-bottle
[![test](https://gitea.dideric.is/didericis/bot-bottle/actions/workflows/test.yml/badge.svg?branch=main)](https://gitea.dideric.is/didericis/bot-bottle/actions?workflow=test.yml) [![test](https://gitea.dideric.is/didericis/bot-bottle/actions/workflows/test.yml/badge.svg?branch=main)](https://gitea.dideric.is/didericis/bot-bottle/actions?workflow=test.yml)
[![coverage](https://img.shields.io/badge/coverage-83%25-brightgreen)](https://coverage.readthedocs.io/) [![coverage](https://img.shields.io/badge/coverage-81%25-brightgreen)](https://coverage.readthedocs.io/)
[![core coverage](https://img.shields.io/badge/core%20coverage-94%25-brightgreen)](https://gitea.dideric.is/didericis/bot-bottle/src/branch/main/docs/decisions/0004-coverage-policy.md) [![core coverage](https://img.shields.io/badge/core%20coverage-94%25-brightgreen)](https://gitea.dideric.is/didericis/bot-bottle/src/branch/main/docs/decisions/0004-coverage-policy.md)
**Problem:** Developer wants to run a coding agent without supervision, but they don't want a prompt injected or misbehaving agent wrecking their environment or exfiltrating sensitive data. **Problem:** Developer wants to run a coding agent without supervision, but they don't want a prompt injected or misbehaving agent wrecking their environment or exfiltrating sensitive data.
+16 -66
View File
@@ -1,23 +1,8 @@
"""Cleanup for the Firecracker backend. """Cleanup for the Firecracker backend.
Reaps *orphans* only — resources with no live VM behind them: Orphans are: firecracker VMM processes whose config lives under our run
dir, and the per-bottle run dirs. TAP slots free themselves (the flock
* orphan run dirs: a per-bottle run dir (holding the ~1G rootfs.ext4) drops when the launcher exits), so there is nothing to reclaim there.
whose firecracker process has exited. These leak when a launch is
hard-killed before its teardown runs (host OOM/crash, a cancelled CI
job, `kill -9`); the clean-exit path already removes its own dir in
launch.py.
* orphan VM pids: a firecracker process whose run dir is already gone
— a VMM left lingering after its dir was removed.
A run dir with a *live* firecracker process is a running bottle and is
left strictly alone: it is neither killed nor removed. (The backend's
`enumerate_active` registry is still a stub — #354 — so a live process
is the only reliable "this bottle is in use" signal we have. Once the
registry lands, registry-orphaned-but-running VMs can be reaped too.)
TAP slots free themselves (the flock drops when the launcher exits), so
there is nothing to reclaim there.
""" """
from __future__ import annotations from __future__ import annotations
@@ -37,73 +22,38 @@ def _run_root() -> Path:
return util.cache_dir() / "run" return util.cache_dir() / "run"
def _run_dir_of(cmd: str, run_root: Path) -> Path | None: def _orphan_vm_pids() -> list[int]:
"""The bottle run dir a firecracker cmdline belongs to, or None. """firecracker processes whose --config-file is under our run dir."""
run_root = str(_run_root())
A bottle VM is launched with `--config-file <run_root>/<slug>/config.json`,
so the run dir is the config file's parent when it sits directly under
the run root. Anything else (a builder VM, the infra VM elsewhere) is
not ours to reap here.
"""
toks = cmd.split()
for i, tok in enumerate(toks):
if tok == "--config-file" and i + 1 < len(toks):
parent = Path(toks[i + 1]).parent
if parent.parent == run_root:
return parent
return None
def _scan_processes(run_root: Path) -> tuple[set[str], list[int]]:
"""Inspect running firecracker VMs under ``run_root``.
Returns ``(live_run_dirs, orphan_pids)``:
* ``live_run_dirs`` — run dirs backed by a running VM (never reaped);
* ``orphan_pids`` — firecracker pids whose run dir no longer exists
(a lingering VMM to kill).
"""
result = subprocess.run( result = subprocess.run(
["pgrep", "-a", "firecracker"], ["pgrep", "-a", "firecracker"],
capture_output=True, text=True, check=False, capture_output=True, text=True, check=False,
) )
if result.returncode != 0: if result.returncode != 0:
return set(), [] return []
live: set[str] = set() pids: list[int] = []
orphan_pids: list[int] = []
for line in result.stdout.splitlines(): for line in result.stdout.splitlines():
parts = line.split(None, 1) parts = line.split(None, 1)
if len(parts) != 2: if len(parts) != 2 or run_root not in parts[1]:
continue continue
try: try:
pid = int(parts[0]) pids.append(int(parts[0]))
except ValueError: except ValueError:
continue continue
run_dir = _run_dir_of(parts[1], run_root) return pids
if run_dir is None:
continue
if run_dir.is_dir():
live.add(str(run_dir))
else:
orphan_pids.append(pid)
return live, orphan_pids
def _orphan_run_dirs(run_root: Path, live: set[str]) -> list[str]: def _run_dirs() -> list[str]:
"""Run dirs with no live VM behind them — the leaked ones to remove.""" run_root = _run_root()
if not run_root.is_dir(): if not run_root.is_dir():
return [] return []
return sorted( return sorted(str(p) for p in run_root.iterdir() if p.is_dir())
str(p) for p in run_root.iterdir()
if p.is_dir() and str(p) not in live
)
def prepare_cleanup() -> FirecrackerBottleCleanupPlan: def prepare_cleanup() -> FirecrackerBottleCleanupPlan:
run_root = _run_root()
live, orphan_pids = _scan_processes(run_root)
return FirecrackerBottleCleanupPlan( return FirecrackerBottleCleanupPlan(
vm_pids=tuple(orphan_pids), vm_pids=tuple(_orphan_vm_pids()),
run_dirs=tuple(_orphan_run_dirs(run_root, live)), run_dirs=tuple(_run_dirs()),
) )
-5
View File
@@ -26,7 +26,6 @@ from __future__ import annotations
import dataclasses import dataclasses
import os import os
import shutil
from contextlib import ExitStack, contextmanager from contextlib import ExitStack, contextmanager
from pathlib import Path from pathlib import Path
from typing import Callable, Generator from typing import Callable, Generator
@@ -168,10 +167,6 @@ def launch(
# Step 6: build the per-bottle rootfs + SSH key, then boot. # Step 6: build the per-bottle rootfs + SSH key, then boot.
run_dir = util.cache_dir() / "run" / plan.slug run_dir = util.cache_dir() / "run" / plan.slug
run_dir.mkdir(parents=True, exist_ok=True) run_dir.mkdir(parents=True, exist_ok=True)
# Remove the run dir on teardown so the per-bottle rootfs.ext4 (~1G)
# doesn't leak. Registered before vm.terminate below so it runs *after*
# it (ExitStack is LIFO): the VM is gone before we rm its rootfs.
stack.callback(lambda: shutil.rmtree(run_dir, ignore_errors=True))
rootfs = run_dir / "rootfs.ext4" rootfs = run_dir / "rootfs.ext4"
util.build_rootfs_ext4(agent_base, rootfs) util.build_rootfs_ext4(agent_base, rootfs)
private_key, pubkey = util.generate_keypair(run_dir) private_key, pubkey = util.generate_keypair(run_dir)
@@ -1,128 +0,0 @@
# PRD prd-new: Encrypted at-rest egress secrets (SecretProvider, interim slice)
- **Status:** Draft
- **Author:** didericis
- **Created:** 2026-07-21
- **Issue:** #355
## Summary
An interim step toward the generic `SecretProvider` (#355) that stops short
of per-request minting. Today the orchestrator holds each bottle's egress
auth tokens **in process memory only**, so any infra-container recreation
silently strips every already-running bottle of its upstream credentials.
This PRD makes those secrets survive a gateway restart by persisting them
**encrypted**, under a key that is not itself sitting next to the
ciphertext.
The end state in #355 — short-lived, scoped credentials minted per request
— removes the need to store anything durable at all. That is a larger
change gated on per-upstream minting support. This slice buys back
restart-survivability now without regressing to plaintext secrets at rest.
## Problem
`Orchestrator._tokens` (`bot_bottle/orchestrator/service.py:74-79`) is a
plain in-memory dict, deliberately never written to the registry DB:
> Held **in memory only** — never written to the registry DB — so the
> gateway can inject each bottle's upstream credential without secrets at
> rest. Lost on restart (re-launch re-registers them); the future
> SecretProvider (#355) replaces this with per-request minting.
The registry itself *is* durable (SQLite on a container-only volume), and
so is the gateway CA since #450 / `2cd44cf7`. The tokens are now the only
piece of gateway state that does not survive a restart, which makes the
failure mode both silent and confusing.
### Observed failure
Checking out a branch that touches `bot_bottle/**/*.py` changes
`source_hash()` (`bot_bottle/orchestrator/lifecycle.py:88-99`).
`MacosInfraService._source_current()`
(`bot_bottle/backend/macos_container/infra.py:159-169`) sees the mismatch
and `ensure_running()` force-removes and recreates the infra container
(`infra.py:198-208`). The registry rows survive on the DB volume; the CA
survives on its host bind-mount; `_tokens` comes back empty.
Every already-running bottle then fails closed, mid-session, on its next
outbound request:
- `/resolve` succeeds — the bottle is still `active` in
`orchestrator_bottles` and its policy blob is served intact, including
`- host: "api.anthropic.com"` with `auth_scheme: Bearer` /
`token_env: EGRESS_TOKEN_0`.
- `tokens_for()` returns `{}`, so the resolved env overlay has no
`EGRESS_TOKEN_0`.
- `decide()` (`bot_bottle/egress_addon_core.py:644-652`) blocks with
`egress: route for 'api.anthropic.com' declared auth but env var
'EGRESS_TOKEN_0' is unset` — an 89-byte `403` on every request.
Confirmed live on the macOS backend on 2026-07-21: two bottles running
since 20:29/20:30 were still registered `active` with valid policy after
the 23:05 infra recreation, and both took 89-byte `403`s from then on,
while a bottle launched *after* the recreation egressed normally. The
recovery today is to relaunch every affected bottle.
Note this is a re-attachment blocker distinct from #443/#445 and from #450
— the CA and the gateway address were both fine. It is specifically the
credential wipe.
## Goals / Success criteria
1. A bottle's egress auth tokens survive infra-container recreation: an
already-running bottle keeps egressing across a gateway restart with no
relaunch and no operator action.
2. Secrets are **never** at rest in plaintext, and never at rest next to a
key that trivially decrypts them.
3. Compromise of the registry DB file alone does not yield usable
upstream credentials.
4. The stored form is revocable and rotatable without relaunching bottles
that are not affected.
5. Reap/teardown destroys a bottle's stored secrets along with its
registry row (no ciphertext outliving its bottle).
6. Migration is transparent: existing bottles keep working, no manifest
changes required.
## Non-goals
- **Per-request minting** of short-lived scoped credentials. That is the
#355 end state; this PRD is explicitly the interim slice and should not
foreclose it.
- Generalizing `DeployKeyProvisioner` into the full `SecretProvider` ABC,
or the manifest-level `{ provider: <name> }` reference surface.
- User-extensible provider discovery (`~/.bot-bottle/contrib/<name>/`).
- Changing the `/resolve` contract's shape (it already carries `tokens`).
- Fixing the *trigger*`source_hash` churn on branch switch. Recreating
infra is legitimate; it just must not cost running bottles their
credentials. A separate guard that refuses recreation while bottles are
active is complementary and out of scope here.
## Design
> **TODO (didericis):** the encryption flow goes here — key custody, where
> the key material lives relative to the ciphertext, the wrap/unwrap path
> at register and at `/resolve`, and what an attacker who holds only the
> DB (or only the host, or only the infra container) can recover.
Constraints the design has to satisfy, for reference while drafting:
- The gateway's `PolicyResolver` needs the cleartext at request time, on
the data-plane path, so unwrap has to be cheap enough to sit in a
per-flow `/resolve` (or be cached in memory after first unwrap).
- The infra container is recreated routinely and unattended. Anything
requiring an interactive unlock on every recreation defeats the goal.
- The DB lives on a container-only volume that the host does not mount, so
host-side and guest-side components see different filesystems — that
asymmetry is available as a place to split custody.
- The agent must never be able to reach the key material. It is a separate
container with no control-plane token, which is the existing boundary.
## Open questions
- Where does the unwrap key live, and what recreates/re-derives it when the
infra container is rebuilt?
- Is the cleartext cached in memory after first unwrap, or unwrapped per
request? (Latency vs. exposure window.)
- What is the rotation story — re-wrap in place, or force re-registration?
- Does this land behind a flag, or replace `_tokens` outright?
@@ -1,178 +0,0 @@
# Firecracker Image Remote Store
**Date:** 2026-07-22
**Context:** PR #459 (run-dir leak fix) surfaced that committed Firecracker snapshots
currently live only on the host machine. Once a host is wiped or a run-dir is
evicted, a user's preserved bottle is gone. This note investigates a secure remote
store so committed images survive host turnover and can be restored without the
user having to pre-flag which sessions to keep.
## Verdict
Backblaze B2 + Cloudflare CDN is the cost-optimal choice for most deployments.
Cloudflare R2 is the simpler zero-config option at slightly higher storage cost.
Self-hosted MinIO is the right call for air-gapped or on-premises installs.
On the image-size front, zstd-compressing the committed tar before upload
produces roughly a 6070% reduction with negligible impact on restore latency.
OverlayFS (for in-flight working rootfs, not for the committed artifact) cuts
per-instance disk use to ~1050 MB per extra bottle sharing the same base.
---
## What Gets Stored
The Firecracker backend produces two artifact types:
| Artifact | Created by | Size | Lifetime |
|----------|-----------|------|---------|
| `rootfs/agent-<digest>/` (dir) | `image_builder.py``mke2fs -d` | ~1 GB as ext4 | Cached per Dockerfile hash; evictable |
| `committed/<slug>/rootfs.tar` | `FirecrackerFreezer._freeze` via SSH tar | 500 MB1 GB | User-preserved; must survive host wipe |
The committed artifact is a tar of the guest's live filesystem streamed out over
SSH (`freezer.py:5890`). At resume time `launch.py` calls `mke2fs -d` to
rebuild a fresh ext4 from this tar. The tar — not the ext4 — is what needs to be
pushed to remote storage and pulled back at restore time.
The base image cache (`rootfs/agent-<digest>/`) is derivable from the Dockerfile
and can be rebuilt on demand; it is lower priority for remote storage.
---
## Storage Candidates
### Object storage
| Provider | Storage | Egress | Notes |
|----------|---------|--------|-------|
| **Backblaze B2** | $0.006/GB | Free (via Cloudflare Bandwidth Alliance) | Cheapest storage; pairs with Cloudflare CDN to eliminate egress |
| **Cloudflare R2** | $0.015/GB | $0 always | Zero-config egress; no lifecycle transitions (limitation) |
| **Wasabi** | $0.0069/GB | Free (1:1 ratio) | 90-day minimum retention; good for archival; lifecycle evaluated daily |
| **AWS S3** | $0.023/GB | $0.09/GB | Richest lifecycle support; expensive at scale; avoid unless already in AWS |
| **MinIO** (self-hosted) | Host cost only | None | S3-compatible; best for private/on-prem deployments |
**B2 + Cloudflare CDN** is effectively $0.006/GB with zero egress — about 18×
cheaper than S3 for restore-heavy workloads. **R2** is the zero-config choice
($0 egress by default, no Bandwidth Alliance pairing needed) at a slightly
higher storage rate.
**Cloudflare R2's missing lifecycle support** is the main caveat: auto-eviction
rules (evict images older than N days) cannot currently be expressed natively in
R2. Wasabi and S3 both support declarative lifecycle policies.
### Retention policy recommendation
The comment proposes:
- Retain images for ~1 week by default
- Warn when approaching a capacity threshold
- Auto-evict oldest images once threshold is exceeded
This maps cleanly to an application-level policy (not a provider lifecycle rule),
which avoids the R2 limitation and works consistently across providers:
1. On `commit`: upload tar, record `(slug, size_bytes, uploaded_at)` in a local
or remote manifest file.
2. On startup / on `list`: scan the manifest, warn if total stored size exceeds
e.g. 80% of the configured threshold.
3. On eviction run (CLI or cron): delete objects older than `retention_days`
(default 7) that push total over `max_capacity`; oldest-first.
This keeps the policy logic in bot-bottle and the storage provider as a dumb
object store — no vendor-specific lifecycle API required.
---
## Image Size Reduction
### Current artifact sizes
A typical committed tar for a Claude Code agent image is 500 MB1 GB uncompressed.
The per-run ext4 (copy of the base, written at `start`) adds another ~1 GB of
local disk. The leak fix in this PR addresses the ext4 copies; the remote store
addresses the committed tars.
### Compression
zstd compression of the committed tar before upload is the highest-leverage
single change:
| Codec | Typical size (1 GB rootfs) | Compress speed | Decompress speed |
|-------|---------------------------|---------------|-----------------|
| gzip | 350430 MB | ~100 MB/s | ~500 MB/s |
| **zstd (default)** | **330360 MB** | **~400 MB/s** | **~2 GB/s** |
| xz | 290320 MB | ~20 MB/s | ~200 MB/s |
**zstd is the best trade-off**: 6567% size reduction, near-instantaneous
decompression. The `tar` call in `freezer.py` could pipe through `zstd` before
writing to disk and to the remote; `resume` decompresses on the way back. A
`.tar.zst` suffix marks compressed artifacts so old tars remain restorable
without the codec.
### SquashFS for the base image cache
The `rootfs/agent-<digest>/` directory (the buildah-exported tree) is rebuilt by
`image_builder.py` and turned into per-run ext4 by `mke2fs -d`. Storing the
base as a SquashFS image instead of a flat directory tree would reduce it from
~1 GB to ~330360 MB and make the cache remote-friendly. Firecracker does not
directly boot SquashFS, but the existing `mke2fs -d` path reads a directory tree
— a SquashFS mount could serve as the source. This is a larger change and lower
priority than tar compression.
### OverlayFS for per-run rootfs
Multiple simultaneous bottles sharing the same agent image today each get a full
`mke2fs -d` copy (~1 GB). OverlayFS (read-only base + writable sparse overlay)
would reduce this to ~1050 MB per instance beyond the first:
- Mount the base image directory as read-only lower layer
- Attach a sparse ext4 or tmpfs writable layer per bottle
- Pass the merged overlay to Firecracker as the block device
E2B's public write-up on Firecracker + OverlayFS confirms this approach works
at scale. The `launch.py` changes would be non-trivial (device mapper or
`fuse-overlayfs` plumbing), so this is a follow-up rather than a prerequisite
for the remote store.
---
## Recommended Approach
**Phase 1 — remote store with zstd (tight scope, actionable now)**
1. Add `--zstd` to the `tar` call in `FirecrackerFreezer._freeze`; name the
artifact `rootfs.tar.zst`. Keep uncompressed restore path for legacy tars.
2. Add a `bb firecracker upload <slug>` / `bb firecracker pull <slug>` pair that
pushes/fetches the compressed tar to the configured object store (S3-compatible
API, so B2, R2, MinIO, and Wasabi all work with the same client).
3. Store a `manifest.json` in the bucket (or a local mirror) tracking slug →
`{size, uploaded_at}`. Use it for threshold warnings and eviction.
4. Default retention: 7 days, configurable via `firecracker.image_retention_days`
in `~/.config/bot-bottle/config.toml` (or equivalent).
5. Warn at 80% of `max_capacity` (default e.g. 50 GB); evict oldest on commit
once at 100%.
**Storage recommendation:** Cloudflare R2 for hosted deployments (zero egress,
zero config), MinIO for private/on-premises.
**Phase 2 — base image cache compression**
Compress the `agent-<digest>` cache dir as a `.tar.zst` to save ~65% on repeated
image uploads. Low urgency since the base image is rebuildable.
**Phase 3 — OverlayFS per-run disk**
Replace the full per-run ext4 copy with an OverlayFS sparse layer. Largest disk
impact (~9095% savings per concurrent bottle) but highest implementation
complexity. Track as a separate PRD.
---
## Open Questions
- Does the host have a configured object-store credential path, or should the
remote store be an opt-in with an explicit `bb config set image-store.url ...`?
- Should `commit` automatically upload, or should upload be an explicit step to
avoid surprise egress?
- What is the acceptable cold-start latency for a restore from remote? A 330 MB
zstd tar at 100 Mbit/s takes ~26 s; at 1 Gbit/s, ~2.6 s. This bounds the
retention strategy (evict from local after successful upload vs keep local copy).
+21 -60
View File
@@ -10,9 +10,7 @@ classmethods forward to their module.
from __future__ import annotations from __future__ import annotations
import subprocess import subprocess
import tempfile
import unittest import unittest
from pathlib import Path
from unittest.mock import patch from unittest.mock import patch
from bot_bottle.backend.firecracker import cleanup as fc_cleanup from bot_bottle.backend.firecracker import cleanup as fc_cleanup
@@ -25,69 +23,32 @@ def _proc(stdout: str = "", returncode: int = 0) -> "subprocess.CompletedProcess
return subprocess.CompletedProcess([], returncode, stdout=stdout, stderr="") return subprocess.CompletedProcess([], returncode, stdout=stdout, stderr="")
class TestProcessScan(unittest.TestCase): class TestOrphanEnumeration(unittest.TestCase):
def test_run_dir_of_matches_only_direct_children(self): def test_orphan_vm_pids_filters_by_run_dir(self):
run_root = Path("/cache/run") run_root = str(fc_cleanup._run_root())
self.assertEqual( out = (
Path("/cache/run/dev-a"), f"111 firecracker --config-file {run_root}/dev-a/config.json\n"
fc_cleanup._run_dir_of( "222 firecracker --config-file /somewhere/else/config.json\n"
f"firecracker --config-file {run_root}/dev-a/config.json", run_root "notanint firecracker --config-file " + run_root + "/x\n"
),
)
# infra/builder VMs elsewhere, or nested paths, are not ours.
self.assertIsNone(
fc_cleanup._run_dir_of("firecracker --config-file /elsewhere/config.json", run_root)
)
self.assertIsNone(
fc_cleanup._run_dir_of("firecracker --no-config", run_root)
) )
with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)):
self.assertEqual([111], fc_cleanup._orphan_vm_pids())
def test_scan_splits_live_dirs_from_orphan_pids(self): def test_orphan_vm_pids_empty_when_pgrep_fails(self):
with tempfile.TemporaryDirectory() as tmp:
run_root = Path(tmp)
(run_root / "live-a").mkdir() # dir present -> live VM, protected
# "gone-b" dir intentionally absent -> lingering VMM, orphan pid
out = (
f"111 firecracker --config-file {run_root}/live-a/config.json\n"
f"222 firecracker --config-file {run_root}/gone-b/config.json\n"
"333 firecracker --config-file /elsewhere/config.json\n"
"notanint firecracker --config-file x\n"
)
with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)):
live, orphan_pids = fc_cleanup._scan_processes(run_root)
self.assertEqual({str(run_root / "live-a")}, live)
self.assertEqual([222], orphan_pids)
def test_scan_empty_when_pgrep_fails(self):
with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(returncode=1)): with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(returncode=1)):
self.assertEqual((set(), []), fc_cleanup._scan_processes(Path("/x"))) self.assertEqual([], fc_cleanup._orphan_vm_pids())
def test_orphan_run_dirs_excludes_live_and_missing_root(self): def test_run_dirs_empty_when_absent(self):
with tempfile.TemporaryDirectory() as tmp: with patch.object(fc_cleanup.util, "cache_dir") as cache:
run_root = Path(tmp) cache.return_value.__truediv__.return_value.is_dir.return_value = False
(run_root / "live-a").mkdir() self.assertEqual([], fc_cleanup._run_dirs())
(run_root / "dead-b").mkdir()
live = {str(run_root / "live-a")}
self.assertEqual(
[str(run_root / "dead-b")],
fc_cleanup._orphan_run_dirs(run_root, live),
)
# absent run root -> nothing to reap
self.assertEqual([], fc_cleanup._orphan_run_dirs(Path("/nope/run"), set()))
def test_prepare_cleanup_reaps_orphans_only(self): def test_prepare_cleanup_assembles_plan(self):
"""The live VM's dir is never in the plan; the dead one is.""" with patch.object(fc_cleanup, "_orphan_vm_pids", return_value=[7]), \
with tempfile.TemporaryDirectory() as tmp: patch.object(fc_cleanup, "_run_dirs", return_value=["/run/x"]):
run_root = Path(tmp) plan = fc_cleanup.prepare_cleanup()
(run_root / "live-a").mkdir() self.assertEqual((7,), plan.vm_pids)
(run_root / "dead-b").mkdir() self.assertEqual(("/run/x",), plan.run_dirs)
out = f"111 firecracker --config-file {run_root}/live-a/config.json\n"
with patch.object(fc_cleanup, "_run_root", return_value=run_root), \
patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)):
plan = fc_cleanup.prepare_cleanup()
self.assertEqual((), plan.vm_pids)
self.assertEqual((str(run_root / "dead-b"),), plan.run_dirs)
self.assertNotIn(str(run_root / "live-a"), plan.run_dirs)
class TestCleanupRemoval(unittest.TestCase): class TestCleanupRemoval(unittest.TestCase):
-52
View File
@@ -173,58 +173,6 @@ if __name__ == "__main__":
unittest.main() unittest.main()
class TestAgentSecrets(unittest.TestCase):
"""store/get/delete for the bottled_agent_secrets table."""
def setUp(self) -> None:
self._tmp = tempfile.TemporaryDirectory()
self.db = Path(self._tmp.name) / "registry.db"
self.store = RegistryStore(self.db)
self.store.migrate()
def tearDown(self) -> None:
self._tmp.cleanup()
def test_store_and_get_roundtrip(self) -> None:
self.store.store_agent_secrets("bottle-1", {"EGRESS_TOKEN_1": "enc-val-a"})
got = self.store.get_agent_secrets("bottle-1")
self.assertEqual({"EGRESS_TOKEN_1": "enc-val-a"}, got)
def test_get_returns_empty_when_none_stored(self) -> None:
self.assertEqual({}, self.store.get_agent_secrets("no-such-bottle"))
def test_store_replaces_existing_rows(self) -> None:
self.store.store_agent_secrets("bottle-1", {"K": "old"})
self.store.store_agent_secrets("bottle-1", {"K": "new", "K2": "v2"})
got = self.store.get_agent_secrets("bottle-1")
self.assertEqual({"K": "new", "K2": "v2"}, got)
def test_delete_removes_secrets(self) -> None:
self.store.store_agent_secrets("bottle-1", {"K": "v"})
self.store.delete_agent_secrets("bottle-1")
self.assertEqual({}, self.store.get_agent_secrets("bottle-1"))
def test_delete_is_idempotent_on_missing(self) -> None:
self.store.delete_agent_secrets("no-such-bottle") # must not raise
def test_secrets_are_isolated_by_bottle_id(self) -> None:
self.store.store_agent_secrets("bottle-1", {"K": "for-1"})
self.store.store_agent_secrets("bottle-2", {"K": "for-2"})
self.assertEqual({"K": "for-1"}, self.store.get_agent_secrets("bottle-1"))
self.assertEqual({"K": "for-2"}, self.store.get_agent_secrets("bottle-2"))
def test_secrets_isolated_by_type(self) -> None:
self.store.store_agent_secrets("bottle-1", {"K": "injected"}, secret_type="injected_env_var")
self.store.store_agent_secrets("bottle-1", {"K": "other"}, secret_type="other_type")
self.assertEqual({"K": "injected"}, self.store.get_agent_secrets("bottle-1"))
self.assertEqual({"K": "other"}, self.store.get_agent_secrets("bottle-1", secret_type="other_type"))
def test_secrets_persist_across_reopen(self) -> None:
self.store.store_agent_secrets("bottle-1", {"K": "v"})
reopened = RegistryStore(self.db)
self.assertEqual({"K": "v"}, reopened.get_agent_secrets("bottle-1"))
class TestReapAbsent(unittest.TestCase): class TestReapAbsent(unittest.TestCase):
"""`reap_absent` — the self-heal for rows whose bottle is gone. """`reap_absent` — the self-heal for rows whose bottle is gone.
@@ -1,96 +0,0 @@
"""Unit tests for per-bottle egress secret encryption (PRD prd-new-secret-provider)."""
from __future__ import annotations
import unittest
from bot_bottle.orchestrator.secret_store import (
ENV_VAR_SECRET_NAME,
decrypt_value,
encrypt_value,
new_env_var_secret,
)
class TestNewEnvVarSecret(unittest.TestCase):
def test_returns_non_empty_string(self) -> None:
s = new_env_var_secret()
self.assertIsInstance(s, str)
self.assertTrue(len(s) > 0)
def test_secrets_are_unique(self) -> None:
keys = {new_env_var_secret() for _ in range(50)}
self.assertEqual(50, len(keys))
def test_no_padding_characters(self) -> None:
# URL-safe base64, padding stripped — should round-trip cleanly
for _ in range(20):
self.assertNotIn("=", new_env_var_secret())
class TestEncryptDecryptRoundtrip(unittest.TestCase):
def setUp(self) -> None:
self.secret = new_env_var_secret()
def _rt(self, plaintext: str) -> str:
return decrypt_value(self.secret, encrypt_value(self.secret, plaintext))
def test_roundtrip_short_value(self) -> None:
self.assertEqual("sk-abc123", self._rt("sk-abc123"))
def test_roundtrip_empty_string(self) -> None:
self.assertEqual("", self._rt(""))
def test_roundtrip_long_value_crosses_block_boundary(self) -> None:
# 32 bytes is exactly one HMAC-SHA256 block; 65 bytes crosses two.
plaintext = "x" * 65
self.assertEqual(plaintext, self._rt(plaintext))
def test_roundtrip_unicode(self) -> None:
self.assertEqual("héllo wörld", self._rt("héllo wörld"))
def test_encrypt_produces_different_ciphertexts_each_call(self) -> None:
ct1 = encrypt_value(self.secret, "same")
ct2 = encrypt_value(self.secret, "same")
self.assertNotEqual(ct1, ct2) # fresh nonce each call
def test_ciphertext_is_url_safe_base64(self) -> None:
ct = encrypt_value(self.secret, "hello")
# no '+', '/', '=' — URL-safe and padding-stripped
for ch in ("+", "/", "="):
self.assertNotIn(ch, ct)
class TestDecryptErrors(unittest.TestCase):
def setUp(self) -> None:
self.secret = new_env_var_secret()
def test_wrong_key_raises_value_error(self) -> None:
ct = encrypt_value(self.secret, "secret-token")
other_key = new_env_var_secret()
# Wrong key produces garbage bytes; decrypt_value raises ValueError
# when the result is non-UTF-8 (which is very likely for 12-char data).
# We allow it to succeed only if garbage happens to be valid UTF-8, but
# the plaintext must not match.
try:
result = decrypt_value(other_key, ct)
self.assertNotEqual("secret-token", result)
except ValueError:
pass
def test_truncated_blob_raises_value_error(self) -> None:
with self.assertRaises(ValueError):
decrypt_value(self.secret, "dG9vc2hvcnQ") # "tooshort" — under 16 nonce bytes
def test_invalid_base64_raises_value_error(self) -> None:
with self.assertRaises(ValueError):
decrypt_value(self.secret, "!!not-base64!!")
class TestConstant(unittest.TestCase):
def test_env_var_secret_name(self) -> None:
self.assertEqual("ENV_VAR_SECRET", ENV_VAR_SECRET_NAME)
if __name__ == "__main__":
unittest.main()