diff --git a/bot_bottle/backend/firecracker/cleanup.py b/bot_bottle/backend/firecracker/cleanup.py index af41e29..6e9fe2d 100644 --- a/bot_bottle/backend/firecracker/cleanup.py +++ b/bot_bottle/backend/firecracker/cleanup.py @@ -1,8 +1,23 @@ """Cleanup for the Firecracker backend. -Orphans are: firecracker VMM processes whose config lives under our run -dir, and the per-bottle run dirs. TAP slots free themselves (the flock -drops when the launcher exits), so there is nothing to reclaim there. +Reaps *orphans* only — resources with no live VM behind them: + + * orphan run dirs: a per-bottle run dir (holding the ~1G rootfs.ext4) + whose firecracker process has exited. These leak when a launch is + hard-killed before its teardown runs (host OOM/crash, a cancelled CI + job, `kill -9`); the clean-exit path already removes its own dir in + launch.py. + * orphan VM pids: a firecracker process whose run dir is already gone + — a VMM left lingering after its dir was removed. + +A run dir with a *live* firecracker process is a running bottle and is +left strictly alone: it is neither killed nor removed. (The backend's +`enumerate_active` registry is still a stub — #354 — so a live process +is the only reliable "this bottle is in use" signal we have. Once the +registry lands, registry-orphaned-but-running VMs can be reaped too.) + +TAP slots free themselves (the flock drops when the launcher exits), so +there is nothing to reclaim there. """ from __future__ import annotations @@ -22,38 +37,73 @@ def _run_root() -> Path: return util.cache_dir() / "run" -def _orphan_vm_pids() -> list[int]: - """firecracker processes whose --config-file is under our run dir.""" - run_root = str(_run_root()) +def _run_dir_of(cmd: str, run_root: Path) -> Path | None: + """The bottle run dir a firecracker cmdline belongs to, or None. + + A bottle VM is launched with `--config-file //config.json`, + so the run dir is the config file's parent when it sits directly under + the run root. Anything else (a builder VM, the infra VM elsewhere) is + not ours to reap here. + """ + toks = cmd.split() + for i, tok in enumerate(toks): + if tok == "--config-file" and i + 1 < len(toks): + parent = Path(toks[i + 1]).parent + if parent.parent == run_root: + return parent + return None + + +def _scan_processes(run_root: Path) -> tuple[set[str], list[int]]: + """Inspect running firecracker VMs under ``run_root``. + + Returns ``(live_run_dirs, orphan_pids)``: + * ``live_run_dirs`` — run dirs backed by a running VM (never reaped); + * ``orphan_pids`` — firecracker pids whose run dir no longer exists + (a lingering VMM to kill). + """ result = subprocess.run( ["pgrep", "-a", "firecracker"], capture_output=True, text=True, check=False, ) if result.returncode != 0: - return [] - pids: list[int] = [] + return set(), [] + live: set[str] = set() + orphan_pids: list[int] = [] for line in result.stdout.splitlines(): parts = line.split(None, 1) - if len(parts) != 2 or run_root not in parts[1]: + if len(parts) != 2: continue try: - pids.append(int(parts[0])) + pid = int(parts[0]) except ValueError: continue - return pids + run_dir = _run_dir_of(parts[1], run_root) + if run_dir is None: + continue + if run_dir.is_dir(): + live.add(str(run_dir)) + else: + orphan_pids.append(pid) + return live, orphan_pids -def _run_dirs() -> list[str]: - run_root = _run_root() +def _orphan_run_dirs(run_root: Path, live: set[str]) -> list[str]: + """Run dirs with no live VM behind them — the leaked ones to remove.""" if not run_root.is_dir(): return [] - return sorted(str(p) for p in run_root.iterdir() if p.is_dir()) + return sorted( + str(p) for p in run_root.iterdir() + if p.is_dir() and str(p) not in live + ) def prepare_cleanup() -> FirecrackerBottleCleanupPlan: + run_root = _run_root() + live, orphan_pids = _scan_processes(run_root) return FirecrackerBottleCleanupPlan( - vm_pids=tuple(_orphan_vm_pids()), - run_dirs=tuple(_run_dirs()), + vm_pids=tuple(orphan_pids), + run_dirs=tuple(_orphan_run_dirs(run_root, live)), ) diff --git a/bot_bottle/backend/firecracker/launch.py b/bot_bottle/backend/firecracker/launch.py index 1880b7f..a76f9ef 100644 --- a/bot_bottle/backend/firecracker/launch.py +++ b/bot_bottle/backend/firecracker/launch.py @@ -26,6 +26,7 @@ from __future__ import annotations import dataclasses import os +import shutil from contextlib import ExitStack, contextmanager from pathlib import Path from typing import Callable, Generator @@ -167,6 +168,10 @@ def launch( # Step 6: build the per-bottle rootfs + SSH key, then boot. run_dir = util.cache_dir() / "run" / plan.slug run_dir.mkdir(parents=True, exist_ok=True) + # Remove the run dir on teardown so the per-bottle rootfs.ext4 (~1G) + # doesn't leak. Registered before vm.terminate below so it runs *after* + # it (ExitStack is LIFO): the VM is gone before we rm its rootfs. + stack.callback(lambda: shutil.rmtree(run_dir, ignore_errors=True)) rootfs = run_dir / "rootfs.ext4" util.build_rootfs_ext4(agent_base, rootfs) private_key, pubkey = util.generate_keypair(run_dir) diff --git a/docs/research/firecracker-image-remote-store.md b/docs/research/firecracker-image-remote-store.md new file mode 100644 index 0000000..f09750c --- /dev/null +++ b/docs/research/firecracker-image-remote-store.md @@ -0,0 +1,178 @@ +# Firecracker Image Remote Store + +**Date:** 2026-07-22 +**Context:** PR #459 (run-dir leak fix) surfaced that committed Firecracker snapshots +currently live only on the host machine. Once a host is wiped or a run-dir is +evicted, a user's preserved bottle is gone. This note investigates a secure remote +store so committed images survive host turnover and can be restored without the +user having to pre-flag which sessions to keep. + +## Verdict + +Backblaze B2 + Cloudflare CDN is the cost-optimal choice for most deployments. +Cloudflare R2 is the simpler zero-config option at slightly higher storage cost. +Self-hosted MinIO is the right call for air-gapped or on-premises installs. + +On the image-size front, zstd-compressing the committed tar before upload +produces roughly a 60–70% reduction with negligible impact on restore latency. +OverlayFS (for in-flight working rootfs, not for the committed artifact) cuts +per-instance disk use to ~10–50 MB per extra bottle sharing the same base. + +--- + +## What Gets Stored + +The Firecracker backend produces two artifact types: + +| Artifact | Created by | Size | Lifetime | +|----------|-----------|------|---------| +| `rootfs/agent-/` (dir) | `image_builder.py` → `mke2fs -d` | ~1 GB as ext4 | Cached per Dockerfile hash; evictable | +| `committed//rootfs.tar` | `FirecrackerFreezer._freeze` via SSH tar | 500 MB–1 GB | User-preserved; must survive host wipe | + +The committed artifact is a tar of the guest's live filesystem streamed out over +SSH (`freezer.py:58–90`). At resume time `launch.py` calls `mke2fs -d` to +rebuild a fresh ext4 from this tar. The tar — not the ext4 — is what needs to be +pushed to remote storage and pulled back at restore time. + +The base image cache (`rootfs/agent-/`) is derivable from the Dockerfile +and can be rebuilt on demand; it is lower priority for remote storage. + +--- + +## Storage Candidates + +### Object storage + +| Provider | Storage | Egress | Notes | +|----------|---------|--------|-------| +| **Backblaze B2** | $0.006/GB | Free (via Cloudflare Bandwidth Alliance) | Cheapest storage; pairs with Cloudflare CDN to eliminate egress | +| **Cloudflare R2** | $0.015/GB | $0 always | Zero-config egress; no lifecycle transitions (limitation) | +| **Wasabi** | $0.0069/GB | Free (1:1 ratio) | 90-day minimum retention; good for archival; lifecycle evaluated daily | +| **AWS S3** | $0.023/GB | $0.09/GB | Richest lifecycle support; expensive at scale; avoid unless already in AWS | +| **MinIO** (self-hosted) | Host cost only | None | S3-compatible; best for private/on-prem deployments | + +**B2 + Cloudflare CDN** is effectively $0.006/GB with zero egress — about 18× +cheaper than S3 for restore-heavy workloads. **R2** is the zero-config choice +($0 egress by default, no Bandwidth Alliance pairing needed) at a slightly +higher storage rate. + +**Cloudflare R2's missing lifecycle support** is the main caveat: auto-eviction +rules (evict images older than N days) cannot currently be expressed natively in +R2. Wasabi and S3 both support declarative lifecycle policies. + +### Retention policy recommendation + +The comment proposes: +- Retain images for ~1 week by default +- Warn when approaching a capacity threshold +- Auto-evict oldest images once threshold is exceeded + +This maps cleanly to an application-level policy (not a provider lifecycle rule), +which avoids the R2 limitation and works consistently across providers: + +1. On `commit`: upload tar, record `(slug, size_bytes, uploaded_at)` in a local + or remote manifest file. +2. On startup / on `list`: scan the manifest, warn if total stored size exceeds + e.g. 80% of the configured threshold. +3. On eviction run (CLI or cron): delete objects older than `retention_days` + (default 7) that push total over `max_capacity`; oldest-first. + +This keeps the policy logic in bot-bottle and the storage provider as a dumb +object store — no vendor-specific lifecycle API required. + +--- + +## Image Size Reduction + +### Current artifact sizes + +A typical committed tar for a Claude Code agent image is 500 MB–1 GB uncompressed. +The per-run ext4 (copy of the base, written at `start`) adds another ~1 GB of +local disk. The leak fix in this PR addresses the ext4 copies; the remote store +addresses the committed tars. + +### Compression + +zstd compression of the committed tar before upload is the highest-leverage +single change: + +| Codec | Typical size (1 GB rootfs) | Compress speed | Decompress speed | +|-------|---------------------------|---------------|-----------------| +| gzip | 350–430 MB | ~100 MB/s | ~500 MB/s | +| **zstd (default)** | **330–360 MB** | **~400 MB/s** | **~2 GB/s** | +| xz | 290–320 MB | ~20 MB/s | ~200 MB/s | + +**zstd is the best trade-off**: 65–67% size reduction, near-instantaneous +decompression. The `tar` call in `freezer.py` could pipe through `zstd` before +writing to disk and to the remote; `resume` decompresses on the way back. A +`.tar.zst` suffix marks compressed artifacts so old tars remain restorable +without the codec. + +### SquashFS for the base image cache + +The `rootfs/agent-/` directory (the buildah-exported tree) is rebuilt by +`image_builder.py` and turned into per-run ext4 by `mke2fs -d`. Storing the +base as a SquashFS image instead of a flat directory tree would reduce it from +~1 GB to ~330–360 MB and make the cache remote-friendly. Firecracker does not +directly boot SquashFS, but the existing `mke2fs -d` path reads a directory tree +— a SquashFS mount could serve as the source. This is a larger change and lower +priority than tar compression. + +### OverlayFS for per-run rootfs + +Multiple simultaneous bottles sharing the same agent image today each get a full +`mke2fs -d` copy (~1 GB). OverlayFS (read-only base + writable sparse overlay) +would reduce this to ~10–50 MB per instance beyond the first: + +- Mount the base image directory as read-only lower layer +- Attach a sparse ext4 or tmpfs writable layer per bottle +- Pass the merged overlay to Firecracker as the block device + +E2B's public write-up on Firecracker + OverlayFS confirms this approach works +at scale. The `launch.py` changes would be non-trivial (device mapper or +`fuse-overlayfs` plumbing), so this is a follow-up rather than a prerequisite +for the remote store. + +--- + +## Recommended Approach + +**Phase 1 — remote store with zstd (tight scope, actionable now)** + +1. Add `--zstd` to the `tar` call in `FirecrackerFreezer._freeze`; name the + artifact `rootfs.tar.zst`. Keep uncompressed restore path for legacy tars. +2. Add a `bb firecracker upload ` / `bb firecracker pull ` pair that + pushes/fetches the compressed tar to the configured object store (S3-compatible + API, so B2, R2, MinIO, and Wasabi all work with the same client). +3. Store a `manifest.json` in the bucket (or a local mirror) tracking slug → + `{size, uploaded_at}`. Use it for threshold warnings and eviction. +4. Default retention: 7 days, configurable via `firecracker.image_retention_days` + in `~/.config/bot-bottle/config.toml` (or equivalent). +5. Warn at 80% of `max_capacity` (default e.g. 50 GB); evict oldest on commit + once at 100%. + +**Storage recommendation:** Cloudflare R2 for hosted deployments (zero egress, +zero config), MinIO for private/on-premises. + +**Phase 2 — base image cache compression** + +Compress the `agent-` cache dir as a `.tar.zst` to save ~65% on repeated +image uploads. Low urgency since the base image is rebuildable. + +**Phase 3 — OverlayFS per-run disk** + +Replace the full per-run ext4 copy with an OverlayFS sparse layer. Largest disk +impact (~90–95% savings per concurrent bottle) but highest implementation +complexity. Track as a separate PRD. + +--- + +## Open Questions + +- Does the host have a configured object-store credential path, or should the + remote store be an opt-in with an explicit `bb config set image-store.url ...`? +- Should `commit` automatically upload, or should upload be an explicit step to + avoid surprise egress? +- What is the acceptable cold-start latency for a restore from remote? A 330 MB + zstd tar at 100 Mbit/s takes ~26 s; at 1 Gbit/s, ~2.6 s. This bounds the + retention strategy (evict from local after successful upload vs keep local copy). diff --git a/tests/unit/test_firecracker_cleanup.py b/tests/unit/test_firecracker_cleanup.py index 27b0097..8670783 100644 --- a/tests/unit/test_firecracker_cleanup.py +++ b/tests/unit/test_firecracker_cleanup.py @@ -10,7 +10,9 @@ classmethods forward to their module. from __future__ import annotations import subprocess +import tempfile import unittest +from pathlib import Path from unittest.mock import patch from bot_bottle.backend.firecracker import cleanup as fc_cleanup @@ -23,32 +25,69 @@ def _proc(stdout: str = "", returncode: int = 0) -> "subprocess.CompletedProcess return subprocess.CompletedProcess([], returncode, stdout=stdout, stderr="") -class TestOrphanEnumeration(unittest.TestCase): - def test_orphan_vm_pids_filters_by_run_dir(self): - run_root = str(fc_cleanup._run_root()) - out = ( - f"111 firecracker --config-file {run_root}/dev-a/config.json\n" - "222 firecracker --config-file /somewhere/else/config.json\n" - "notanint firecracker --config-file " + run_root + "/x\n" +class TestProcessScan(unittest.TestCase): + def test_run_dir_of_matches_only_direct_children(self): + run_root = Path("/cache/run") + self.assertEqual( + Path("/cache/run/dev-a"), + fc_cleanup._run_dir_of( + f"firecracker --config-file {run_root}/dev-a/config.json", run_root + ), + ) + # infra/builder VMs elsewhere, or nested paths, are not ours. + self.assertIsNone( + fc_cleanup._run_dir_of("firecracker --config-file /elsewhere/config.json", run_root) + ) + self.assertIsNone( + fc_cleanup._run_dir_of("firecracker --no-config", run_root) ) - with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)): - self.assertEqual([111], fc_cleanup._orphan_vm_pids()) - def test_orphan_vm_pids_empty_when_pgrep_fails(self): + def test_scan_splits_live_dirs_from_orphan_pids(self): + with tempfile.TemporaryDirectory() as tmp: + run_root = Path(tmp) + (run_root / "live-a").mkdir() # dir present -> live VM, protected + # "gone-b" dir intentionally absent -> lingering VMM, orphan pid + out = ( + f"111 firecracker --config-file {run_root}/live-a/config.json\n" + f"222 firecracker --config-file {run_root}/gone-b/config.json\n" + "333 firecracker --config-file /elsewhere/config.json\n" + "notanint firecracker --config-file x\n" + ) + with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)): + live, orphan_pids = fc_cleanup._scan_processes(run_root) + self.assertEqual({str(run_root / "live-a")}, live) + self.assertEqual([222], orphan_pids) + + def test_scan_empty_when_pgrep_fails(self): with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(returncode=1)): - self.assertEqual([], fc_cleanup._orphan_vm_pids()) + self.assertEqual((set(), []), fc_cleanup._scan_processes(Path("/x"))) - def test_run_dirs_empty_when_absent(self): - with patch.object(fc_cleanup.util, "cache_dir") as cache: - cache.return_value.__truediv__.return_value.is_dir.return_value = False - self.assertEqual([], fc_cleanup._run_dirs()) + def test_orphan_run_dirs_excludes_live_and_missing_root(self): + with tempfile.TemporaryDirectory() as tmp: + run_root = Path(tmp) + (run_root / "live-a").mkdir() + (run_root / "dead-b").mkdir() + live = {str(run_root / "live-a")} + self.assertEqual( + [str(run_root / "dead-b")], + fc_cleanup._orphan_run_dirs(run_root, live), + ) + # absent run root -> nothing to reap + self.assertEqual([], fc_cleanup._orphan_run_dirs(Path("/nope/run"), set())) - def test_prepare_cleanup_assembles_plan(self): - with patch.object(fc_cleanup, "_orphan_vm_pids", return_value=[7]), \ - patch.object(fc_cleanup, "_run_dirs", return_value=["/run/x"]): - plan = fc_cleanup.prepare_cleanup() - self.assertEqual((7,), plan.vm_pids) - self.assertEqual(("/run/x",), plan.run_dirs) + def test_prepare_cleanup_reaps_orphans_only(self): + """The live VM's dir is never in the plan; the dead one is.""" + with tempfile.TemporaryDirectory() as tmp: + run_root = Path(tmp) + (run_root / "live-a").mkdir() + (run_root / "dead-b").mkdir() + out = f"111 firecracker --config-file {run_root}/live-a/config.json\n" + with patch.object(fc_cleanup, "_run_root", return_value=run_root), \ + patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)): + plan = fc_cleanup.prepare_cleanup() + self.assertEqual((), plan.vm_pids) + self.assertEqual((str(run_root / "dead-b"),), plan.run_dirs) + self.assertNotIn(str(run_root / "live-a"), plan.run_dirs) class TestCleanupRemoval(unittest.TestCase):