fix(firecracker): stop leaking per-bottle run dirs #459

Merged
didericis merged 3 commits from fix/firecracker-run-dir-leak into main 2026-07-21 22:54:09 -04:00
4 changed files with 309 additions and 37 deletions
+66 -16
View File
@@ -1,8 +1,23 @@
"""Cleanup for the Firecracker backend.
Orphans are: firecracker VMM processes whose config lives under our run
dir, and the per-bottle run dirs. TAP slots free themselves (the flock
drops when the launcher exits), so there is nothing to reclaim there.
Reaps *orphans* only — resources with no live VM behind them:
* orphan run dirs: a per-bottle run dir (holding the ~1G rootfs.ext4)
whose firecracker process has exited. These leak when a launch is
hard-killed before its teardown runs (host OOM/crash, a cancelled CI
job, `kill -9`); the clean-exit path already removes its own dir in
launch.py.
* orphan VM pids: a firecracker process whose run dir is already gone
— a VMM left lingering after its dir was removed.
A run dir with a *live* firecracker process is a running bottle and is
left strictly alone: it is neither killed nor removed. (The backend's
`enumerate_active` registry is still a stub — #354 — so a live process
is the only reliable "this bottle is in use" signal we have. Once the
registry lands, registry-orphaned-but-running VMs can be reaped too.)
TAP slots free themselves (the flock drops when the launcher exits), so
there is nothing to reclaim there.
"""
from __future__ import annotations
@@ -22,38 +37,73 @@ def _run_root() -> Path:
return util.cache_dir() / "run"
def _orphan_vm_pids() -> list[int]:
"""firecracker processes whose --config-file is under our run dir."""
run_root = str(_run_root())
def _run_dir_of(cmd: str, run_root: Path) -> Path | None:
"""The bottle run dir a firecracker cmdline belongs to, or None.
A bottle VM is launched with `--config-file <run_root>/<slug>/config.json`,
so the run dir is the config file's parent when it sits directly under
the run root. Anything else (a builder VM, the infra VM elsewhere) is
not ours to reap here.
"""
toks = cmd.split()
for i, tok in enumerate(toks):
if tok == "--config-file" and i + 1 < len(toks):
parent = Path(toks[i + 1]).parent
if parent.parent == run_root:
return parent
return None
def _scan_processes(run_root: Path) -> tuple[set[str], list[int]]:
"""Inspect running firecracker VMs under ``run_root``.
Returns ``(live_run_dirs, orphan_pids)``:
* ``live_run_dirs`` — run dirs backed by a running VM (never reaped);
* ``orphan_pids`` — firecracker pids whose run dir no longer exists
(a lingering VMM to kill).
"""
result = subprocess.run(
["pgrep", "-a", "firecracker"],
capture_output=True, text=True, check=False,
)
if result.returncode != 0:
return []
pids: list[int] = []
return set(), []
live: set[str] = set()
orphan_pids: list[int] = []
for line in result.stdout.splitlines():
parts = line.split(None, 1)
if len(parts) != 2 or run_root not in parts[1]:
if len(parts) != 2:
continue
try:
pids.append(int(parts[0]))
pid = int(parts[0])
except ValueError:
continue
return pids
run_dir = _run_dir_of(parts[1], run_root)
if run_dir is None:
continue
if run_dir.is_dir():
live.add(str(run_dir))
else:
orphan_pids.append(pid)
return live, orphan_pids
def _run_dirs() -> list[str]:
run_root = _run_root()
def _orphan_run_dirs(run_root: Path, live: set[str]) -> list[str]:
"""Run dirs with no live VM behind them — the leaked ones to remove."""
if not run_root.is_dir():
return []
return sorted(str(p) for p in run_root.iterdir() if p.is_dir())
return sorted(
str(p) for p in run_root.iterdir()
if p.is_dir() and str(p) not in live
)
def prepare_cleanup() -> FirecrackerBottleCleanupPlan:
run_root = _run_root()
live, orphan_pids = _scan_processes(run_root)
return FirecrackerBottleCleanupPlan(
vm_pids=tuple(_orphan_vm_pids()),
run_dirs=tuple(_run_dirs()),
vm_pids=tuple(orphan_pids),
run_dirs=tuple(_orphan_run_dirs(run_root, live)),
)
+5
View File
@@ -26,6 +26,7 @@ from __future__ import annotations
import dataclasses
import os
import shutil
from contextlib import ExitStack, contextmanager
from pathlib import Path
from typing import Callable, Generator
@@ -167,6 +168,10 @@ def launch(
# Step 6: build the per-bottle rootfs + SSH key, then boot.
run_dir = util.cache_dir() / "run" / plan.slug
run_dir.mkdir(parents=True, exist_ok=True)
# Remove the run dir on teardown so the per-bottle rootfs.ext4 (~1G)
# doesn't leak. Registered before vm.terminate below so it runs *after*
# it (ExitStack is LIFO): the VM is gone before we rm its rootfs.
stack.callback(lambda: shutil.rmtree(run_dir, ignore_errors=True))
rootfs = run_dir / "rootfs.ext4"
util.build_rootfs_ext4(agent_base, rootfs)
private_key, pubkey = util.generate_keypair(run_dir)
@@ -0,0 +1,178 @@
# Firecracker Image Remote Store
**Date:** 2026-07-22
**Context:** PR #459 (run-dir leak fix) surfaced that committed Firecracker snapshots
currently live only on the host machine. Once a host is wiped or a run-dir is
evicted, a user's preserved bottle is gone. This note investigates a secure remote
store so committed images survive host turnover and can be restored without the
user having to pre-flag which sessions to keep.
## Verdict
Backblaze B2 + Cloudflare CDN is the cost-optimal choice for most deployments.
Cloudflare R2 is the simpler zero-config option at slightly higher storage cost.
Self-hosted MinIO is the right call for air-gapped or on-premises installs.
On the image-size front, zstd-compressing the committed tar before upload
produces roughly a 6070% reduction with negligible impact on restore latency.
OverlayFS (for in-flight working rootfs, not for the committed artifact) cuts
per-instance disk use to ~1050 MB per extra bottle sharing the same base.
---
## What Gets Stored
The Firecracker backend produces two artifact types:
| Artifact | Created by | Size | Lifetime |
|----------|-----------|------|---------|
| `rootfs/agent-<digest>/` (dir) | `image_builder.py``mke2fs -d` | ~1 GB as ext4 | Cached per Dockerfile hash; evictable |
| `committed/<slug>/rootfs.tar` | `FirecrackerFreezer._freeze` via SSH tar | 500 MB1 GB | User-preserved; must survive host wipe |
The committed artifact is a tar of the guest's live filesystem streamed out over
SSH (`freezer.py:5890`). At resume time `launch.py` calls `mke2fs -d` to
rebuild a fresh ext4 from this tar. The tar — not the ext4 — is what needs to be
pushed to remote storage and pulled back at restore time.
The base image cache (`rootfs/agent-<digest>/`) is derivable from the Dockerfile
and can be rebuilt on demand; it is lower priority for remote storage.
---
## Storage Candidates
### Object storage
| Provider | Storage | Egress | Notes |
|----------|---------|--------|-------|
| **Backblaze B2** | $0.006/GB | Free (via Cloudflare Bandwidth Alliance) | Cheapest storage; pairs with Cloudflare CDN to eliminate egress |
| **Cloudflare R2** | $0.015/GB | $0 always | Zero-config egress; no lifecycle transitions (limitation) |
| **Wasabi** | $0.0069/GB | Free (1:1 ratio) | 90-day minimum retention; good for archival; lifecycle evaluated daily |
| **AWS S3** | $0.023/GB | $0.09/GB | Richest lifecycle support; expensive at scale; avoid unless already in AWS |
| **MinIO** (self-hosted) | Host cost only | None | S3-compatible; best for private/on-prem deployments |
**B2 + Cloudflare CDN** is effectively $0.006/GB with zero egress — about 18×
cheaper than S3 for restore-heavy workloads. **R2** is the zero-config choice
($0 egress by default, no Bandwidth Alliance pairing needed) at a slightly
higher storage rate.
**Cloudflare R2's missing lifecycle support** is the main caveat: auto-eviction
rules (evict images older than N days) cannot currently be expressed natively in
R2. Wasabi and S3 both support declarative lifecycle policies.
### Retention policy recommendation
The comment proposes:
- Retain images for ~1 week by default
- Warn when approaching a capacity threshold
- Auto-evict oldest images once threshold is exceeded
This maps cleanly to an application-level policy (not a provider lifecycle rule),
which avoids the R2 limitation and works consistently across providers:
1. On `commit`: upload tar, record `(slug, size_bytes, uploaded_at)` in a local
or remote manifest file.
2. On startup / on `list`: scan the manifest, warn if total stored size exceeds
e.g. 80% of the configured threshold.
3. On eviction run (CLI or cron): delete objects older than `retention_days`
(default 7) that push total over `max_capacity`; oldest-first.
This keeps the policy logic in bot-bottle and the storage provider as a dumb
object store — no vendor-specific lifecycle API required.
---
## Image Size Reduction
### Current artifact sizes
A typical committed tar for a Claude Code agent image is 500 MB1 GB uncompressed.
The per-run ext4 (copy of the base, written at `start`) adds another ~1 GB of
local disk. The leak fix in this PR addresses the ext4 copies; the remote store
addresses the committed tars.
### Compression
zstd compression of the committed tar before upload is the highest-leverage
single change:
| Codec | Typical size (1 GB rootfs) | Compress speed | Decompress speed |
|-------|---------------------------|---------------|-----------------|
| gzip | 350430 MB | ~100 MB/s | ~500 MB/s |
| **zstd (default)** | **330360 MB** | **~400 MB/s** | **~2 GB/s** |
| xz | 290320 MB | ~20 MB/s | ~200 MB/s |
**zstd is the best trade-off**: 6567% size reduction, near-instantaneous
decompression. The `tar` call in `freezer.py` could pipe through `zstd` before
writing to disk and to the remote; `resume` decompresses on the way back. A
`.tar.zst` suffix marks compressed artifacts so old tars remain restorable
without the codec.
### SquashFS for the base image cache
The `rootfs/agent-<digest>/` directory (the buildah-exported tree) is rebuilt by
`image_builder.py` and turned into per-run ext4 by `mke2fs -d`. Storing the
base as a SquashFS image instead of a flat directory tree would reduce it from
~1 GB to ~330360 MB and make the cache remote-friendly. Firecracker does not
directly boot SquashFS, but the existing `mke2fs -d` path reads a directory tree
— a SquashFS mount could serve as the source. This is a larger change and lower
priority than tar compression.
### OverlayFS for per-run rootfs
Multiple simultaneous bottles sharing the same agent image today each get a full
`mke2fs -d` copy (~1 GB). OverlayFS (read-only base + writable sparse overlay)
would reduce this to ~1050 MB per instance beyond the first:
- Mount the base image directory as read-only lower layer
- Attach a sparse ext4 or tmpfs writable layer per bottle
- Pass the merged overlay to Firecracker as the block device
E2B's public write-up on Firecracker + OverlayFS confirms this approach works
at scale. The `launch.py` changes would be non-trivial (device mapper or
`fuse-overlayfs` plumbing), so this is a follow-up rather than a prerequisite
for the remote store.
---
## Recommended Approach
**Phase 1 — remote store with zstd (tight scope, actionable now)**
1. Add `--zstd` to the `tar` call in `FirecrackerFreezer._freeze`; name the
artifact `rootfs.tar.zst`. Keep uncompressed restore path for legacy tars.
2. Add a `bb firecracker upload <slug>` / `bb firecracker pull <slug>` pair that
pushes/fetches the compressed tar to the configured object store (S3-compatible
API, so B2, R2, MinIO, and Wasabi all work with the same client).
3. Store a `manifest.json` in the bucket (or a local mirror) tracking slug →
`{size, uploaded_at}`. Use it for threshold warnings and eviction.
4. Default retention: 7 days, configurable via `firecracker.image_retention_days`
in `~/.config/bot-bottle/config.toml` (or equivalent).
5. Warn at 80% of `max_capacity` (default e.g. 50 GB); evict oldest on commit
once at 100%.
**Storage recommendation:** Cloudflare R2 for hosted deployments (zero egress,
zero config), MinIO for private/on-premises.
**Phase 2 — base image cache compression**
Compress the `agent-<digest>` cache dir as a `.tar.zst` to save ~65% on repeated
image uploads. Low urgency since the base image is rebuildable.
**Phase 3 — OverlayFS per-run disk**
Replace the full per-run ext4 copy with an OverlayFS sparse layer. Largest disk
impact (~9095% savings per concurrent bottle) but highest implementation
complexity. Track as a separate PRD.
---
## Open Questions
- Does the host have a configured object-store credential path, or should the
remote store be an opt-in with an explicit `bb config set image-store.url ...`?
- Should `commit` automatically upload, or should upload be an explicit step to
avoid surprise egress?
- What is the acceptable cold-start latency for a restore from remote? A 330 MB
zstd tar at 100 Mbit/s takes ~26 s; at 1 Gbit/s, ~2.6 s. This bounds the
retention strategy (evict from local after successful upload vs keep local copy).
+60 -21
View File
@@ -10,7 +10,9 @@ classmethods forward to their module.
from __future__ import annotations
import subprocess
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch
from bot_bottle.backend.firecracker import cleanup as fc_cleanup
@@ -23,32 +25,69 @@ def _proc(stdout: str = "", returncode: int = 0) -> "subprocess.CompletedProcess
return subprocess.CompletedProcess([], returncode, stdout=stdout, stderr="")
class TestOrphanEnumeration(unittest.TestCase):
def test_orphan_vm_pids_filters_by_run_dir(self):
run_root = str(fc_cleanup._run_root())
out = (
f"111 firecracker --config-file {run_root}/dev-a/config.json\n"
"222 firecracker --config-file /somewhere/else/config.json\n"
"notanint firecracker --config-file " + run_root + "/x\n"
class TestProcessScan(unittest.TestCase):
def test_run_dir_of_matches_only_direct_children(self):
run_root = Path("/cache/run")
self.assertEqual(
Path("/cache/run/dev-a"),
fc_cleanup._run_dir_of(
f"firecracker --config-file {run_root}/dev-a/config.json", run_root
),
)
# infra/builder VMs elsewhere, or nested paths, are not ours.
self.assertIsNone(
fc_cleanup._run_dir_of("firecracker --config-file /elsewhere/config.json", run_root)
)
self.assertIsNone(
fc_cleanup._run_dir_of("firecracker --no-config", run_root)
)
with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)):
self.assertEqual([111], fc_cleanup._orphan_vm_pids())
def test_orphan_vm_pids_empty_when_pgrep_fails(self):
def test_scan_splits_live_dirs_from_orphan_pids(self):
with tempfile.TemporaryDirectory() as tmp:
run_root = Path(tmp)
(run_root / "live-a").mkdir() # dir present -> live VM, protected
# "gone-b" dir intentionally absent -> lingering VMM, orphan pid
out = (
f"111 firecracker --config-file {run_root}/live-a/config.json\n"
f"222 firecracker --config-file {run_root}/gone-b/config.json\n"
"333 firecracker --config-file /elsewhere/config.json\n"
"notanint firecracker --config-file x\n"
)
with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)):
live, orphan_pids = fc_cleanup._scan_processes(run_root)
self.assertEqual({str(run_root / "live-a")}, live)
self.assertEqual([222], orphan_pids)
def test_scan_empty_when_pgrep_fails(self):
with patch.object(fc_cleanup.subprocess, "run", return_value=_proc(returncode=1)):
self.assertEqual([], fc_cleanup._orphan_vm_pids())
self.assertEqual((set(), []), fc_cleanup._scan_processes(Path("/x")))
def test_run_dirs_empty_when_absent(self):
with patch.object(fc_cleanup.util, "cache_dir") as cache:
cache.return_value.__truediv__.return_value.is_dir.return_value = False
self.assertEqual([], fc_cleanup._run_dirs())
def test_orphan_run_dirs_excludes_live_and_missing_root(self):
with tempfile.TemporaryDirectory() as tmp:
run_root = Path(tmp)
(run_root / "live-a").mkdir()
(run_root / "dead-b").mkdir()
live = {str(run_root / "live-a")}
self.assertEqual(
[str(run_root / "dead-b")],
fc_cleanup._orphan_run_dirs(run_root, live),
)
# absent run root -> nothing to reap
self.assertEqual([], fc_cleanup._orphan_run_dirs(Path("/nope/run"), set()))
def test_prepare_cleanup_assembles_plan(self):
with patch.object(fc_cleanup, "_orphan_vm_pids", return_value=[7]), \
patch.object(fc_cleanup, "_run_dirs", return_value=["/run/x"]):
plan = fc_cleanup.prepare_cleanup()
self.assertEqual((7,), plan.vm_pids)
self.assertEqual(("/run/x",), plan.run_dirs)
def test_prepare_cleanup_reaps_orphans_only(self):
"""The live VM's dir is never in the plan; the dead one is."""
with tempfile.TemporaryDirectory() as tmp:
run_root = Path(tmp)
(run_root / "live-a").mkdir()
(run_root / "dead-b").mkdir()
out = f"111 firecracker --config-file {run_root}/live-a/config.json\n"
with patch.object(fc_cleanup, "_run_root", return_value=run_root), \
patch.object(fc_cleanup.subprocess, "run", return_value=_proc(out)):
plan = fc_cleanup.prepare_cleanup()
self.assertEqual((), plan.vm_pids)
self.assertEqual((str(run_root / "dead-b"),), plan.run_dirs)
self.assertNotIn(str(run_root / "live-a"), plan.run_dirs)
class TestCleanupRemoval(unittest.TestCase):