Compare commits

..

5 Commits

Author SHA1 Message Date
didericis 314b30c013 docs(prd): renumber 0083 -> 0081, and point at the umbrella issue
prd-number-check / require-numbered-prds (pull_request) Successful in 14s
test / image-input-builds (pull_request) Successful in 46s
test / unit (pull_request) Successful in 1m0s
test / integration-docker (pull_request) Successful in 1m8s
test / coverage (pull_request) Successful in 18s
tracker-policy-pr / check-pr (pull_request) Successful in 14s
cf6bbc43 renumbered this PRD 0081 -> 0083 as "0081 already taken", but 0081 is
free: main goes 0080, 0082, and no other branch claims it. 0083 is the number
that is actually taken — feat/pinned-infra-artifacts holds
docs/prds/0083-packaged-infra-artifacts.md, renamed there from its prd-new-
form, so both PRDs were sitting on 0083 and whichever merged second would have
collided. Move back to 0081 and leave 0083 to the infra-artifacts PRD.

The Issue header pointed at #510 (CA) and #512 (git-gate), which have since
been folded into #516 and closed. #516 carries both symptoms plus the shared
root cause, so it is the reference that stays useful.

requirements.gateway.lock also matches "0083" — inside a sha256 hash, left
alone.
2026-07-27 11:33:23 -04:00
didericis-claude cf6bbc43da docs(prd): renumber 0081 -> 0083 (0081 already taken)
test / integration-docker (pull_request) Has been cancelled
prd-number-check / require-numbered-prds (pull_request) Successful in 14s
lint / lint (push) Successful in 1m4s
test / image-input-builds (pull_request) Successful in 59s
tracker-policy-pr / check-pr (pull_request) Successful in 11s
test / unit (pull_request) Failing after 13m55s
test / coverage (pull_request) Has been skipped
PRD number 0081 was claimed by another PRD before this one merged (main is on
0082); renumber this design to the next free slot, 0083. File + title only.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-27 15:14:54 +00:00
didericis 4e26e075c8 docs(prd): reconcile via a backend-ABC method across all backends (0081 review)
Address review: target all backends, not firecracker-only. Make the reconcile
an @abc.abstractmethod `attach_bottled_agents_to_gateway()` on BottleBackend so
every backend implements it (cross-backend by construction) and the host calls
it on gateway bring-up. Drop the docker/macOS non-goal; add the open question of
retiring docker's now-redundant CA bind-mount.

Refs #516
2026-07-27 15:14:33 +00:00
didericis cca724244e feat(orchestrator): update a single egress secret in place (0081)
Add a single-secret counterpart to reprovision_from_secret's restore-all:
update ONE egress token for a running bottle without a relaunch, for
refreshing a short-lived host credential (e.g. the Codex access token)
whose launch-time snapshot has expired.

- registry_store.store_agent_secret: per-key upsert (delete+insert of the
  one row), the counterpart of store_agent_secrets' replace-all.
- OrchestratorCore.update_agent_secret: set the in-memory token AND upsert
  the re-encrypted row under the bottle's env_var_secret, leaving other
  tokens untouched.
- POST /bottles/<id>/secret (cli-only) + client.update_agent_secret.

Refs #510, #512

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-27 15:14:33 +00:00
didericis cecc1b4d60 docs(prd): reprovision gateway-dependent state on gateway bring-up (0081)
Reconcile every running bottle against a freshly-booted gateway (replace
each agent's CA, re-provision git-gate, restore egress tokens) instead of
persisting gateway state on volumes. Supersedes the abandoned CA-volume
(#511) and git-gate-volume (#513) PRs.

Refs #510, #512

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-27 15:14:33 +00:00
15 changed files with 308 additions and 1111 deletions
+2 -5
View File
@@ -20,7 +20,6 @@ import subprocess
import sys
from pathlib import Path
from ... import invocation
from ... import resources
from . import netpool
from . import util
@@ -180,11 +179,9 @@ def _setup_systemd() -> None:
f"sudo systemctl daemon-reload\n"
f"sudo systemctl enable --now {netpool.SYSTEMD_UNIT}\n"
)
# Absolute path, not `sudo bot-bottle`: sudo's secure_path drops
# ~/.local/bin, where both pipx and install.sh put the entry point.
sys.stderr.write(
f"\n(Or re-run this as root to install it directly:\n"
f" {invocation.sudo_command('backend', 'setup', '--backend=firecracker')})\n"
f"\n(Or re-run this as root to install it directly: "
f"sudo bot-bottle backend setup --backend=firecracker)\n"
)
-48
View File
@@ -1,48 +0,0 @@
"""How to tell a user to re-run this CLI.
`bot-bottle …` is the right thing to print for anything the user runs as
themselves — it is on their PATH, since that is how they got here.
Under `sudo` it is not. sudo replaces PATH with sudoers' `secure_path`
(`/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin` on Debian and
Ubuntu, similar elsewhere), which deliberately excludes user-writable
directories. Both supported install paths put the entry point in one of those:
pipx uses `~/.local/bin`, and `install.sh`'s venv fallback symlinks there too.
So `sudo bot-bottle …` fails with "command not found" for exactly the users who
followed the documented install, while working for anyone who happened to
install system-wide — which is why it survives review so easily.
Naming the absolute path sidesteps secure_path entirely.
"""
from __future__ import annotations
import os
import shutil
import sys
def self_path() -> str:
"""Absolute path to this CLI's entry point.
Falls back to the bare name when the entry point cannot be resolved (an
unusual invocation such as `python -m`), because a slightly wrong hint is
better than a traceback while reporting an unrelated problem.
"""
argv0 = sys.argv[0] or "bot-bottle"
resolved = shutil.which(argv0) or argv0
if not os.path.isabs(resolved):
if os.path.exists(resolved):
resolved = os.path.abspath(resolved)
else:
return "bot-bottle"
return resolved
def sudo_command(*args: str) -> str:
"""A copy-pasteable `sudo …` invocation of this CLI.
>>> sudo_command("backend", "setup", "--backend=firecracker")
'sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker'
"""
return " ".join(["sudo", self_path(), *args])
+21
View File
@@ -50,6 +50,12 @@ class ReprovisionBody(_StrictModel):
env_var_secret: StrictStr
class SecretBody(_StrictModel):
name: StrictStr
value: StrictStr
env_var_secret: StrictStr
class ReconcileBody(_StrictModel):
live_source_ips: list[StrictStr]
grace_seconds: float | None = None
@@ -259,6 +265,21 @@ def create_app(orch: OrchestratorCore, *, signing_key: str) -> FastAPI:
return {"reprovisioned": True}
raise HTTPException(404, "no stored secrets for this bottle")
@app.post("/bottles/{bottle_id}/secret")
def update_secret(bottle_id: str, body: SecretBody) -> dict[str, object]:
# Update ONE egress token for a running bottle in place — the
# single-secret form of reprovision, for refreshing a short-lived host
# credential (e.g. the Codex access token) without a relaunch. cli-only
# (not in _GATEWAY_ROUTES): a bottle must never set its own tokens.
if orch.update_agent_secret(
bottle_id,
_required(body.name, "name"),
_required(body.value, "value"),
_required(body.env_var_secret, "env_var_secret"),
):
return {"updated": True}
raise HTTPException(404, "no such bottle")
@app.delete("/bottles/{bottle_id}")
def teardown(bottle_id: str) -> dict[str, object]:
if orch.teardown_bottle(bottle_id):
+21
View File
@@ -184,6 +184,27 @@ class OrchestratorClient:
)
return True
def update_agent_secret(
self, bottle_id: str, name: str, value: str, env_var_secret: str,
) -> bool:
"""Update ONE egress token for a running bottle in place
(`POST /bottles/<id>/secret`) — the single-secret form of
`reprovision_gateway`, for pushing a freshly-refreshed host credential
into a bottle without a relaunch. Returns True on success, False when the
orchestrator doesn't know the bottle (404)."""
status, _ = self._request(
"POST",
f"/bottles/{bottle_id}/secret",
{"name": name, "value": value, "env_var_secret": env_var_secret},
)
if status == 404:
return False
if not 200 <= status < 300:
raise OrchestratorClientError(
f"update_agent_secret {bottle_id}: HTTP {status}"
)
return True
def teardown_bottle(self, bottle_id: str) -> bool:
"""Tear a bottle down (`DELETE /bottles/<id>`). False if the
orchestrator didn't know it (404) — idempotent for cleanup paths."""
+24
View File
@@ -379,6 +379,30 @@ class OrchestratorCore:
self._tokens[bottle_id] = decrypted
return True
def update_agent_secret(
self, bottle_id: str, name: str, value: str, env_var_secret: str,
) -> bool:
"""Update ONE egress token for a known bottle in place — the
single-secret form of ``reprovision_from_secret``.
Sets the in-memory token AND upserts the single re-encrypted row under
*env_var_secret* (the same key the rest of the rows are encrypted with, so
the whole set stays decryptable by a later ``reprovision_from_secret``),
leaving every other token untouched. Returns False if the bottle is
unknown.
Unlike ``reprovision_from_secret`` (which restores the values captured at
launch), this pushes a *caller-supplied* value — used to refresh a
short-lived host credential (e.g. the Codex access token) into a
still-running bottle without a relaunch."""
from .store.secret_store import encrypt_value
if self.registry.get(bottle_id) is None:
return False
self._tokens.setdefault(bottle_id, {})[name] = value
self.registry.store_agent_secret(
bottle_id, name, encrypt_value(env_var_secret, value))
return True
# --- consolidated gateway ----------------------------------------------
def gateway_status(self) -> dict[str, object]:
@@ -370,6 +370,30 @@ class RegistryStore(DbStore):
)
self._chmod()
def store_agent_secret(
self,
bottle_id: str,
key: str,
encrypted_value: str,
secret_type: str = "injected_env_var",
) -> None:
"""Upsert ONE encrypted secret row (env-var name → ciphertext) for
*bottle_id*, leaving the bottle's other secrets untouched — the per-key
counterpart of ``store_agent_secrets``' replace-all. Delete-then-insert
because the table carries no unique constraint to `ON CONFLICT` against."""
with self._connection() as conn:
conn.execute(
"DELETE FROM bottled_agent_secrets "
"WHERE bottled_agent_id = ? AND key = ? AND type = ?",
(bottle_id, key, secret_type),
)
conn.execute(
"INSERT INTO bottled_agent_secrets "
"(bottled_agent_id, key, value, type) VALUES (?, ?, ?, ?)",
(bottle_id, key, encrypted_value, secret_type),
)
self._chmod()
def get_agent_secrets(
self,
bottle_id: str,
@@ -0,0 +1,117 @@
# PRD 0081: Reprovision gateway-dependent state on gateway bring-up
- **Status:** Active
- **Author:** claude
- **Created:** 2026-07-26
- **Issue:** #516
## Summary
When the gateway is (re)built, reconcile every already-running bottle against the
fresh gateway instead of persisting the gateway's state. On a gateway cold boot,
the gateway reconciles all live bottles in one flow: **replace** each agent's
trusted CA with the freshly-minted gateway CA, **re-provision** each bottle's
git-gate repos + creds onto the gateway, and **restore** each bottle's egress
tokens. One mechanism across all three services; the CA rotates for free on every
bring-up.
## Problem
A gateway rebuild/restart silently breaks every already-running bottle:
- **CA (#510).** The gateway's mitmproxy mints a new CA on a fresh rootfs; agents
still trust the old one, so egress fails TLS verification (`SSL certificate
verification failed`).
- **git-gate (#512).** Per-bottle bare repos (`/git/<id>`) + deploy creds
(`/git-gate/creds/<id>`) live in the gateway's ephemeral rootfs; a rebuild wipes
them and the agent 404s on fetch/push.
Both are the same root cause: per-bottle gateway-dependent state is provisioned
**once, at bottle launch**, and nothing restores it for already-running bottles
when the gateway comes back. The existing launch-time reprovision
(`reprovision_bottles`) restores **only** egress tokens, and only as a side effect
of the *next* launch.
Persisting the state (a host bind-mount / a per-VM data volume per service) was
prototyped and rejected: it differs per service (a volume for the CA, another for
git-gate, reprovision for tokens), it pins the CA static forever (no rotation),
and it adds volume surface that `docker volume prune` / a wiped cache can silently
destroy (the original #450 failure mode).
## Goals / Success Criteria
- A gateway (re)boot restores **all** running bottles' gateway-dependent state
with no manual step and no relaunch — the agent's next egress / fetch / push
just works.
- **One** reconcile flow covering CA, git-gate, and egress tokens, rather than a
different mechanism per service.
- The CA **rotates** on every gateway bring-up (no long-lived CA), distributed to
running agents by the same reconcile.
- Per-bottle failures are tolerated: one unreachable or malformed bottle does not
block the others or the gateway coming up.
- **All backends** (Firecracker, docker, macOS) reconcile through the *same*
contract — an abstract method on the backend base class, so a new backend
cannot forget to implement it and none drifts onto a bespoke mechanism.
## Non-goals
- Deliberate mid-session CA rotation *without* a gateway restart — `rotate_ca`
stays for that operator action.
- Changing source-IP attribution, `/resolve`, or the plane split (#469).
## Design
**The contract — `attach_bottled_agents_to_gateway()` on the backend ABC.**
Reconciling running bottles against the current gateway is a backend
responsibility (only the backend can enumerate its agents and reach them —
firecracker over SSH, docker/macOS over `exec`/`cp`), so it is an
`@abc.abstractmethod` on `BottleBackend` (`backend/base.py`). Every backend
implements it; the host calls it whenever the gateway is (re)brought up. This is
what makes the fix cross-backend by construction rather than a per-backend
follow-up. It reprovisions **all** registered bottles' gateway-dependent state:
CA, git-gate, and egress tokens.
**Trigger — the gateway bring-up path.** The host calls
`attach_bottled_agents_to_gateway()` only when the gateway was actually
(re)brought up — the cold-boot branch of the infra bring-up (e.g.
`FirecrackerInfraService.ensure_running` after it boots a fresh pair), never on an
adopt of a healthy, current gateway (state intact). So it fires exactly when the
gateway was (re)booted — including orchestrator restarts, since the pair boots
together — and there is no bare-restart path that bypasses bring-up.
**Per-backend implementation.** Each backend's `attach_bottled_agents_to_gateway`
enumerates its live bottles and, for each, reconciles the three services against
the current gateway. The firecracker implementation, once the gateway VM is up
and its CA is available:
1. Map each live bottle's guest IP → `bottle_id` from the orchestrator registry
(`list_bottles`).
2. Install the **current** shared git-gate hooks (`git_gate_render_hook` /
`git_gate_render_access_hook`) into the fresh gateway once — rendered from
code, never from a bottle's possibly-stale state dir.
3. For each live agent VM (enumerated from its run dir):
- **CA:** SSH the current gateway CA into the agent's trust store and run
`update-ca-certificates` (unconditional replace — there is one gateway, so no
fingerprint match is needed).
- **git-gate:** rebuild the bottle's upstreams from its persisted git-gate
state dir (deploy key, known_hosts, upstream URL) and re-init its bare repos
+ per-repo creds under `/git/<bottle_id>`.
- **egress token:** read the agent's `ENV_VAR_SECRET` and feed
`reprovision_bottles`, restoring the orchestrator's in-memory tokens.
4. Every per-bottle step is wrapped so one failure is logged and skipped.
**Retire the persistence prototype.** No CA data volume, no git-gate data volume
(the abandoned PRs #511 / #513). The launch-time `_reprovision_running_bottles`
call folds into this bring-up reconcile, so egress tokens are restored on the same
cold-boot trigger (an adopt needs no restore — the orchestrator never restarted).
**CA rotation.** Because the gateway rootfs is ephemeral, every cold boot mints a
fresh CA; reconcile is what distributes it, so a routine gateway rebuild doubles
as a CA rotation with zero extra machinery.
## Open questions
- Docker's existing host-bind-mounted CA (`host_gateway_ca_dir`): once docker's
`attach_bottled_agents_to_gateway` pushes the CA to running agents on bring-up,
the bind-mount is redundant — drop it (so docker rotates like firecracker) or
keep it as belt-and-suspenders? Leaning drop, for one behaviour across backends.
@@ -1,183 +0,0 @@
# Testing a clean bot-bottle install on Linux
How do you exercise `install.sh` the way a brand-new user would — on a
pristine Linux environment you can throw away afterward — *without*
polluting your daily-driver host, and across the several package-management
regimes Linux fragments into? This is the Linux counterpart to
[`testing-clean-install-on-macos.md`](testing-clean-install-on-macos.md);
the conclusion is different because Linux gives us a boundary macOS doesn't.
## Summary
On macOS the honest options were a throwaway user or a VM, and the throwaway
user won on pragmatics (nested virtualization is gated to M3+). On Linux the
calculus flips: a **disposable KVM virtual machine, booted from a distro
cloud image and deleted per run, is both the cleanest boundary and the one
that lets a single harness cover Ubuntu, Fedora, Arch, Alpine, and NixOS**.
The host already requires KVM for the Firecracker backend, so the VM is cheap
here.
The harness lives at [`scripts/linux-install-test.sh`](../../scripts/linux-install-test.sh).
Per run it caches one read-only base image, boots a throwaway copy-on-write
overlay (`qemu-img create -b base`), installs the distro's prerequisites,
pipes *this checkout's* `install.sh` into the guest exactly as `curl … | sh`
would, asserts the CLI installed, and deletes the overlay — the Linux
equivalent of `docker run --rm`, for a whole machine.
## Why a VM, not a container or a throwaway user
| Mechanism | Why it's the wrong boundary here |
|---|---|
| **Container** (`docker run --rm`) | Shares the host kernel and ships a deliberately minimal userland — no systemd, a stubbed-out package manager story, and (crucially) it doesn't reproduce the *externally-managed Python* (PEP 668) that real desktop/server installs put in front of the user. It tests "does install.sh run in a container," not "does it run on a real distro." |
| **Throwaway user** (`useradd`/`userdel`) | The macOS pick, but weaker on Linux: it reaches the real host, yet every system package it installs (python, pipx, git via `apt`/`dnf`/…) stays behind, and it can only ever test the *one* distro the host runs. The whole Linux-specific value is the cross-distro matrix. |
| **Disposable KVM VM** (this harness) | A genuine kernel + userland + package-manager boundary that wipes to nothing on teardown, and swaps freely between distro cloud images. The one real cost — nested virtualization for the *backend* — doesn't apply, because we gate the installer, not the runtime (below). |
## Two variants: `test` (bare host) and `test-ready` (prepared host)
`install.sh` never installs a backend, and never installs its own toolchain
prerequisites (python3, git, pipx) — it installs the `bot-bottle` package and
runs `doctor`, which *reports* what's missing
([`install.sh`](../../install.sh) header,
[`bot_bottle/cli/commands/doctor.py`](../../bot_bottle/cli/commands/doctor.py)).
That leaves two distinct things worth testing, split into two subcommands that
mirror the macOS harness's `test` / `test-ready` convention (there the split is
the backend service; here it is the toolchain the installer needs):
- **`test`** — `install.sh` runs on the **bare cloud image**, prerequisites and
all left as the vendor ships them. This exercises `install.sh`'s own
prerequisite-guard logic — the entire first half of the script (python
version gate, git-for-git-specs gate, pipx/pip PEP-668 handling).
- **`test-ready`** — the harness installs python3 + git + pipx first (the
`prereqs` step), then runs `install.sh`. This is the *prepared-host happy
path*: does a clean install actually land and produce a working CLI?
**Pass criteria differ by variant:**
| Variant | PASS when |
|---|---|
| `test` | `install.sh` **either** installs cleanly (the image already carried enough) **or** declines with one of its own recognized, actionable prerequisite errors (missing python3/git, no usable pip, PEP 668). A crash or an *unrecognized* failure is a FAIL. |
| `test-ready` | `install.sh` actually lands: the `bot-bottle` entry point is present and runs, and `doctor` reports a usable python and config without crashing. A graceful decline is no longer good enough. |
Neither variant requires a green `doctor`: inside the VM there is no nested KVM
or Docker, so **the backend is correctly reported not-ready** — install.sh does
not install a backend and cannot regress one, and this harness does not
provision the Docker backend. This is where Linux necessarily diverges from the
macOS `test-ready`, which reaches the host backend; `BB_TEST_REQUIRE_BACKEND=1`
makes readiness fatal anyway, for a nested-virt host that can satisfy it. The
verdict instead classifies `doctor`'s output the way the macOS harness does — a
`Traceback` is an install defect (fail), a missing `python`/`config` line is a
fail, a not-ready backend is reported — so a genuine installer regression (a
broken shim, an import error, a botched PATH) stays visible.
`test-all` runs the full matrix — every distro × both variants — each cell in
its own throwaway VM, and prints a per-cell PASS/FAIL summary.
## The distro matrix is the point
Each distro exercises a different corner of the installer:
| Distro | Cloud image | What it stresses |
|---|---|---|
| **Ubuntu** (noble) | `cloud-images.ubuntu.com` | The common case; `apt`'s `pipx`, externally-managed Python (PEP 668) → install.sh's pipx path. |
| **Fedora** | Fedora Cloud Base Generic | `dnf` packaging, a different default Python, BSD-style checksum file. |
| **Arch** | `geo.mirror.pkgbuild.com/images/latest` | Rolling / newest Python; `python-pipx`. |
| **Alpine** | Alpine `nocloud_` (cloudinit) image | musl libc + BusyBox `sh` — the harshest POSIX-`sh` host for a `#!/bin/sh` installer. |
| **NixOS** | locally built with `nixos-generators` | No FHS `~/.local` on PATH by default; `nix profile install` prereqs; pipx laying a self-contained venv on a non-FHS host. |
### Validation run (2026-07-27) — full green
Full matrix on the delphi KVM host, QEMU 11.0.2, both variants × all five
distros passing:
| Distro | `test` (bare) | `test-ready` (prepared) |
|---|---|---|
| Ubuntu 24.04 | ✅ declines at git gate | ✅ installs, doctor python+config green |
| Fedora 44 | ✅ declines at git gate | ✅ installs |
| Arch (latest) | ✅ declines at git gate | ✅ installs |
| Alpine 3.21 | ✅ declines at git gate | ✅ installs |
| NixOS 24.11 | ✅ declines (no python3) | ✅ installs |
The bare `test` sees `install.sh` decline soundly — exit 1 at the
git-for-git-specs gate on the Debian/Fedora/Arch/Alpine images (they ship
python3 but not git), and at the python3 gate on NixOS (no python3 on PATH) —
and `test-ready` installs cleanly with `doctor` reporting a usable python and
config (backends all not-ready, as expected in a plain VM).
Getting to green surfaced and fixed a series of real defects:
- **Fedora 41 was EOL/404** → bumped to 44.
- The liveness probe used `bot-bottle --version`, which the CLI does not
implement (unknown args die non-zero), so every *successful* install was
misreported as failed → switched to `bot-bottle --help`.
- **Alpine** needed three fixes: the `generic_` image ignores a NoCloud seed
(switched to the `nocloud_` variant); OpenRC does not auto-start sshd after
cloud-init injects the key (start it via `runcmd`); and Alpine's non-PAM
sshd refuses pubkey auth for a cloud-init-*locked* account (give it a
throwaway password). It also has no `sudo` by default (install it via
cloud-init `packages:`).
- **NixOS** publishes no downloadable cloud qcow2 (its cloud images are
Hydra-built AMIs), so the harness builds one with `nixos-generators`
([`linux-install-test-nixos.nix`](../../scripts/linux-install-test-nixos.nix)):
cloud-init for the key, flakes enabled, deliberately no python/git/pipx. The
`test-ready` prereq install pins `nixpkgs/nixos-24.11` because the guest's
default unstable registry builds pipx from source (and its test suite
currently fails to build).
- Two harness-hygiene bugs also fixed: `cmd_down` left `serial.log` behind
(orphaned run dirs), and the teardown trap was armed after `cmd_up`, leaking
a VM when `wait_for_ssh` timed out.
In `test-ready` the harness installs `python3 + git + pipx` first on each
distro (install.sh installs none of them), so all five drive the recommended
pipx path. `test` then removes that scaffolding and lets each distro's bare
image collide with install.sh's guards — on most cloud images python3 is
present (cloud-init needs it) but git and pipx are not, so install.sh is
expected to decline at the git-for-git-specs gate or the PEP-668 pip check with
an actionable message. Both are legitimate, and the two variants together cover
the whole first half of the installer as well as the happy path.
## What a clean install touches (the footprint that decides "wipeable")
| Artifact | Location | In the guest's `$HOME`? | Survives VM teardown? |
|---|---|---|---|
| Config / state / db | `~/.bot-bottle/{agents,bottles,contrib,…}` ([`install.sh`](../../install.sh)) | ✅ | ❌ overlay deleted |
| pipx venv + shim | `~/.local/pipx/venvs/bot-bottle`, shim in `~/.local/bin` | ✅ | ❌ overlay deleted |
| pip `--user` fallback | `~/.local/lib` + `~/.local/bin` | ✅ | ❌ overlay deleted |
| **Distro prerequisites** (python/git/pipx, `test-ready` only) | system paths via `apt`/`dnf`/`pacman`/`apk`/`nix profile` | ❌ | ❌ **overlay deleted** |
Unlike the macOS throwaway user (whose Homebrew / Apple-Container / Rosetta
footprint *survives*), **every row here dies with the overlay** — that is the
VM's whole advantage. The cached base image is read-only backing and is the
only thing that persists between runs, on purpose.
## Design notes baked into the harness
- **User-mode networking** (`-netdev user,hostfwd=tcp:127.0.0.1:PORT-:22`):
no root, no bridge, no host network state touched. Only SSH is forwarded.
- **cloud-init seed ISO** (`cloud-localds`) injects an ephemeral SSH keypair
and a passwordless-sudo login. The keypair is generated per run and deleted
on teardown; the guest can't be logged into after it's gone.
- **Copy-on-write overlay**: the cached base is never mutated, so a corrupt or
interrupted run can't poison the cache; downloads land at `*.partial` and
are renamed only after checksum verification.
- **Checksums**: verified against each vendor's published sums file at
download time (GNU `hash file`, bare-hash, and Fedora's BSD
`SHA256 (file) = hash` formats are all handled). Alpine ships `.sha512` only
(this verifier is sha256) so it is skipped; NixOS is built locally, not
downloaded, so there is nothing to verify.
- **NixOS is built, not downloaded**: `ensure_base_image` runs
`nixos-generate -f qcow` against
[`linux-install-test-nixos.nix`](../../scripts/linux-install-test-nixos.nix)
once and caches the result; the per-run seed/overlay flow is otherwise
identical to the downloaded distros.
- **`test-all`** runs every distro × both variants (`test` and `test-ready`),
each cell in its own subshell on its own forwarded port, so one cell's
failure (or teardown trap) can't abort the matrix; it prints a per-cell
PASS/FAIL summary.
## Not wired into PR CI
Like the macOS harness, the runtime is host-specific (needs `/dev/kvm`,
`qemu`, and `cloud-localds`) and is not exercised by the Linux pull-request
runner. It is validated statically (`bash -n`, `shellcheck`) and run by hand
on a KVM-capable host. The cloud-image URLs in the `DISTRO` table are the one
place to bump when a distro cuts a newer build.
-24
View File
@@ -1,24 +0,0 @@
# NixOS image for scripts/linux-install-test.sh.
#
# NixOS publishes no downloadable cloud qcow2 (its cloud images are Hydra-built
# AMIs), so the harness BUILDS this one with nixos-generators (-f qcow). It is
# deliberately minimal — no python3/git/pipx — so the bare `test` variant is
# genuinely under-provisioned and exercises install.sh's guards; `test-ready`
# provisions them with `nix profile install` (hence flakes below).
{ lib, ... }:
{
# Consume the same NoCloud seed the other distros use: cloud-init injects the
# per-run ephemeral SSH key for root. Leave networking to NixOS's default
# dhcpcd (QEMU user-mode NAT) — enabling cloud-init's networkd here conflicts
# with dhcpcd and can drop the guest's network.
services.cloud-init.enable = true;
services.cloud-init.network.enable = false;
services.openssh.enable = true;
services.openssh.settings.PermitRootLogin = lib.mkForce "prohibit-password";
# Flakes so the `test-ready` prereq step can `nix profile install nixpkgs#...`.
nix.settings.experimental-features = [ "nix-command" "flakes" ];
system.stateVersion = "24.11";
}
-753
View File
@@ -1,753 +0,0 @@
#!/usr/bin/env bash
# Clean-install test harness for the Linux path.
#
# Exercises install.sh the way a brand-new user would, inside a THROWAWAY
# QEMU/KVM virtual machine that is booted from a distro cloud image and
# deleted afterward. install.sh's entire footprint is user-home-local (the
# pipx venv under ~/.local, the ~/.bot-bottle config dir, and a printed PATH
# hint), so a fresh VM's fresh $HOME is the clean surface we want — and unlike
# a throwaway user account, tearing the VM down also wipes any OS-level
# prerequisites installed into it, so the reset is total. Full rationale in
# docs/research/testing-clean-install-on-linux.md.
#
# Why a VM and not a container or a throwaway user: a container shares the
# host kernel and cannot exercise a genuinely pristine OS (systemd, the distro
# package manager, PEP 668 externally-managed Python) the way a real guest
# does, and a throwaway user leaves every system package it installs behind.
# The host already needs KVM for the Firecracker backend, so a per-run,
# copy-on-write VM is cheap here: one cached base image, a throwaway overlay
# per run (`qemu-img create -b base`), deleted on teardown — the Linux
# equivalent of `docker run --rm`, but for a whole machine.
#
# Usage:
# ./scripts/linux-install-test.sh test # up -> run -> verdict -> down (bare host)
# ./scripts/linux-install-test.sh test-ready # ... with prerequisites installed first
# ./scripts/linux-install-test.sh test-all # every distro × both variants, with a summary
# ./scripts/linux-install-test.sh up # fetch base image, boot a fresh VM
# ./scripts/linux-install-test.sh prereqs # install python3 + git + pipx in the VM
# ./scripts/linux-install-test.sh run # pipe install.sh into the VM
# ./scripts/linux-install-test.sh status # VM reachable? is the install sound?
# ./scripts/linux-install-test.sh down # kill the VM, delete overlay + seed (the reset)
# ./scripts/linux-install-test.sh ssh # open an interactive shell in the running VM
#
# TWO TEST VARIANTS, because "does install.sh handle an unprepared host" and
# "does a prepared host get a clean install" are different questions (mirrors
# the macOS harness's test / test-ready split):
#
# test A Linux system WITHOUT the prerequisites set up — the default
# state of a stock cloud image (python3 is usually present for
# cloud-init, but git and pipx are not). This exercises
# install.sh's own prerequisite-guard logic. It is SOUND — a
# PASS — when install.sh EITHER installs cleanly (the image
# already had enough) OR declines with one of its own recognized,
# actionable errors (missing python3/git, no usable pip, PEP 668
# externally-managed). A crash or an unrecognized failure fails.
#
# test-ready A Linux system WITH the prerequisites satisfied — the harness
# installs python3 + git + pipx first (see the DISTRO table),
# then runs install.sh. A graceful decline is no longer good
# enough here: the install MUST land, the entry point must run,
# and doctor must report a usable python and config without
# crashing.
#
# Neither variant requires a green backend. install.sh does not install a
# backend and cannot regress one, and inside a plain VM there is no nested KVM
# for Firecracker; the harness does not provision the Docker backend either. So
# backend readiness is reported, not required (this is where Linux necessarily
# diverges from the macOS test-ready, which reaches the host backend).
# BB_TEST_REQUIRE_BACKEND=1 makes it fatal anyway, for a nested-virt host.
#
# Config via env:
# BB_TEST_DISTRO ubuntu | fedora | arch | alpine | nixos (default: ubuntu)
# BB_TEST_SSH_PORT host port forwarded to the guest's :22 (default: 2222)
# BB_TEST_CACHE_DIR where base images are cached (default: ~/.cache/bot-bottle-install-test)
# BB_TEST_RUN_DIR per-run scratch (overlay, seed, key, …) (default: a mktemp dir)
# BB_TEST_MEM_MB guest RAM (default: 2048)
# BB_TEST_CPUS guest vCPUs (default: 2)
# BB_TEST_DISK overlay virtual size (default: 12G)
# BB_TEST_BOOT_TIMEOUT seconds to wait for SSH after boot (default: 300)
# BB_TEST_KEEP 1 = the test cycles skip teardown, to poke at a failure
# BB_TEST_REQUIRE_BACKEND 1 = make a not-ready backend fatal (needs nested virt)
# BB_TEST_INSTALL_URL curl this install.sh in the guest instead of piping the local checkout
# BB_TEST_SKIP_VERIFY 1 = skip base-image checksum verification (not recommended)
# BOT_BOTTLE_INSTALL_SPEC passed through to install.sh (pip / git spec)
#
# Notes:
# * Needs /dev/kvm, qemu-system-x86_64, and cloud-localds (cloud-image-utils
# / cloud-utils); the nixos distro additionally needs nixos-generate. On the
# NixOS host: nix shell nixpkgs#qemu nixpkgs#cloud-utils nixpkgs#nixos-generators
# * Networking is user-mode (`-netdev user,hostfwd`) so the harness needs no
# root, no bridge, and touches no host network state. Only SSH is forwarded.
# * The cloud-image URLs in the DISTRO table are the one place to bump when a
# distro cuts a new build; each is verified against the vendor's published
# checksum at download time (guarding against truncated/corrupt pulls).
set -euo pipefail
DISTRO="${BB_TEST_DISTRO:-ubuntu}"
SSH_PORT="${BB_TEST_SSH_PORT:-2222}"
CACHE_DIR="${BB_TEST_CACHE_DIR:-${XDG_CACHE_HOME:-$HOME/.cache}/bot-bottle-install-test}"
MEM_MB="${BB_TEST_MEM_MB:-2048}"
CPUS="${BB_TEST_CPUS:-2}"
DISK="${BB_TEST_DISK:-12G}"
BOOT_TIMEOUT="${BB_TEST_BOOT_TIMEOUT:-300}"
_SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
_REPO_ROOT="$(cd "$_SCRIPT_DIR/.." && pwd)"
# Per-run scratch. Persisted across sub-commands (up/prereqs/run/status/down)
# via a marker file so `up` in one invocation and `down` in the next find the
# same VM; the test cycles set RUN_DIR themselves and never write the marker.
RUN_DIR="${BB_TEST_RUN_DIR:-}"
# Set by the test cycles, which chain the steps and suppress the per-step "next
# command" hints. _STEPS / _PASS_CLAIM / _REQUIRE_INSTALL are the per-variant
# knobs the shared cycle and teardown read.
IN_TEST=0
_STEPS=4
_PASS_CLAIM=""
_REQUIRE_INSTALL=0
# --- distro table ----------------------------------------------------
# For each distro: cloud-image URL | checksum-file URL | default SSH user |
# prerequisite-install command (run in the guest; uses sudo when user != root).
#
# The prerequisite command installs python3 + git + pipx — install.sh installs
# none of them — so `test-ready` (and the `prereqs` sub-command) drive
# install.sh down its recommended pipx path. Bump the URLs here when a distro
# publishes a newer build.
declare -A IMAGE_URL SUM_URL SSH_USER PREREQ
IMAGE_URL[ubuntu]="https://cloud-images.ubuntu.com/noble/current/noble-server-cloudimg-amd64.img"
SUM_URL[ubuntu]="https://cloud-images.ubuntu.com/noble/current/SHA256SUMS"
SSH_USER[ubuntu]="ubuntu"
PREREQ[ubuntu]="sudo apt-get update && sudo DEBIAN_FRONTEND=noninteractive apt-get install -y python3 git pipx"
IMAGE_URL[fedora]="https://download.fedoraproject.org/pub/fedora/linux/releases/44/Cloud/x86_64/images/Fedora-Cloud-Base-Generic-44-1.7.x86_64.qcow2"
SUM_URL[fedora]="https://download.fedoraproject.org/pub/fedora/linux/releases/44/Cloud/x86_64/images/Fedora-Cloud-44-1.7-x86_64-CHECKSUM"
SSH_USER[fedora]="fedora"
PREREQ[fedora]="sudo dnf install -y python3 git pipx"
IMAGE_URL[arch]="https://geo.mirror.pkgbuild.com/images/latest/Arch-Linux-x86_64-cloudimg.qcow2"
SUM_URL[arch]="https://geo.mirror.pkgbuild.com/images/latest/Arch-Linux-x86_64-cloudimg.qcow2.SHA256"
SSH_USER[arch]="arch"
PREREQ[arch]="sudo pacman -Sy --noconfirm python git python-pipx"
# Use the *nocloud_* Alpine variant, not generic_: the generic image probes
# network datasources and ignores the local NoCloud seed, so cloud-init never
# runs and the SSH key is never injected. (Only published at .0 patch levels.)
IMAGE_URL[alpine]="https://dl-cdn.alpinelinux.org/alpine/v3.21/releases/cloud/nocloud_alpine-3.21.0-x86_64-bios-cloudinit-r0.qcow2"
SUM_URL[alpine]="" # Alpine cloud images ship .sha512 only; this verifier is sha256.
SSH_USER[alpine]="alpine"
PREREQ[alpine]="sudo apk add --no-cache python3 git pipx"
# NixOS publishes no downloadable cloud qcow2 (its cloud images are Hydra-built
# AMIs), so the harness BUILDS one with nixos-generators — see build_nixos_image
# and linux-install-test-nixos.nix. It's externally-managed in its own way (no
# FHS ~/.local on PATH by default); `nix profile install` provisions the
# prerequisites into root's profile (the image enables flakes for this).
IMAGE_URL[nixos]="nix:build" # sentinel: ensure_base_image builds instead of downloading
SUM_URL[nixos]=""
SSH_USER[nixos]="root"
# Pin to a stable release: the guest's default `nixpkgs` registry is unstable,
# where pipx isn't in the binary cache and builds from source (its test suite
# currently fails to build). nixos-24.11 has these cached as substitutes.
PREREQ[nixos]="nix profile install nixpkgs/nixos-24.11#python3 nixpkgs/nixos-24.11#git nixpkgs/nixos-24.11#pipx"
ALL_DISTROS=(ubuntu fedora arch alpine nixos)
# --- guards ----------------------------------------------------------
require_linux() {
[ "$(uname -s)" = "Linux" ] \
|| { echo "error: this harness is Linux-only (uname is $(uname -s))" >&2; exit 1; }
}
require_kvm() {
[ -e /dev/kvm ] && [ -r /dev/kvm ] && [ -w /dev/kvm ] \
|| { echo "error: /dev/kvm is missing or not accessible (add yourself to the 'kvm' group)" >&2; exit 1; }
}
require_tools() {
local missing=()
command -v qemu-system-x86_64 >/dev/null 2>&1 || missing+=(qemu-system-x86_64)
command -v qemu-img >/dev/null 2>&1 || missing+=(qemu-img)
command -v cloud-localds >/dev/null 2>&1 || missing+=(cloud-localds)
command -v ssh >/dev/null 2>&1 || missing+=(ssh)
command -v curl >/dev/null 2>&1 || missing+=(curl)
if [ "${#missing[@]}" -ne 0 ]; then
echo "error: missing required tools: ${missing[*]}" >&2
echo " on NixOS: nix shell nixpkgs#qemu nixpkgs#cloud-utils nixpkgs#openssh nixpkgs#curl" >&2
exit 1
fi
}
known_distro() {
[ -n "${IMAGE_URL[$DISTRO]:-}" ] \
|| { echo "error: unknown distro '$DISTRO' (known: ${ALL_DISTROS[*]})" >&2; exit 1; }
}
# --- run-dir bookkeeping ---------------------------------------------
# The marker lets prereqs/run/status/down in separate invocations find the VM
# that `up` started. The test cycles set RUN_DIR themselves and never write it.
_marker() { echo "${TMPDIR:-/tmp}/bot-bottle-install-test.$DISTRO.run"; }
_ensure_run_dir() {
if [ -z "$RUN_DIR" ]; then
RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/bb-install-test.$DISTRO.XXXXXX")"
fi
mkdir -p "$RUN_DIR"
}
_load_run_dir() {
if [ -z "$RUN_DIR" ] && [ -f "$(_marker)" ]; then
RUN_DIR="$(cat "$(_marker)")"
fi
[ -n "$RUN_DIR" ] && [ -d "$RUN_DIR" ]
}
_ssh_key() { echo "$RUN_DIR/id_ed25519"; }
_overlay() { echo "$RUN_DIR/overlay.qcow2"; }
_seed() { echo "$RUN_DIR/seed.iso"; }
_pidfile() { echo "$RUN_DIR/qemu.pid"; }
_serial() { echo "$RUN_DIR/serial.log"; }
# --- ssh helpers -----------------------------------------------------
_ssh_opts() {
# No host-key pinning: the guest is thrown away every run.
printf '%s\0' \
-i "$(_ssh_key)" \
-p "$SSH_PORT" \
-o StrictHostKeyChecking=no \
-o UserKnownHostsFile=/dev/null \
-o LogLevel=ERROR \
-o ConnectTimeout=8 \
-o BatchMode=yes
}
guest() {
local -a opts
mapfile -d '' -t opts < <(_ssh_opts)
ssh "${opts[@]}" "${SSH_USER[$DISTRO]}@127.0.0.1" "$@"
}
wait_for_ssh() {
local deadline=$(( SECONDS + BOOT_TIMEOUT ))
echo "== waiting for SSH on 127.0.0.1:$SSH_PORT (up to ${BOOT_TIMEOUT}s) =="
while [ "$SECONDS" -lt "$deadline" ]; do
if guest true 2>/dev/null; then
echo " guest is up"
return 0
fi
# Bail early if QEMU has died — no point waiting out the timeout.
if [ -f "$(_pidfile)" ] && ! kill -0 "$(cat "$(_pidfile)")" 2>/dev/null; then
echo "error: QEMU exited before SSH came up; see $(_serial)" >&2
return 1
fi
sleep 3
done
echo "error: timed out waiting for SSH; see $(_serial)" >&2
return 1
}
# --- image cache -----------------------------------------------------
_base_image() {
# One cached file per distro. NixOS is built (not downloaded), so it has a
# fixed cache name; the rest are keyed by the image's basename so a URL bump
# lands as a new cache entry rather than a stale hit.
if [ "$DISTRO" = nixos ]; then
echo "$CACHE_DIR/nixos-built.qcow2"
return
fi
local url="${IMAGE_URL[$DISTRO]}"
echo "$CACHE_DIR/$DISTRO-$(basename "$url")"
}
# NixOS has no upstream cloud qcow2; build one with nixos-generators and copy it
# out of the (immutable, GC-able) store into the cache.
build_nixos_image() {
local out; out="$(_base_image)"
[ -f "$out" ] && { echo "== nixos base image cached: $out =="; return 0; }
command -v nixos-generate >/dev/null 2>&1 || {
echo "error: 'nixos-generate' is required to build the NixOS image" >&2
echo " run inside: nix shell nixpkgs#nixos-generators nixpkgs#qemu nixpkgs#cloud-utils" >&2
return 1
}
echo "== building NixOS cloud image with nixos-generators (first run is slow) =="
local link="$CACHE_DIR/nixos-result"
nixos-generate -f qcow --system x86_64-linux \
-c "$_SCRIPT_DIR/linux-install-test-nixos.nix" -o "$link"
cp -L "$link"/*.qcow2 "$out"
rm -f "$link"
echo " built + cached: $out"
}
verify_checksum() {
local file="$1" sums_url="${SUM_URL[$DISTRO]}" base
base="$(basename "${IMAGE_URL[$DISTRO]}")"
if [ "${BB_TEST_SKIP_VERIFY:-0}" = "1" ] || [ -z "$sums_url" ]; then
echo " checksum: SKIPPED (${sums_url:+set BB_TEST_SKIP_VERIFY=0 to enable}${sums_url:-no sums URL for $DISTRO})" >&2
return 0
fi
local want
# Vendors publish either "HASH filename" tables or a bare "HASH" (or a
# "SHA256 (file) = HASH" BSD line, e.g. Fedora). Cover all three.
local sums; sums="$(curl -fsSL "$sums_url")"
want="$(printf '%s\n' "$sums" | awk -v f="$base" '
$0 ~ f && $1 ~ /^[0-9a-fA-F]{64}$/ { print $1; exit } # GNU "hash file"
$1=="SHA256" && $0 ~ f { gsub(/[()]/,""); print $NF; exit } # BSD "SHA256 (file) = hash"
')"
[ -z "$want" ] && want="$(printf '%s\n' "$sums" | awk '/^[0-9a-fA-F]{64}$/ {print $1; exit}')"
[ -n "$want" ] || { echo "error: could not find a sha256 for $base in $sums_url" >&2; return 1; }
local got; got="$(sha256sum "$file" | awk '{print $1}')"
if [ "$want" != "$got" ]; then
echo "error: checksum mismatch for $base" >&2
echo " want $want" >&2
echo " got $got" >&2
return 1
fi
echo " checksum: OK"
}
ensure_base_image() {
mkdir -p "$CACHE_DIR"
if [ "$DISTRO" = nixos ]; then
build_nixos_image
return
fi
local base; base="$(_base_image)"
if [ -f "$base" ]; then
echo "== base image cached: $base =="
return 0
fi
echo "== downloading $DISTRO cloud image =="
echo " ${IMAGE_URL[$DISTRO]}"
# Download to a temp name and rename on success so an interrupted pull
# never poisons the cache with a truncated image.
local tmp="$base.partial"
curl -fSL --retry 3 -o "$tmp" "${IMAGE_URL[$DISTRO]}"
verify_checksum "$tmp"
mv "$tmp" "$base"
echo " cached: $base"
}
# --- cloud-init seed -------------------------------------------------
make_seed() {
ssh-keygen -t ed25519 -N '' -f "$(_ssh_key)" -q
local pub; pub="$(cat "$(_ssh_key).pub")"
local user="${SSH_USER[$DISTRO]}"
local user_data="$RUN_DIR/user-data"
if [ "$user" = "root" ]; then
# NixOS' cloud-init lands the key straight on root; no sudo needed.
cat > "$user_data" <<EOF
#cloud-config
ssh_authorized_keys:
- $pub
EOF
elif [ "$DISTRO" = alpine ]; then
# Alpine needs three things the systemd distros don't: cloud-init locks
# the account (lock_passwd), but Alpine's non-PAM sshd then refuses
# pubkey auth for a locked account — so give it a throwaway password;
# and OpenRC does not auto-start sshd after the key is injected, so
# start it via runcmd.
cat > "$user_data" <<EOF
#cloud-config
users:
- name: $user
sudo: ALL=(ALL) NOPASSWD:ALL
shell: /bin/sh
lock_passwd: false
ssh_authorized_keys:
- $pub
chpasswd:
expire: false
list: |
$user:bbtest
packages:
- sudo
runcmd:
- [ sh, -c, "rc-service sshd start 2>/dev/null || true" ]
EOF
else
cat > "$user_data" <<EOF
#cloud-config
users:
- name: $user
sudo: ALL=(ALL) NOPASSWD:ALL
shell: /bin/sh
lock_passwd: true
ssh_authorized_keys:
- $pub
EOF
fi
# NoCloud wants a meta-data with an instance-id, or cloud-init may not treat
# the seed as a new instance (the Alpine nocloud image is strict about this).
printf 'instance-id: bbtest-%s\nlocal-hostname: bbtest-%s\n' "$DISTRO" "$DISTRO" \
> "$RUN_DIR/meta-data"
cloud-localds "$(_seed)" "$user_data" "$RUN_DIR/meta-data"
}
# --- commands --------------------------------------------------------
cmd_up() {
require_linux; require_kvm; require_tools; known_distro
_ensure_run_dir
ensure_base_image
# Throwaway copy-on-write overlay: the cached base is read-only backing,
# all guest writes land in the overlay, and `down` deletes it. Resize so
# pipx + a git build have headroom (cloud-init grows the rootfs to fit).
qemu-img create -q -f qcow2 -F qcow2 -b "$(_base_image)" "$(_overlay)" "$DISK"
make_seed
echo "== booting $DISTRO VM (mem=${MEM_MB}M cpus=$CPUS, ssh -> :$SSH_PORT) =="
qemu-system-x86_64 \
-machine accel=kvm -cpu host -smp "$CPUS" -m "$MEM_MB" \
-display none -daemonize \
-pidfile "$(_pidfile)" \
-serial "file:$(_serial)" \
-drive "file=$(_overlay),if=virtio,format=qcow2" \
-drive "file=$(_seed),if=virtio,format=raw" \
-netdev "user,id=n0,hostfwd=tcp:127.0.0.1:$SSH_PORT-:22" \
-device virtio-net-pci,netdev=n0
# Only publish the marker (so a later prereqs/run/down finds this VM) when
# we aren't inside a test cycle, which manages its own RUN_DIR + teardown.
[ "$IN_TEST" = 1 ] || echo "$RUN_DIR" > "$(_marker)"
wait_for_ssh
if [ "$IN_TEST" != 1 ]; then
echo "== VM is up. Prepare it with: BB_TEST_DISTRO=$DISTRO $0 prereqs (or go straight to 'run') =="
fi
}
# Install install.sh's toolchain prerequisites (python3 + git + pipx) into the
# running VM. This is what separates `test-ready` from `test`, and it is a
# distinct sub-command so a manual up/prereqs/run/down cycle is possible.
cmd_prereqs() {
require_linux
_load_run_dir || { echo "error: no running VM for $DISTRO; run '$0 up' first" >&2; return 1; }
echo "== installing prerequisites (python3 + git + pipx) on $DISTRO =="
# Runs via the guest login shell; PREREQ is a client-side table value.
guest "${PREREQ[$DISTRO]}"
}
cmd_run() {
require_linux
_load_run_dir || { echo "error: no running VM for $DISTRO; run '$0 up' first" >&2; return 1; }
local spec_env=""
[ -n "${BOT_BOTTLE_INSTALL_SPEC:-}" ] \
&& spec_env="BOT_BOTTLE_INSTALL_SPEC='$BOT_BOTTLE_INSTALL_SPEC' "
echo "== installing bot-bottle as ${SSH_USER[$DISTRO]} =="
# Capture install.sh's exit code and full output rather than aborting on
# non-zero: on a bare host a clean prerequisite *decline* is sound, so the
# verdict step — not set -e — decides the outcome.
local rc
set +e
if [ -n "${BB_TEST_INSTALL_URL:-}" ]; then
guest "curl -fsSL '$BB_TEST_INSTALL_URL' | ${spec_env}sh" 2>&1 | tee "$RUN_DIR/install.log"
else
# Feed THIS checkout's install.sh in over stdin — the same `curl … | sh`
# shape a real user runs, and nothing is staged in the guest to leak.
guest "${spec_env}sh -s" < "$_REPO_ROOT/install.sh" 2>&1 | tee "$RUN_DIR/install.log"
fi
rc="${PIPESTATUS[0]}"
set -e
printf '%s\n' "$rc" > "$RUN_DIR/install.rc"
echo "== install.sh exited $rc =="
[ "$IN_TEST" = 1 ] \
|| echo "== verdict anytime with: BB_TEST_DISTRO=$DISTRO $0 status =="
}
# Quietly report whether a runnable bot-bottle entry point exists for the
# guest user, checking the pipx/pip locations install.sh may leave off PATH.
entry_point_runnable() {
# shellcheck disable=SC2016 # expand in the GUEST shell.
guest '
for bb in "$HOME/.local/bin/bot-bottle" "$HOME/.bot-bottle/venv/bin/bot-bottle" "$(command -v bot-bottle 2>/dev/null)"; do
[ -n "$bb" ] && [ -x "$bb" ] || continue
# --help exits 0 before any DB/migration/network work; it is the
# cheapest proof the package imports and the shim runs. (bot-bottle
# has no --version: an unknown arg would die non-zero.)
"$bb" --help >/dev/null 2>&1 && exit 0
done
exit 1
' >/dev/null 2>&1
}
# `bot-bottle doctor` in the guest, classified. doctor's own exit code
# conflates "is the install sound" with "is a backend ready to run a bottle" —
# and inside a plain VM no backend can be ready (no nested KVM/Docker), so the
# raw exit code is non-zero by design. This separates the two: an unhandled
# traceback, or a missing python/config line, is an install defect and fails;
# a not-ready backend is reported, not fatal (unless BB_TEST_REQUIRE_BACKEND=1,
# for a nested-virt host that can actually satisfy it).
doctor_in_guest() {
local out rc=0 bad=0
out="$(mktemp "${TMPDIR:-/tmp}/bb-doctor.XXXXXX")"
# shellcheck disable=SC2016 # $HOME/$bb must expand in the GUEST shell.
guest '
for bb in bot-bottle "$HOME/.local/bin/bot-bottle" "$HOME/.bot-bottle/venv/bin/bot-bottle"; do
if command -v "$bb" >/dev/null 2>&1; then
case "$bb" in
bot-bottle) : ;;
*) echo " (not on PATH — running $bb directly, as install.sh advises)" ;;
esac
exec "$bb" doctor
fi
done
echo " no bot-bottle entry point found for this user" >&2
exit 1
' >"$out" 2>&1 || rc=$?
cat "$out"
# An unhandled exception is always an install/product defect, never an
# environment fact — doctor's non-zero exit alone would not distinguish it.
if grep -q 'Traceback (most recent call last)' "$out"; then
echo " doctor crashed (traceback above) — a defect, not a missing prerequisite" >&2
bad=1
fi
grep -qE '^ok: +python:' "$out" \
|| { echo " doctor never reported a usable python" >&2; bad=1; }
grep -qE '^ok: +config:' "$out" \
|| { echo " doctor never reported a usable config dir" >&2; bad=1; }
if [ "$rc" -ne 0 ] && ! grep -qE '^(fail|warn): +backend' "$out"; then
echo " doctor failed for something other than backend readiness" >&2
bad=1
fi
local backend_ready=1
grep -qE '^fail: +backend' "$out" && backend_ready=0
rm -f "$out"
[ "$bad" -eq 0 ] || return 1
if [ "${BB_TEST_REQUIRE_BACKEND:-0}" = "1" ] && [ "$backend_ready" -eq 0 ]; then
echo " backend is not ready and BB_TEST_REQUIRE_BACKEND=1 — failing" >&2
return 1
fi
if [ "$backend_ready" -eq 1 ]; then
echo " doctor: install sound; a backend is ready"
else
echo " doctor: install sound; no backend ready (expected in a plain VM — install gate only)"
fi
return 0
}
# The prerequisite-decline messages install.sh prints via die(). On a bare host
# ANY of these means install.sh correctly refused rather than half-installing —
# a sound outcome for `test`.
PREREQ_ERR_RE='is required but was not found|or newer is required|git is required to install from|neither pipx nor a usable|externally managed \(PEP 668\)|is not on PATH'
# The verdict: is the install sound? Reads install.sh's captured exit code and
# output (from cmd_run) plus the guest's resulting state.
# - installed & runnable -> the verdict is doctor's soundness classification.
# - no entry point, but a recognized prerequisite decline, and declines are
# allowed (bare `test`, _REQUIRE_INSTALL=0) -> sound.
# - anything else -> not sound.
install_verdict() {
local rc=""
[ -f "$RUN_DIR/install.rc" ] && rc="$(cat "$RUN_DIR/install.rc")"
if [ "${rc:-1}" = 0 ] && entry_point_runnable; then
echo "doctor (in guest):"
doctor_in_guest
return $?
fi
if [ "${_REQUIRE_INSTALL:-0}" != "1" ] \
&& [ -n "$rc" ] && [ "$rc" != 0 ] \
&& [ -f "$RUN_DIR/install.log" ] \
&& grep -Eiq "$PREREQ_ERR_RE" "$RUN_DIR/install.log"; then
echo " install.sh declined with an actionable prerequisite error (rc=$rc)"
echo " — sound on a bare host; run 'test-ready' (or 'prereqs') to install."
return 0
fi
if [ "${_REQUIRE_INSTALL:-0}" = "1" ]; then
echo " prerequisites were provisioned, but install.sh left no runnable entry point (rc=${rc:-?})" >&2
else
echo " install.sh neither installed nor gave a recognized prerequisite error (rc=${rc:-?})" >&2
fi
return 1
}
cmd_status() {
require_linux
if ! _load_run_dir; then
echo "vm: no running VM for $DISTRO"
return 0
fi
if [ -f "$(_pidfile)" ] && kill -0 "$(cat "$(_pidfile)")" 2>/dev/null; then
echo "vm: $DISTRO running (pid $(cat "$(_pidfile)"), ssh :$SSH_PORT)"
else
echo "vm: $DISTRO run-dir present but QEMU not alive"
return 1
fi
if install_verdict; then
echo "OK[$DISTRO]: the install is sound"
return 0
fi
echo "FAIL[$DISTRO]: the install is not sound (see above)" >&2
return 1
}
cmd_down() {
require_linux
if ! _load_run_dir; then
echo "$DISTRO: nothing running"
return 0
fi
if [ -f "$(_pidfile)" ]; then
local pid; pid="$(cat "$(_pidfile)")"
if kill -0 "$pid" 2>/dev/null; then
kill "$pid" 2>/dev/null || true
for _ in 1 2 3 4 5; do kill -0 "$pid" 2>/dev/null || break; sleep 1; done
kill -9 "$pid" 2>/dev/null || true
fi
fi
# Deleting the overlay + seed is the reset; the read-only base stays cached.
rm -f "$(_overlay)" "$(_seed)" "$(_ssh_key)" "$(_ssh_key).pub" \
"$RUN_DIR/user-data" "$RUN_DIR/meta-data" \
"$RUN_DIR/install.rc" "$RUN_DIR/install.log" "$(_serial)"
# Only remove a scratch dir we created (leave a user-provided one alone).
[ -n "${BB_TEST_RUN_DIR:-}" ] || rmdir "$RUN_DIR" 2>/dev/null || true
rm -f "$(_marker)"
echo "removed $DISTRO VM and its overlay — install surface is clean."
}
cmd_ssh() {
require_linux
_load_run_dir || { echo "error: no running VM for $DISTRO" >&2; return 1; }
local -a opts
mapfile -d '' -t opts < <(_ssh_opts)
exec ssh -t "${opts[@]}" "${SSH_USER[$DISTRO]}@127.0.0.1"
}
# Teardown half of the test cycles, armed the moment the VM exists so a failure
# or a Ctrl-C still leaves nothing running.
_test_teardown() {
local rc=$?
trap - EXIT INT TERM
if [ "${BB_TEST_KEEP:-0}" = "1" ]; then
echo
echo "== [$_STEPS/$_STEPS] down: SKIPPED (BB_TEST_KEEP=1) =="
echo " VM still up; remove with: BB_TEST_DISTRO=$DISTRO $0 down"
echo " ssh in with: BB_TEST_RUN_DIR=$RUN_DIR BB_TEST_DISTRO=$DISTRO $0 ssh"
exit "$rc"
fi
echo
echo "== [$_STEPS/$_STEPS] down =="
cmd_down || rc=1
if [ "$rc" -eq 0 ]; then
echo
echo "PASS[$DISTRO]: $_PASS_CLAIM"
else
echo
echo "FAIL[$DISTRO]: see above (the VM was torn down regardless)." >&2
fi
exit "$rc"
}
# The two variants differ only in whether the prerequisites get installed
# before install.sh runs, which is exactly the question each one asks:
#
# test a bare host — assert install.sh is SOUND (installs cleanly, or
# declines with an actionable prerequisite error).
# test-ready prerequisites satisfied — assert install.sh actually LANDS.
_test_cycle() {
local with_prereqs="$1"
require_linux; require_kvm; require_tools; known_distro
IN_TEST=1
RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/bb-install-test.$DISTRO.XXXXXX")"
# Arm teardown BEFORE cmd_up: its wait_for_ssh can fail after QEMU is
# already running (e.g. a guest that never opens SSH), and without the trap
# in place that would leak the VM.
trap _test_teardown EXIT INT TERM
echo "== [1/$_STEPS] up ($DISTRO) =="
cmd_up
local step=2
if [ "$with_prereqs" = 1 ]; then
echo
echo "== [$step/$_STEPS] prereqs =="
cmd_prereqs || { echo "error: could not install prerequisites (see above)." >&2; return 1; }
step=$(( step + 1 ))
fi
echo
echo "== [$step/$_STEPS] run =="
cmd_run
step=$(( step + 1 ))
echo
echo "== [$step/$_STEPS] verdict =="
# install.sh exits 0 even when doctor reports unmet prerequisites, so the
# install succeeding is not the verdict — this is.
cmd_status || {
echo "error: the install is not sound for $DISTRO (see above)." >&2
echo " re-run with BB_TEST_KEEP=1 to keep the VM and dig in." >&2
return 1
}
}
cmd_test() {
_STEPS=4
_PASS_CLAIM="on a bare $DISTRO host, install.sh behaves soundly."
_test_cycle 0
}
cmd_test_ready() {
_STEPS=5
_PASS_CLAIM="a $DISTRO host with prerequisites satisfied installs bot-bottle cleanly."
# Prerequisites are provisioned, so a graceful decline is no longer an
# acceptable outcome — the install must actually land.
_REQUIRE_INSTALL=1
_test_cycle 1
}
cmd_test_all() {
require_linux; require_kvm; require_tools
local -a passed=() failed=()
local port="$SSH_PORT"
local script
script="$_SCRIPT_DIR/$(basename "${BASH_SOURCE[0]}")"
# Every distro × both variants. Each cell gets its own forwarded port so a
# leftover from a prior cell can't collide, and its own subshell so one
# cell's failure (or teardown trap) doesn't abort the matrix.
for sub in test test-ready; do
for d in "${ALL_DISTROS[@]}"; do
echo
echo "########################################################"
echo "# $d ($sub)"
echo "########################################################"
if ( DISTRO="$d" SSH_PORT="$port" \
BB_TEST_DISTRO="$d" BB_TEST_SSH_PORT="$port" \
bash "$script" "$sub" ); then
passed+=("$d/$sub")
else
failed+=("$d/$sub")
fi
port=$(( port + 1 ))
done
done
echo
echo "== matrix summary =="
echo " PASS: ${passed[*]:-(none)}"
echo " FAIL: ${failed[*]:-(none)}"
[ "${#failed[@]}" -eq 0 ]
}
case "${1:-}" in
test) cmd_test ;;
test-ready) cmd_test_ready ;;
test-all) cmd_test_all ;;
up) cmd_up ;;
prereqs) cmd_prereqs ;;
run) cmd_run ;;
status) cmd_status ;;
down) cmd_down ;;
ssh) cmd_ssh ;;
*) echo "usage: $0 {test|test-ready|test-all|up|prereqs|run|status|down|ssh} (distro via BB_TEST_DISTRO)" >&2; exit 2 ;;
esac
-98
View File
@@ -1,98 +0,0 @@
"""Unit: how the CLI tells users to re-run it under sudo.
`sudo bot-bottle ` is wrong for the users who followed the documented
install: sudo's secure_path excludes ~/.local/bin, where both pipx and
install.sh put the entry point. These lock in the absolute-path form.
"""
from __future__ import annotations
import contextlib
import io
import os
import tempfile
import unittest
from pathlib import Path
from unittest import mock
from bot_bottle import invocation
class TestSelfPath(unittest.TestCase):
def test_absolute_argv0_is_used_as_is(self):
with mock.patch.object(invocation.sys, "argv", ["/opt/venv/bin/bot-bottle"]):
self.assertEqual("/opt/venv/bin/bot-bottle", invocation.self_path())
def test_bare_name_is_resolved_through_path(self):
# The case that matters: invoked as `bot-bottle`, installed in a
# directory sudo would drop.
with tempfile.TemporaryDirectory() as d:
entry = Path(d, "bot-bottle")
entry.write_text("#!/bin/sh\n")
entry.chmod(0o755)
with mock.patch.object(invocation.sys, "argv", ["bot-bottle"]), \
mock.patch.dict(os.environ, {"PATH": d}):
self.assertEqual(str(entry), invocation.self_path())
def test_relative_path_is_made_absolute(self):
with tempfile.TemporaryDirectory() as d:
entry = Path(d, "bot-bottle")
entry.write_text("#!/bin/sh\n")
entry.chmod(0o755)
cwd = os.getcwd()
try:
os.chdir(d)
with mock.patch.object(invocation.sys, "argv", ["./bot-bottle"]):
self.assertTrue(os.path.isabs(invocation.self_path()))
finally:
os.chdir(cwd)
def test_unresolvable_entry_point_falls_back_to_the_name(self):
# `python -m`-style invocation, or an argv[0] that no longer exists.
# A slightly wrong hint beats a traceback raised while reporting some
# unrelated problem.
with mock.patch.object(invocation.sys, "argv", ["/nonexistent/gone"]), \
mock.patch.object(invocation.shutil, "which", return_value=None):
self.assertEqual("/nonexistent/gone", invocation.self_path())
with mock.patch.object(invocation.sys, "argv", [""]), \
mock.patch.object(invocation.shutil, "which", return_value=None):
self.assertEqual("bot-bottle", invocation.self_path())
class TestSudoCommand(unittest.TestCase):
def test_names_an_absolute_path_not_the_bare_command(self):
with mock.patch.object(invocation, "self_path",
return_value="/home/u/.local/bin/bot-bottle"):
cmd = invocation.sudo_command("backend", "setup", "--backend=firecracker")
self.assertEqual(
"sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker",
cmd,
)
# The regression this exists to prevent.
self.assertNotIn("sudo bot-bottle", cmd)
class TestFirecrackerSetupUsesIt(unittest.TestCase):
def test_root_reinvocation_hint_names_an_absolute_path(self):
# The message that prompted all this. Drive the real code path rather
# than scanning the source, which would also match the comment
# explaining why the bare form is wrong.
from bot_bottle.backend.firecracker import setup as fc_setup
err, out = io.StringIO(), io.StringIO()
with mock.patch.object(fc_setup.os, "geteuid", return_value=501), \
mock.patch.object(fc_setup.invocation, "self_path",
return_value="/home/u/.local/bin/bot-bottle"), \
contextlib.redirect_stderr(err), contextlib.redirect_stdout(out):
fc_setup._setup_systemd()
printed = err.getvalue()
self.assertIn(
"sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker",
printed,
)
self.assertNotIn("sudo bot-bottle", printed)
if __name__ == "__main__":
unittest.main()
+25
View File
@@ -151,6 +151,31 @@ class TestReprovisionGateway(unittest.TestCase):
self.c.reprovision_gateway("b1", "key")
class TestUpdateAgentSecret(unittest.TestCase):
def setUp(self) -> None:
self.c = OrchestratorClient("http://orch:8080")
def test_success_posts_single_secret(self) -> None:
with patch(_URLOPEN, return_value=_resp(200, {"updated": True})) as opened:
self.assertTrue(self.c.update_agent_secret("b1", "A", "fresh", "key"))
request = opened.call_args.args[0]
self.assertEqual("POST", request.get_method())
self.assertTrue(request.full_url.endswith("/bottles/b1/secret"))
self.assertEqual(
{"name": "A", "value": "fresh", "env_var_secret": "key"},
json.loads(request.data),
)
def test_unknown_bottle_is_false(self) -> None:
with patch(_URLOPEN, side_effect=_http_error(404)):
self.assertFalse(self.c.update_agent_secret("b1", "A", "v", "key"))
def test_other_status_raises(self) -> None:
with patch(_URLOPEN, side_effect=_http_error(400)):
with self.assertRaises(OrchestratorClientError):
self.c.update_agent_secret("b1", "A", "v", "key")
class TestHealthAndPolicy(unittest.TestCase):
def setUp(self) -> None:
self.c = OrchestratorClient("http://orch:8080")
@@ -199,6 +199,20 @@ class TestAgentSecrets(unittest.TestCase):
got = self.store.get_agent_secrets("bottle-1")
self.assertEqual({"K": "new", "K2": "v2"}, got)
def test_store_agent_secret_upserts_one_key_leaving_others(self) -> None:
self.store.store_agent_secrets("bottle-1", {"K": "v1", "K2": "v2"})
self.store.store_agent_secret("bottle-1", "K", "v1-new") # update existing
self.store.store_agent_secret("bottle-1", "K3", "v3") # insert new
self.assertEqual(
{"K": "v1-new", "K2": "v2", "K3": "v3"},
self.store.get_agent_secrets("bottle-1"),
)
def test_store_agent_secret_does_not_duplicate_rows(self) -> None:
self.store.store_agent_secret("bottle-1", "K", "v1")
self.store.store_agent_secret("bottle-1", "K", "v2")
self.assertEqual({"K": "v2"}, self.store.get_agent_secrets("bottle-1"))
def test_delete_removes_secrets(self) -> None:
self.store.store_agent_secrets("bottle-1", {"K": "v"})
self.store.delete_agent_secrets("bottle-1")
+41
View File
@@ -125,6 +125,47 @@ class TestDispatch(unittest.TestCase):
{"EGRESS_TOKEN_0": "upstream-secret"}, self.orch.tokens_for(bottle_id),
)
def test_update_agent_secret_in_place(self) -> None:
key = base64.urlsafe_b64encode(b"unit-test-key").rstrip(b"=").decode()
status, payload = dispatch(
self.orch, "POST", "/bottles", _body({
"source_ip": "10.243.0.21",
"tokens": {"A": "old", "B": "keep"},
"env_var_secret": key,
}),
)
self.assertEqual(201, status)
bottle_id = payload["bottle_id"]
assert isinstance(bottle_id, str)
status, response = dispatch(
self.orch, "POST", f"/bottles/{bottle_id}/secret",
_body({"name": "A", "value": "fresh", "env_var_secret": key}),
)
self.assertEqual((200, {"updated": True}), (status, response))
self.assertEqual({"A": "fresh", "B": "keep"}, self.orch.tokens_for(bottle_id))
def test_update_agent_secret_validates_and_unknown_bottle(self) -> None:
status, _ = dispatch(self.orch, "POST", "/bottles/b1/secret", b"not-json")
self.assertEqual(400, status)
status, _ = dispatch(
self.orch, "POST", "/bottles/b1/secret", _body({"name": "A"}),
)
self.assertEqual(400, status) # value + env_var_secret missing
status, _ = dispatch(
self.orch, "POST", "/bottles/ghost/secret",
_body({"name": "A", "value": "v", "env_var_secret": "k"}),
)
self.assertEqual(404, status)
def test_update_agent_secret_is_cli_only(self) -> None:
# A data-plane (`gateway`) caller must not set a bottle's tokens.
status, _ = dispatch(
self.orch, "POST", "/bottles/b1/secret",
_body({"name": "A", "value": "v", "env_var_secret": "k"}),
role=ROLE_GATEWAY,
)
self.assertEqual(403, status)
def test_reprovision_validates_request_and_missing_rows(self) -> None:
status, _ = dispatch(
self.orch, "POST", "/bottles/b1/reprovision_gateway", b"not-json",
+19
View File
@@ -103,6 +103,25 @@ class TestOrchestrator(unittest.TestCase):
self.assertTrue(self.orch.reprovision_from_secret(rec.bottle_id, key))
self.assertEqual({"EGRESS_TOKEN_0": "secret"}, self.orch.tokens_for(rec.bottle_id))
def test_update_agent_secret_refreshes_one_token_in_place(self) -> None:
key = new_env_var_secret()
rec = self.orch.launch_bottle(
"10.243.0.20", tokens={"A": "old-a", "B": "keep-b"}, env_var_secret=key,
)
self.assertTrue(self.orch.update_agent_secret(rec.bottle_id, "A", "new-a", key))
# In memory: A refreshed, B left untouched.
self.assertEqual({"A": "new-a", "B": "keep-b"}, self.orch.tokens_for(rec.bottle_id))
# At rest: the whole set stays decryptable with the SAME env_var_secret,
# so a later reprovision restores the refreshed value (not the launch one).
self.orch._tokens.clear()
self.assertTrue(self.orch.reprovision_from_secret(rec.bottle_id, key))
self.assertEqual({"A": "new-a", "B": "keep-b"}, self.orch.tokens_for(rec.bottle_id))
def test_update_agent_secret_unknown_bottle_is_false(self) -> None:
self.assertFalse(
self.orch.update_agent_secret("ghost", "A", "v", new_env_var_secret())
)
def test_reprovision_rejects_missing_rows_and_wrong_key(self) -> None:
self.assertFalse(self.orch.reprovision_from_secret("missing", new_env_var_secret()))
key = "AQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQE"