Compare commits
8 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| ec38458c94 | |||
| f6df85a8cd | |||
| 9bf2961d13 | |||
| 6385752040 | |||
| 496608fc25 | |||
| 218f29cb05 | |||
| 1518f73de5 | |||
| 9d82535390 |
@@ -19,6 +19,7 @@ from __future__ import annotations
|
||||
import fcntl
|
||||
import hashlib
|
||||
import os
|
||||
import re
|
||||
import shlex
|
||||
import shutil
|
||||
import subprocess
|
||||
@@ -41,19 +42,72 @@ _BUILD_TIMEOUT_SECONDS = 900.0
|
||||
|
||||
|
||||
def _dockerfile_hash(dockerfile: Path) -> str:
|
||||
"""The Dockerfile's content hash. The shipped agent Dockerfiles COPY
|
||||
nothing from the build context (see .dockerignore), so their content fully
|
||||
determines the built image; a Dockerfile that adds COPY will want the
|
||||
context folded in here too."""
|
||||
"""The Dockerfile's content hash. Agent Dockerfiles mostly COPY nothing
|
||||
from the build context, so their text nearly determines the built image;
|
||||
any files they *do* COPY are folded into `_rootfs_digest` (so a changed
|
||||
input busts the cache) and shipped to the VM-side context by
|
||||
`_send_build_context`."""
|
||||
return hashlib.sha256(dockerfile.read_bytes()).hexdigest()[:16]
|
||||
|
||||
|
||||
def _context_copy_sources(dockerfile: Path) -> list[str]:
|
||||
"""The build-context-relative paths a Dockerfile ``COPY``s in.
|
||||
|
||||
Agent Dockerfiles are meant to COPY nothing from the context (the VM-side
|
||||
build ships only the Dockerfile), but one may pin an input by COPYing a
|
||||
committed file — a checksum list, an npm lockfile. Return those source
|
||||
paths so the builder can both ship them to the VM context and fold them
|
||||
into the cache key. ``COPY --from=<stage>`` reads a build stage, not the
|
||||
context, so it is excluded; the JSON/exec COPY form is unused by the
|
||||
shipped images and is skipped rather than mis-parsed."""
|
||||
joined = re.sub(r"\\\n", " ", dockerfile.read_text(encoding="utf-8"))
|
||||
sources: list[str] = []
|
||||
for line in joined.splitlines():
|
||||
stripped = line.strip()
|
||||
if not re.match(r"(?i)^COPY\s", stripped):
|
||||
continue
|
||||
tokens = stripped.split()[1:]
|
||||
if any(t.startswith("--from=") for t in tokens):
|
||||
continue
|
||||
args = [t for t in tokens if not t.startswith("--")]
|
||||
if len(args) < 2 or args[0].startswith("["):
|
||||
continue
|
||||
sources.extend(args[:-1])
|
||||
return sources
|
||||
|
||||
|
||||
def _context_files(dockerfile: Path) -> list[tuple[str, Path]]:
|
||||
"""``(context-relative path, host path)`` for every existing file a
|
||||
Dockerfile COPYs from the build root — globs expanded, sorted, de-duped.
|
||||
Absolute or traversing (`..`) sources are dropped: the shipped context
|
||||
only ever mirrors files under the build root."""
|
||||
root = resources.build_root()
|
||||
resolved: dict[str, Path] = {}
|
||||
for src in _context_copy_sources(dockerfile):
|
||||
if src.startswith("/") or ".." in Path(src).parts:
|
||||
continue
|
||||
if any(ch in src for ch in "*?["):
|
||||
matches = [p for p in root.glob(src) if p.is_file()]
|
||||
else:
|
||||
candidate = root / src
|
||||
if candidate.is_dir():
|
||||
matches = [path for path in candidate.rglob("*") if path.is_file()]
|
||||
else:
|
||||
matches = [candidate] if candidate.is_file() else []
|
||||
for path in matches:
|
||||
resolved[str(path.relative_to(root))] = path
|
||||
return sorted(resolved.items())
|
||||
|
||||
|
||||
def _rootfs_digest(dockerfile: Path) -> str:
|
||||
"""Cache key for the built AND boot-injected agent rootfs. Two inputs
|
||||
determine the on-disk rootfs: the Dockerfile (the image) and the guest init
|
||||
injected into it (`util._GUEST_INIT`). Folding the init in means a fix to
|
||||
"""Cache key for the built AND boot-injected agent rootfs. Its inputs are
|
||||
the Dockerfile (the image), the centralized build args, the guest init
|
||||
injected into it (`util._GUEST_INIT`), and the content of any files the
|
||||
Dockerfile COPYs from the build context. Folding the init in means a fix to
|
||||
it — e.g. making /tmp world-writable — busts the cache instead of silently
|
||||
reusing a stale rootfs built with the old init."""
|
||||
reusing a stale rootfs built with the old init; folding the COPYed context
|
||||
files in means a repinned input (e.g. a changed checksum list) rebuilds
|
||||
rather than reusing a rootfs baked from the old bytes."""
|
||||
h = hashlib.sha256()
|
||||
h.update(_dockerfile_hash(dockerfile).encode())
|
||||
h.update(b"\0")
|
||||
@@ -63,6 +117,11 @@ def _rootfs_digest(dockerfile: Path) -> str:
|
||||
h.update(value.encode())
|
||||
h.update(b"\0")
|
||||
h.update(util._GUEST_INIT.encode())
|
||||
for rel, path in _context_files(dockerfile):
|
||||
h.update(b"\0")
|
||||
h.update(rel.encode())
|
||||
h.update(b"\0")
|
||||
h.update(path.read_bytes())
|
||||
return h.hexdigest()[:16]
|
||||
|
||||
|
||||
@@ -154,6 +213,7 @@ def _build_in_infra(
|
||||
if prep.returncode != 0:
|
||||
die(f"preparing build dir in the infra VM failed: {prep.stderr.strip()}")
|
||||
_send_dockerfile(key, ip, dockerfile, ctx)
|
||||
_send_build_context(key, ip, dockerfile, ctx)
|
||||
_buildah_build(
|
||||
key,
|
||||
ip,
|
||||
@@ -197,6 +257,40 @@ def _send_dockerfile(private_key: Path, guest_ip: str, dockerfile: Path, ctx: st
|
||||
f"{proc.stderr.decode(errors='replace').strip()}")
|
||||
|
||||
|
||||
def _send_build_context(private_key: Path, guest_ip: str, dockerfile: Path, ctx: str) -> None:
|
||||
"""Ship the files ``dockerfile`` COPYs from the build root into the infra
|
||||
VM's ``{ctx}/ctx``, preserving their build-root-relative paths.
|
||||
|
||||
Usually a no-op — agent Dockerfiles COPY nothing — so `{ctx}/ctx` stays the
|
||||
empty context the build otherwise runs against. It exists so a Dockerfile
|
||||
that pins an input by COPYing a committed file (a checksum list, an npm
|
||||
lockfile) still finds that file in the VM-side context. Streamed as a tar
|
||||
so directories and multiple files land in one round trip."""
|
||||
files = _context_files(dockerfile)
|
||||
if not files:
|
||||
return
|
||||
root = resources.build_root()
|
||||
rels = [rel for rel, _ in files]
|
||||
tar = subprocess.Popen(
|
||||
["tar", "-C", str(root), "-cf", "-", "--", *rels],
|
||||
stdout=subprocess.PIPE,
|
||||
)
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
util.ssh_base_argv(private_key, guest_ip) + [f"tar -C {ctx}/ctx -xf -"],
|
||||
stdin=tar.stdout, capture_output=True, timeout=120, check=False,
|
||||
)
|
||||
finally:
|
||||
if tar.stdout is not None:
|
||||
tar.stdout.close()
|
||||
tar.wait()
|
||||
if tar.returncode != 0:
|
||||
die(f"packing the agent build context failed (tar exit {tar.returncode})")
|
||||
if proc.returncode != 0:
|
||||
die("sending the agent build context to the infra VM failed: "
|
||||
f"{proc.stderr.decode(errors='replace').strip() or '<no stderr>'}")
|
||||
|
||||
|
||||
def _buildah_build(
|
||||
private_key: Path,
|
||||
guest_ip: str,
|
||||
|
||||
@@ -20,6 +20,7 @@ import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from ... import invocation
|
||||
from ... import resources
|
||||
from . import netpool
|
||||
from . import util
|
||||
@@ -179,9 +180,11 @@ def _setup_systemd() -> None:
|
||||
f"sudo systemctl daemon-reload\n"
|
||||
f"sudo systemctl enable --now {netpool.SYSTEMD_UNIT}\n"
|
||||
)
|
||||
# Absolute path, not `sudo bot-bottle`: sudo's secure_path drops
|
||||
# ~/.local/bin, where both pipx and install.sh put the entry point.
|
||||
sys.stderr.write(
|
||||
f"\n(Or re-run this as root to install it directly: "
|
||||
f"sudo bot-bottle backend setup --backend=firecracker)\n"
|
||||
f"\n(Or re-run this as root to install it directly:\n"
|
||||
f" {invocation.sudo_command('backend', 'setup', '--backend=firecracker')})\n"
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
"""How to tell a user to re-run this CLI.
|
||||
|
||||
`bot-bottle …` is the right thing to print for anything the user runs as
|
||||
themselves — it is on their PATH, since that is how they got here.
|
||||
|
||||
Under `sudo` it is not. sudo replaces PATH with sudoers' `secure_path`
|
||||
(`/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin` on Debian and
|
||||
Ubuntu, similar elsewhere), which deliberately excludes user-writable
|
||||
directories. Both supported install paths put the entry point in one of those:
|
||||
pipx uses `~/.local/bin`, and `install.sh`'s venv fallback symlinks there too.
|
||||
So `sudo bot-bottle …` fails with "command not found" for exactly the users who
|
||||
followed the documented install, while working for anyone who happened to
|
||||
install system-wide — which is why it survives review so easily.
|
||||
|
||||
Naming the absolute path sidesteps secure_path entirely.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
|
||||
|
||||
def self_path() -> str:
|
||||
"""Absolute path to this CLI's entry point.
|
||||
|
||||
Falls back to the bare name when the entry point cannot be resolved (an
|
||||
unusual invocation such as `python -m`), because a slightly wrong hint is
|
||||
better than a traceback while reporting an unrelated problem.
|
||||
"""
|
||||
argv0 = sys.argv[0] or "bot-bottle"
|
||||
resolved = shutil.which(argv0) or argv0
|
||||
if not os.path.isabs(resolved):
|
||||
if os.path.exists(resolved):
|
||||
resolved = os.path.abspath(resolved)
|
||||
else:
|
||||
return "bot-bottle"
|
||||
return resolved
|
||||
|
||||
|
||||
def sudo_command(*args: str) -> str:
|
||||
"""A copy-pasteable `sudo …` invocation of this CLI.
|
||||
|
||||
>>> sudo_command("backend", "setup", "--backend=firecracker")
|
||||
'sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker'
|
||||
"""
|
||||
return " ".join(["sudo", self_path(), *args])
|
||||
@@ -50,12 +50,6 @@ class ReprovisionBody(_StrictModel):
|
||||
env_var_secret: StrictStr
|
||||
|
||||
|
||||
class SecretBody(_StrictModel):
|
||||
name: StrictStr
|
||||
value: StrictStr
|
||||
env_var_secret: StrictStr
|
||||
|
||||
|
||||
class ReconcileBody(_StrictModel):
|
||||
live_source_ips: list[StrictStr]
|
||||
grace_seconds: float | None = None
|
||||
@@ -265,21 +259,6 @@ def create_app(orch: OrchestratorCore, *, signing_key: str) -> FastAPI:
|
||||
return {"reprovisioned": True}
|
||||
raise HTTPException(404, "no stored secrets for this bottle")
|
||||
|
||||
@app.post("/bottles/{bottle_id}/secret")
|
||||
def update_secret(bottle_id: str, body: SecretBody) -> dict[str, object]:
|
||||
# Update ONE egress token for a running bottle in place — the
|
||||
# single-secret form of reprovision, for refreshing a short-lived host
|
||||
# credential (e.g. the Codex access token) without a relaunch. cli-only
|
||||
# (not in _GATEWAY_ROUTES): a bottle must never set its own tokens.
|
||||
if orch.update_agent_secret(
|
||||
bottle_id,
|
||||
_required(body.name, "name"),
|
||||
_required(body.value, "value"),
|
||||
_required(body.env_var_secret, "env_var_secret"),
|
||||
):
|
||||
return {"updated": True}
|
||||
raise HTTPException(404, "no such bottle")
|
||||
|
||||
@app.delete("/bottles/{bottle_id}")
|
||||
def teardown(bottle_id: str) -> dict[str, object]:
|
||||
if orch.teardown_bottle(bottle_id):
|
||||
|
||||
@@ -184,27 +184,6 @@ class OrchestratorClient:
|
||||
)
|
||||
return True
|
||||
|
||||
def update_agent_secret(
|
||||
self, bottle_id: str, name: str, value: str, env_var_secret: str,
|
||||
) -> bool:
|
||||
"""Update ONE egress token for a running bottle in place
|
||||
(`POST /bottles/<id>/secret`) — the single-secret form of
|
||||
`reprovision_gateway`, for pushing a freshly-refreshed host credential
|
||||
into a bottle without a relaunch. Returns True on success, False when the
|
||||
orchestrator doesn't know the bottle (404)."""
|
||||
status, _ = self._request(
|
||||
"POST",
|
||||
f"/bottles/{bottle_id}/secret",
|
||||
{"name": name, "value": value, "env_var_secret": env_var_secret},
|
||||
)
|
||||
if status == 404:
|
||||
return False
|
||||
if not 200 <= status < 300:
|
||||
raise OrchestratorClientError(
|
||||
f"update_agent_secret {bottle_id}: HTTP {status}"
|
||||
)
|
||||
return True
|
||||
|
||||
def teardown_bottle(self, bottle_id: str) -> bool:
|
||||
"""Tear a bottle down (`DELETE /bottles/<id>`). False if the
|
||||
orchestrator didn't know it (404) — idempotent for cleanup paths."""
|
||||
|
||||
@@ -379,30 +379,6 @@ class OrchestratorCore:
|
||||
self._tokens[bottle_id] = decrypted
|
||||
return True
|
||||
|
||||
def update_agent_secret(
|
||||
self, bottle_id: str, name: str, value: str, env_var_secret: str,
|
||||
) -> bool:
|
||||
"""Update ONE egress token for a known bottle in place — the
|
||||
single-secret form of ``reprovision_from_secret``.
|
||||
|
||||
Sets the in-memory token AND upserts the single re-encrypted row under
|
||||
*env_var_secret* (the same key the rest of the rows are encrypted with, so
|
||||
the whole set stays decryptable by a later ``reprovision_from_secret``),
|
||||
leaving every other token untouched. Returns False if the bottle is
|
||||
unknown.
|
||||
|
||||
Unlike ``reprovision_from_secret`` (which restores the values captured at
|
||||
launch), this pushes a *caller-supplied* value — used to refresh a
|
||||
short-lived host credential (e.g. the Codex access token) into a
|
||||
still-running bottle without a relaunch."""
|
||||
from .store.secret_store import encrypt_value
|
||||
if self.registry.get(bottle_id) is None:
|
||||
return False
|
||||
self._tokens.setdefault(bottle_id, {})[name] = value
|
||||
self.registry.store_agent_secret(
|
||||
bottle_id, name, encrypt_value(env_var_secret, value))
|
||||
return True
|
||||
|
||||
# --- consolidated gateway ----------------------------------------------
|
||||
|
||||
def gateway_status(self) -> dict[str, object]:
|
||||
|
||||
@@ -370,30 +370,6 @@ class RegistryStore(DbStore):
|
||||
)
|
||||
self._chmod()
|
||||
|
||||
def store_agent_secret(
|
||||
self,
|
||||
bottle_id: str,
|
||||
key: str,
|
||||
encrypted_value: str,
|
||||
secret_type: str = "injected_env_var",
|
||||
) -> None:
|
||||
"""Upsert ONE encrypted secret row (env-var name → ciphertext) for
|
||||
*bottle_id*, leaving the bottle's other secrets untouched — the per-key
|
||||
counterpart of ``store_agent_secrets``' replace-all. Delete-then-insert
|
||||
because the table carries no unique constraint to `ON CONFLICT` against."""
|
||||
with self._connection() as conn:
|
||||
conn.execute(
|
||||
"DELETE FROM bottled_agent_secrets "
|
||||
"WHERE bottled_agent_id = ? AND key = ? AND type = ?",
|
||||
(bottle_id, key, secret_type),
|
||||
)
|
||||
conn.execute(
|
||||
"INSERT INTO bottled_agent_secrets "
|
||||
"(bottled_agent_id, key, value, type) VALUES (?, ?, ?, ?)",
|
||||
(bottle_id, key, encrypted_value, secret_type),
|
||||
)
|
||||
self._chmod()
|
||||
|
||||
def get_agent_secrets(
|
||||
self,
|
||||
bottle_id: str,
|
||||
|
||||
@@ -1,117 +0,0 @@
|
||||
# PRD 0081: Reprovision gateway-dependent state on gateway bring-up
|
||||
|
||||
- **Status:** Active
|
||||
- **Author:** claude
|
||||
- **Created:** 2026-07-26
|
||||
- **Issue:** #516
|
||||
|
||||
## Summary
|
||||
|
||||
When the gateway is (re)built, reconcile every already-running bottle against the
|
||||
fresh gateway instead of persisting the gateway's state. On a gateway cold boot,
|
||||
the gateway reconciles all live bottles in one flow: **replace** each agent's
|
||||
trusted CA with the freshly-minted gateway CA, **re-provision** each bottle's
|
||||
git-gate repos + creds onto the gateway, and **restore** each bottle's egress
|
||||
tokens. One mechanism across all three services; the CA rotates for free on every
|
||||
bring-up.
|
||||
|
||||
## Problem
|
||||
|
||||
A gateway rebuild/restart silently breaks every already-running bottle:
|
||||
|
||||
- **CA (#510).** The gateway's mitmproxy mints a new CA on a fresh rootfs; agents
|
||||
still trust the old one, so egress fails TLS verification (`SSL certificate
|
||||
verification failed`).
|
||||
- **git-gate (#512).** Per-bottle bare repos (`/git/<id>`) + deploy creds
|
||||
(`/git-gate/creds/<id>`) live in the gateway's ephemeral rootfs; a rebuild wipes
|
||||
them and the agent 404s on fetch/push.
|
||||
|
||||
Both are the same root cause: per-bottle gateway-dependent state is provisioned
|
||||
**once, at bottle launch**, and nothing restores it for already-running bottles
|
||||
when the gateway comes back. The existing launch-time reprovision
|
||||
(`reprovision_bottles`) restores **only** egress tokens, and only as a side effect
|
||||
of the *next* launch.
|
||||
|
||||
Persisting the state (a host bind-mount / a per-VM data volume per service) was
|
||||
prototyped and rejected: it differs per service (a volume for the CA, another for
|
||||
git-gate, reprovision for tokens), it pins the CA static forever (no rotation),
|
||||
and it adds volume surface that `docker volume prune` / a wiped cache can silently
|
||||
destroy (the original #450 failure mode).
|
||||
|
||||
## Goals / Success Criteria
|
||||
|
||||
- A gateway (re)boot restores **all** running bottles' gateway-dependent state
|
||||
with no manual step and no relaunch — the agent's next egress / fetch / push
|
||||
just works.
|
||||
- **One** reconcile flow covering CA, git-gate, and egress tokens, rather than a
|
||||
different mechanism per service.
|
||||
- The CA **rotates** on every gateway bring-up (no long-lived CA), distributed to
|
||||
running agents by the same reconcile.
|
||||
- Per-bottle failures are tolerated: one unreachable or malformed bottle does not
|
||||
block the others or the gateway coming up.
|
||||
- **All backends** (Firecracker, docker, macOS) reconcile through the *same*
|
||||
contract — an abstract method on the backend base class, so a new backend
|
||||
cannot forget to implement it and none drifts onto a bespoke mechanism.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Deliberate mid-session CA rotation *without* a gateway restart — `rotate_ca`
|
||||
stays for that operator action.
|
||||
- Changing source-IP attribution, `/resolve`, or the plane split (#469).
|
||||
|
||||
## Design
|
||||
|
||||
**The contract — `attach_bottled_agents_to_gateway()` on the backend ABC.**
|
||||
Reconciling running bottles against the current gateway is a backend
|
||||
responsibility (only the backend can enumerate its agents and reach them —
|
||||
firecracker over SSH, docker/macOS over `exec`/`cp`), so it is an
|
||||
`@abc.abstractmethod` on `BottleBackend` (`backend/base.py`). Every backend
|
||||
implements it; the host calls it whenever the gateway is (re)brought up. This is
|
||||
what makes the fix cross-backend by construction rather than a per-backend
|
||||
follow-up. It reprovisions **all** registered bottles' gateway-dependent state:
|
||||
CA, git-gate, and egress tokens.
|
||||
|
||||
**Trigger — the gateway bring-up path.** The host calls
|
||||
`attach_bottled_agents_to_gateway()` only when the gateway was actually
|
||||
(re)brought up — the cold-boot branch of the infra bring-up (e.g.
|
||||
`FirecrackerInfraService.ensure_running` after it boots a fresh pair), never on an
|
||||
adopt of a healthy, current gateway (state intact). So it fires exactly when the
|
||||
gateway was (re)booted — including orchestrator restarts, since the pair boots
|
||||
together — and there is no bare-restart path that bypasses bring-up.
|
||||
|
||||
**Per-backend implementation.** Each backend's `attach_bottled_agents_to_gateway`
|
||||
enumerates its live bottles and, for each, reconciles the three services against
|
||||
the current gateway. The firecracker implementation, once the gateway VM is up
|
||||
and its CA is available:
|
||||
|
||||
1. Map each live bottle's guest IP → `bottle_id` from the orchestrator registry
|
||||
(`list_bottles`).
|
||||
2. Install the **current** shared git-gate hooks (`git_gate_render_hook` /
|
||||
`git_gate_render_access_hook`) into the fresh gateway once — rendered from
|
||||
code, never from a bottle's possibly-stale state dir.
|
||||
3. For each live agent VM (enumerated from its run dir):
|
||||
- **CA:** SSH the current gateway CA into the agent's trust store and run
|
||||
`update-ca-certificates` (unconditional replace — there is one gateway, so no
|
||||
fingerprint match is needed).
|
||||
- **git-gate:** rebuild the bottle's upstreams from its persisted git-gate
|
||||
state dir (deploy key, known_hosts, upstream URL) and re-init its bare repos
|
||||
+ per-repo creds under `/git/<bottle_id>`.
|
||||
- **egress token:** read the agent's `ENV_VAR_SECRET` and feed
|
||||
`reprovision_bottles`, restoring the orchestrator's in-memory tokens.
|
||||
4. Every per-bottle step is wrapped so one failure is logged and skipped.
|
||||
|
||||
**Retire the persistence prototype.** No CA data volume, no git-gate data volume
|
||||
(the abandoned PRs #511 / #513). The launch-time `_reprovision_running_bottles`
|
||||
call folds into this bring-up reconcile, so egress tokens are restored on the same
|
||||
cold-boot trigger (an adopt needs no restore — the orchestrator never restarted).
|
||||
|
||||
**CA rotation.** Because the gateway rootfs is ephemeral, every cold boot mints a
|
||||
fresh CA; reconcile is what distributes it, so a routine gateway rebuild doubles
|
||||
as a CA rotation with zero extra machinery.
|
||||
|
||||
## Open questions
|
||||
|
||||
- Docker's existing host-bind-mounted CA (`host_gateway_ca_dir`): once docker's
|
||||
`attach_bottled_agents_to_gateway` pushes the CA to running agents on bring-up,
|
||||
the bind-mount is redundant — drop it (so docker rotates like firecracker) or
|
||||
keep it as belt-and-suspenders? Leaning drop, for one behaviour across backends.
|
||||
@@ -0,0 +1,183 @@
|
||||
# Testing a clean bot-bottle install on Linux
|
||||
|
||||
How do you exercise `install.sh` the way a brand-new user would — on a
|
||||
pristine Linux environment you can throw away afterward — *without*
|
||||
polluting your daily-driver host, and across the several package-management
|
||||
regimes Linux fragments into? This is the Linux counterpart to
|
||||
[`testing-clean-install-on-macos.md`](testing-clean-install-on-macos.md);
|
||||
the conclusion is different because Linux gives us a boundary macOS doesn't.
|
||||
|
||||
## Summary
|
||||
|
||||
On macOS the honest options were a throwaway user or a VM, and the throwaway
|
||||
user won on pragmatics (nested virtualization is gated to M3+). On Linux the
|
||||
calculus flips: a **disposable KVM virtual machine, booted from a distro
|
||||
cloud image and deleted per run, is both the cleanest boundary and the one
|
||||
that lets a single harness cover Ubuntu, Fedora, Arch, Alpine, and NixOS**.
|
||||
The host already requires KVM for the Firecracker backend, so the VM is cheap
|
||||
here.
|
||||
|
||||
The harness lives at [`scripts/linux-install-test.sh`](../../scripts/linux-install-test.sh).
|
||||
Per run it caches one read-only base image, boots a throwaway copy-on-write
|
||||
overlay (`qemu-img create -b base`), installs the distro's prerequisites,
|
||||
pipes *this checkout's* `install.sh` into the guest exactly as `curl … | sh`
|
||||
would, asserts the CLI installed, and deletes the overlay — the Linux
|
||||
equivalent of `docker run --rm`, for a whole machine.
|
||||
|
||||
## Why a VM, not a container or a throwaway user
|
||||
|
||||
| Mechanism | Why it's the wrong boundary here |
|
||||
|---|---|
|
||||
| **Container** (`docker run --rm`) | Shares the host kernel and ships a deliberately minimal userland — no systemd, a stubbed-out package manager story, and (crucially) it doesn't reproduce the *externally-managed Python* (PEP 668) that real desktop/server installs put in front of the user. It tests "does install.sh run in a container," not "does it run on a real distro." |
|
||||
| **Throwaway user** (`useradd`/`userdel`) | The macOS pick, but weaker on Linux: it reaches the real host, yet every system package it installs (python, pipx, git via `apt`/`dnf`/…) stays behind, and it can only ever test the *one* distro the host runs. The whole Linux-specific value is the cross-distro matrix. |
|
||||
| **Disposable KVM VM** (this harness) | A genuine kernel + userland + package-manager boundary that wipes to nothing on teardown, and swaps freely between distro cloud images. The one real cost — nested virtualization for the *backend* — doesn't apply, because we gate the installer, not the runtime (below). |
|
||||
|
||||
## Two variants: `test` (bare host) and `test-ready` (prepared host)
|
||||
|
||||
`install.sh` never installs a backend, and never installs its own toolchain
|
||||
prerequisites (python3, git, pipx) — it installs the `bot-bottle` package and
|
||||
runs `doctor`, which *reports* what's missing
|
||||
([`install.sh`](../../install.sh) header,
|
||||
[`bot_bottle/cli/commands/doctor.py`](../../bot_bottle/cli/commands/doctor.py)).
|
||||
That leaves two distinct things worth testing, split into two subcommands that
|
||||
mirror the macOS harness's `test` / `test-ready` convention (there the split is
|
||||
the backend service; here it is the toolchain the installer needs):
|
||||
|
||||
- **`test`** — `install.sh` runs on the **bare cloud image**, prerequisites and
|
||||
all left as the vendor ships them. This exercises `install.sh`'s own
|
||||
prerequisite-guard logic — the entire first half of the script (python
|
||||
version gate, git-for-git-specs gate, pipx/pip PEP-668 handling).
|
||||
- **`test-ready`** — the harness installs python3 + git + pipx first (the
|
||||
`prereqs` step), then runs `install.sh`. This is the *prepared-host happy
|
||||
path*: does a clean install actually land and produce a working CLI?
|
||||
|
||||
**Pass criteria differ by variant:**
|
||||
|
||||
| Variant | PASS when |
|
||||
|---|---|
|
||||
| `test` | `install.sh` **either** installs cleanly (the image already carried enough) **or** declines with one of its own recognized, actionable prerequisite errors (missing python3/git, no usable pip, PEP 668). A crash or an *unrecognized* failure is a FAIL. |
|
||||
| `test-ready` | `install.sh` actually lands: the `bot-bottle` entry point is present and runs, and `doctor` reports a usable python and config without crashing. A graceful decline is no longer good enough. |
|
||||
|
||||
Neither variant requires a green `doctor`: inside the VM there is no nested KVM
|
||||
or Docker, so **the backend is correctly reported not-ready** — install.sh does
|
||||
not install a backend and cannot regress one, and this harness does not
|
||||
provision the Docker backend. This is where Linux necessarily diverges from the
|
||||
macOS `test-ready`, which reaches the host backend; `BB_TEST_REQUIRE_BACKEND=1`
|
||||
makes readiness fatal anyway, for a nested-virt host that can satisfy it. The
|
||||
verdict instead classifies `doctor`'s output the way the macOS harness does — a
|
||||
`Traceback` is an install defect (fail), a missing `python`/`config` line is a
|
||||
fail, a not-ready backend is reported — so a genuine installer regression (a
|
||||
broken shim, an import error, a botched PATH) stays visible.
|
||||
|
||||
`test-all` runs the full matrix — every distro × both variants — each cell in
|
||||
its own throwaway VM, and prints a per-cell PASS/FAIL summary.
|
||||
|
||||
## The distro matrix is the point
|
||||
|
||||
Each distro exercises a different corner of the installer:
|
||||
|
||||
| Distro | Cloud image | What it stresses |
|
||||
|---|---|---|
|
||||
| **Ubuntu** (noble) | `cloud-images.ubuntu.com` | The common case; `apt`'s `pipx`, externally-managed Python (PEP 668) → install.sh's pipx path. |
|
||||
| **Fedora** | Fedora Cloud Base Generic | `dnf` packaging, a different default Python, BSD-style checksum file. |
|
||||
| **Arch** | `geo.mirror.pkgbuild.com/images/latest` | Rolling / newest Python; `python-pipx`. |
|
||||
| **Alpine** | Alpine `nocloud_` (cloudinit) image | musl libc + BusyBox `sh` — the harshest POSIX-`sh` host for a `#!/bin/sh` installer. |
|
||||
| **NixOS** | locally built with `nixos-generators` | No FHS `~/.local` on PATH by default; `nix profile install` prereqs; pipx laying a self-contained venv on a non-FHS host. |
|
||||
|
||||
### Validation run (2026-07-27) — full green
|
||||
|
||||
Full matrix on the delphi KVM host, QEMU 11.0.2, both variants × all five
|
||||
distros passing:
|
||||
|
||||
| Distro | `test` (bare) | `test-ready` (prepared) |
|
||||
|---|---|---|
|
||||
| Ubuntu 24.04 | ✅ declines at git gate | ✅ installs, doctor python+config green |
|
||||
| Fedora 44 | ✅ declines at git gate | ✅ installs |
|
||||
| Arch (latest) | ✅ declines at git gate | ✅ installs |
|
||||
| Alpine 3.21 | ✅ declines at git gate | ✅ installs |
|
||||
| NixOS 24.11 | ✅ declines (no python3) | ✅ installs |
|
||||
|
||||
The bare `test` sees `install.sh` decline soundly — exit 1 at the
|
||||
git-for-git-specs gate on the Debian/Fedora/Arch/Alpine images (they ship
|
||||
python3 but not git), and at the python3 gate on NixOS (no python3 on PATH) —
|
||||
and `test-ready` installs cleanly with `doctor` reporting a usable python and
|
||||
config (backends all not-ready, as expected in a plain VM).
|
||||
|
||||
Getting to green surfaced and fixed a series of real defects:
|
||||
|
||||
- **Fedora 41 was EOL/404** → bumped to 44.
|
||||
- The liveness probe used `bot-bottle --version`, which the CLI does not
|
||||
implement (unknown args die non-zero), so every *successful* install was
|
||||
misreported as failed → switched to `bot-bottle --help`.
|
||||
- **Alpine** needed three fixes: the `generic_` image ignores a NoCloud seed
|
||||
(switched to the `nocloud_` variant); OpenRC does not auto-start sshd after
|
||||
cloud-init injects the key (start it via `runcmd`); and Alpine's non-PAM
|
||||
sshd refuses pubkey auth for a cloud-init-*locked* account (give it a
|
||||
throwaway password). It also has no `sudo` by default (install it via
|
||||
cloud-init `packages:`).
|
||||
- **NixOS** publishes no downloadable cloud qcow2 (its cloud images are
|
||||
Hydra-built AMIs), so the harness builds one with `nixos-generators`
|
||||
([`linux-install-test-nixos.nix`](../../scripts/linux-install-test-nixos.nix)):
|
||||
cloud-init for the key, flakes enabled, deliberately no python/git/pipx. The
|
||||
`test-ready` prereq install pins `nixpkgs/nixos-24.11` because the guest's
|
||||
default unstable registry builds pipx from source (and its test suite
|
||||
currently fails to build).
|
||||
- Two harness-hygiene bugs also fixed: `cmd_down` left `serial.log` behind
|
||||
(orphaned run dirs), and the teardown trap was armed after `cmd_up`, leaking
|
||||
a VM when `wait_for_ssh` timed out.
|
||||
|
||||
In `test-ready` the harness installs `python3 + git + pipx` first on each
|
||||
distro (install.sh installs none of them), so all five drive the recommended
|
||||
pipx path. `test` then removes that scaffolding and lets each distro's bare
|
||||
image collide with install.sh's guards — on most cloud images python3 is
|
||||
present (cloud-init needs it) but git and pipx are not, so install.sh is
|
||||
expected to decline at the git-for-git-specs gate or the PEP-668 pip check with
|
||||
an actionable message. Both are legitimate, and the two variants together cover
|
||||
the whole first half of the installer as well as the happy path.
|
||||
|
||||
## What a clean install touches (the footprint that decides "wipeable")
|
||||
|
||||
| Artifact | Location | In the guest's `$HOME`? | Survives VM teardown? |
|
||||
|---|---|---|---|
|
||||
| Config / state / db | `~/.bot-bottle/{agents,bottles,contrib,…}` ([`install.sh`](../../install.sh)) | ✅ | ❌ overlay deleted |
|
||||
| pipx venv + shim | `~/.local/pipx/venvs/bot-bottle`, shim in `~/.local/bin` | ✅ | ❌ overlay deleted |
|
||||
| pip `--user` fallback | `~/.local/lib` + `~/.local/bin` | ✅ | ❌ overlay deleted |
|
||||
| **Distro prerequisites** (python/git/pipx, `test-ready` only) | system paths via `apt`/`dnf`/`pacman`/`apk`/`nix profile` | ❌ | ❌ **overlay deleted** |
|
||||
|
||||
Unlike the macOS throwaway user (whose Homebrew / Apple-Container / Rosetta
|
||||
footprint *survives*), **every row here dies with the overlay** — that is the
|
||||
VM's whole advantage. The cached base image is read-only backing and is the
|
||||
only thing that persists between runs, on purpose.
|
||||
|
||||
## Design notes baked into the harness
|
||||
|
||||
- **User-mode networking** (`-netdev user,hostfwd=tcp:127.0.0.1:PORT-:22`):
|
||||
no root, no bridge, no host network state touched. Only SSH is forwarded.
|
||||
- **cloud-init seed ISO** (`cloud-localds`) injects an ephemeral SSH keypair
|
||||
and a passwordless-sudo login. The keypair is generated per run and deleted
|
||||
on teardown; the guest can't be logged into after it's gone.
|
||||
- **Copy-on-write overlay**: the cached base is never mutated, so a corrupt or
|
||||
interrupted run can't poison the cache; downloads land at `*.partial` and
|
||||
are renamed only after checksum verification.
|
||||
- **Checksums**: verified against each vendor's published sums file at
|
||||
download time (GNU `hash file`, bare-hash, and Fedora's BSD
|
||||
`SHA256 (file) = hash` formats are all handled). Alpine ships `.sha512` only
|
||||
(this verifier is sha256) so it is skipped; NixOS is built locally, not
|
||||
downloaded, so there is nothing to verify.
|
||||
- **NixOS is built, not downloaded**: `ensure_base_image` runs
|
||||
`nixos-generate -f qcow` against
|
||||
[`linux-install-test-nixos.nix`](../../scripts/linux-install-test-nixos.nix)
|
||||
once and caches the result; the per-run seed/overlay flow is otherwise
|
||||
identical to the downloaded distros.
|
||||
- **`test-all`** runs every distro × both variants (`test` and `test-ready`),
|
||||
each cell in its own subshell on its own forwarded port, so one cell's
|
||||
failure (or teardown trap) can't abort the matrix; it prints a per-cell
|
||||
PASS/FAIL summary.
|
||||
|
||||
## Not wired into PR CI
|
||||
|
||||
Like the macOS harness, the runtime is host-specific (needs `/dev/kvm`,
|
||||
`qemu`, and `cloud-localds`) and is not exercised by the Linux pull-request
|
||||
runner. It is validated statically (`bash -n`, `shellcheck`) and run by hand
|
||||
on a KVM-capable host. The cloud-image URLs in the `DISTRO` table are the one
|
||||
place to bump when a distro cuts a newer build.
|
||||
@@ -0,0 +1,24 @@
|
||||
# NixOS image for scripts/linux-install-test.sh.
|
||||
#
|
||||
# NixOS publishes no downloadable cloud qcow2 (its cloud images are Hydra-built
|
||||
# AMIs), so the harness BUILDS this one with nixos-generators (-f qcow). It is
|
||||
# deliberately minimal — no python3/git/pipx — so the bare `test` variant is
|
||||
# genuinely under-provisioned and exercises install.sh's guards; `test-ready`
|
||||
# provisions them with `nix profile install` (hence flakes below).
|
||||
{ lib, ... }:
|
||||
{
|
||||
# Consume the same NoCloud seed the other distros use: cloud-init injects the
|
||||
# per-run ephemeral SSH key for root. Leave networking to NixOS's default
|
||||
# dhcpcd (QEMU user-mode NAT) — enabling cloud-init's networkd here conflicts
|
||||
# with dhcpcd and can drop the guest's network.
|
||||
services.cloud-init.enable = true;
|
||||
services.cloud-init.network.enable = false;
|
||||
|
||||
services.openssh.enable = true;
|
||||
services.openssh.settings.PermitRootLogin = lib.mkForce "prohibit-password";
|
||||
|
||||
# Flakes so the `test-ready` prereq step can `nix profile install nixpkgs#...`.
|
||||
nix.settings.experimental-features = [ "nix-command" "flakes" ];
|
||||
|
||||
system.stateVersion = "24.11";
|
||||
}
|
||||
Executable
+753
@@ -0,0 +1,753 @@
|
||||
#!/usr/bin/env bash
|
||||
# Clean-install test harness for the Linux path.
|
||||
#
|
||||
# Exercises install.sh the way a brand-new user would, inside a THROWAWAY
|
||||
# QEMU/KVM virtual machine that is booted from a distro cloud image and
|
||||
# deleted afterward. install.sh's entire footprint is user-home-local (the
|
||||
# pipx venv under ~/.local, the ~/.bot-bottle config dir, and a printed PATH
|
||||
# hint), so a fresh VM's fresh $HOME is the clean surface we want — and unlike
|
||||
# a throwaway user account, tearing the VM down also wipes any OS-level
|
||||
# prerequisites installed into it, so the reset is total. Full rationale in
|
||||
# docs/research/testing-clean-install-on-linux.md.
|
||||
#
|
||||
# Why a VM and not a container or a throwaway user: a container shares the
|
||||
# host kernel and cannot exercise a genuinely pristine OS (systemd, the distro
|
||||
# package manager, PEP 668 externally-managed Python) the way a real guest
|
||||
# does, and a throwaway user leaves every system package it installs behind.
|
||||
# The host already needs KVM for the Firecracker backend, so a per-run,
|
||||
# copy-on-write VM is cheap here: one cached base image, a throwaway overlay
|
||||
# per run (`qemu-img create -b base`), deleted on teardown — the Linux
|
||||
# equivalent of `docker run --rm`, but for a whole machine.
|
||||
#
|
||||
# Usage:
|
||||
# ./scripts/linux-install-test.sh test # up -> run -> verdict -> down (bare host)
|
||||
# ./scripts/linux-install-test.sh test-ready # ... with prerequisites installed first
|
||||
# ./scripts/linux-install-test.sh test-all # every distro × both variants, with a summary
|
||||
# ./scripts/linux-install-test.sh up # fetch base image, boot a fresh VM
|
||||
# ./scripts/linux-install-test.sh prereqs # install python3 + git + pipx in the VM
|
||||
# ./scripts/linux-install-test.sh run # pipe install.sh into the VM
|
||||
# ./scripts/linux-install-test.sh status # VM reachable? is the install sound?
|
||||
# ./scripts/linux-install-test.sh down # kill the VM, delete overlay + seed (the reset)
|
||||
# ./scripts/linux-install-test.sh ssh # open an interactive shell in the running VM
|
||||
#
|
||||
# TWO TEST VARIANTS, because "does install.sh handle an unprepared host" and
|
||||
# "does a prepared host get a clean install" are different questions (mirrors
|
||||
# the macOS harness's test / test-ready split):
|
||||
#
|
||||
# test A Linux system WITHOUT the prerequisites set up — the default
|
||||
# state of a stock cloud image (python3 is usually present for
|
||||
# cloud-init, but git and pipx are not). This exercises
|
||||
# install.sh's own prerequisite-guard logic. It is SOUND — a
|
||||
# PASS — when install.sh EITHER installs cleanly (the image
|
||||
# already had enough) OR declines with one of its own recognized,
|
||||
# actionable errors (missing python3/git, no usable pip, PEP 668
|
||||
# externally-managed). A crash or an unrecognized failure fails.
|
||||
#
|
||||
# test-ready A Linux system WITH the prerequisites satisfied — the harness
|
||||
# installs python3 + git + pipx first (see the DISTRO table),
|
||||
# then runs install.sh. A graceful decline is no longer good
|
||||
# enough here: the install MUST land, the entry point must run,
|
||||
# and doctor must report a usable python and config without
|
||||
# crashing.
|
||||
#
|
||||
# Neither variant requires a green backend. install.sh does not install a
|
||||
# backend and cannot regress one, and inside a plain VM there is no nested KVM
|
||||
# for Firecracker; the harness does not provision the Docker backend either. So
|
||||
# backend readiness is reported, not required (this is where Linux necessarily
|
||||
# diverges from the macOS test-ready, which reaches the host backend).
|
||||
# BB_TEST_REQUIRE_BACKEND=1 makes it fatal anyway, for a nested-virt host.
|
||||
#
|
||||
# Config via env:
|
||||
# BB_TEST_DISTRO ubuntu | fedora | arch | alpine | nixos (default: ubuntu)
|
||||
# BB_TEST_SSH_PORT host port forwarded to the guest's :22 (default: 2222)
|
||||
# BB_TEST_CACHE_DIR where base images are cached (default: ~/.cache/bot-bottle-install-test)
|
||||
# BB_TEST_RUN_DIR per-run scratch (overlay, seed, key, …) (default: a mktemp dir)
|
||||
# BB_TEST_MEM_MB guest RAM (default: 2048)
|
||||
# BB_TEST_CPUS guest vCPUs (default: 2)
|
||||
# BB_TEST_DISK overlay virtual size (default: 12G)
|
||||
# BB_TEST_BOOT_TIMEOUT seconds to wait for SSH after boot (default: 300)
|
||||
# BB_TEST_KEEP 1 = the test cycles skip teardown, to poke at a failure
|
||||
# BB_TEST_REQUIRE_BACKEND 1 = make a not-ready backend fatal (needs nested virt)
|
||||
# BB_TEST_INSTALL_URL curl this install.sh in the guest instead of piping the local checkout
|
||||
# BB_TEST_SKIP_VERIFY 1 = skip base-image checksum verification (not recommended)
|
||||
# BOT_BOTTLE_INSTALL_SPEC passed through to install.sh (pip / git spec)
|
||||
#
|
||||
# Notes:
|
||||
# * Needs /dev/kvm, qemu-system-x86_64, and cloud-localds (cloud-image-utils
|
||||
# / cloud-utils); the nixos distro additionally needs nixos-generate. On the
|
||||
# NixOS host: nix shell nixpkgs#qemu nixpkgs#cloud-utils nixpkgs#nixos-generators
|
||||
# * Networking is user-mode (`-netdev user,hostfwd`) so the harness needs no
|
||||
# root, no bridge, and touches no host network state. Only SSH is forwarded.
|
||||
# * The cloud-image URLs in the DISTRO table are the one place to bump when a
|
||||
# distro cuts a new build; each is verified against the vendor's published
|
||||
# checksum at download time (guarding against truncated/corrupt pulls).
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
DISTRO="${BB_TEST_DISTRO:-ubuntu}"
|
||||
SSH_PORT="${BB_TEST_SSH_PORT:-2222}"
|
||||
CACHE_DIR="${BB_TEST_CACHE_DIR:-${XDG_CACHE_HOME:-$HOME/.cache}/bot-bottle-install-test}"
|
||||
MEM_MB="${BB_TEST_MEM_MB:-2048}"
|
||||
CPUS="${BB_TEST_CPUS:-2}"
|
||||
DISK="${BB_TEST_DISK:-12G}"
|
||||
BOOT_TIMEOUT="${BB_TEST_BOOT_TIMEOUT:-300}"
|
||||
|
||||
_SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
_REPO_ROOT="$(cd "$_SCRIPT_DIR/.." && pwd)"
|
||||
|
||||
# Per-run scratch. Persisted across sub-commands (up/prereqs/run/status/down)
|
||||
# via a marker file so `up` in one invocation and `down` in the next find the
|
||||
# same VM; the test cycles set RUN_DIR themselves and never write the marker.
|
||||
RUN_DIR="${BB_TEST_RUN_DIR:-}"
|
||||
|
||||
# Set by the test cycles, which chain the steps and suppress the per-step "next
|
||||
# command" hints. _STEPS / _PASS_CLAIM / _REQUIRE_INSTALL are the per-variant
|
||||
# knobs the shared cycle and teardown read.
|
||||
IN_TEST=0
|
||||
_STEPS=4
|
||||
_PASS_CLAIM=""
|
||||
_REQUIRE_INSTALL=0
|
||||
|
||||
# --- distro table ----------------------------------------------------
|
||||
# For each distro: cloud-image URL | checksum-file URL | default SSH user |
|
||||
# prerequisite-install command (run in the guest; uses sudo when user != root).
|
||||
#
|
||||
# The prerequisite command installs python3 + git + pipx — install.sh installs
|
||||
# none of them — so `test-ready` (and the `prereqs` sub-command) drive
|
||||
# install.sh down its recommended pipx path. Bump the URLs here when a distro
|
||||
# publishes a newer build.
|
||||
declare -A IMAGE_URL SUM_URL SSH_USER PREREQ
|
||||
|
||||
IMAGE_URL[ubuntu]="https://cloud-images.ubuntu.com/noble/current/noble-server-cloudimg-amd64.img"
|
||||
SUM_URL[ubuntu]="https://cloud-images.ubuntu.com/noble/current/SHA256SUMS"
|
||||
SSH_USER[ubuntu]="ubuntu"
|
||||
PREREQ[ubuntu]="sudo apt-get update && sudo DEBIAN_FRONTEND=noninteractive apt-get install -y python3 git pipx"
|
||||
|
||||
IMAGE_URL[fedora]="https://download.fedoraproject.org/pub/fedora/linux/releases/44/Cloud/x86_64/images/Fedora-Cloud-Base-Generic-44-1.7.x86_64.qcow2"
|
||||
SUM_URL[fedora]="https://download.fedoraproject.org/pub/fedora/linux/releases/44/Cloud/x86_64/images/Fedora-Cloud-44-1.7-x86_64-CHECKSUM"
|
||||
SSH_USER[fedora]="fedora"
|
||||
PREREQ[fedora]="sudo dnf install -y python3 git pipx"
|
||||
|
||||
IMAGE_URL[arch]="https://geo.mirror.pkgbuild.com/images/latest/Arch-Linux-x86_64-cloudimg.qcow2"
|
||||
SUM_URL[arch]="https://geo.mirror.pkgbuild.com/images/latest/Arch-Linux-x86_64-cloudimg.qcow2.SHA256"
|
||||
SSH_USER[arch]="arch"
|
||||
PREREQ[arch]="sudo pacman -Sy --noconfirm python git python-pipx"
|
||||
|
||||
# Use the *nocloud_* Alpine variant, not generic_: the generic image probes
|
||||
# network datasources and ignores the local NoCloud seed, so cloud-init never
|
||||
# runs and the SSH key is never injected. (Only published at .0 patch levels.)
|
||||
IMAGE_URL[alpine]="https://dl-cdn.alpinelinux.org/alpine/v3.21/releases/cloud/nocloud_alpine-3.21.0-x86_64-bios-cloudinit-r0.qcow2"
|
||||
SUM_URL[alpine]="" # Alpine cloud images ship .sha512 only; this verifier is sha256.
|
||||
SSH_USER[alpine]="alpine"
|
||||
PREREQ[alpine]="sudo apk add --no-cache python3 git pipx"
|
||||
|
||||
# NixOS publishes no downloadable cloud qcow2 (its cloud images are Hydra-built
|
||||
# AMIs), so the harness BUILDS one with nixos-generators — see build_nixos_image
|
||||
# and linux-install-test-nixos.nix. It's externally-managed in its own way (no
|
||||
# FHS ~/.local on PATH by default); `nix profile install` provisions the
|
||||
# prerequisites into root's profile (the image enables flakes for this).
|
||||
IMAGE_URL[nixos]="nix:build" # sentinel: ensure_base_image builds instead of downloading
|
||||
SUM_URL[nixos]=""
|
||||
SSH_USER[nixos]="root"
|
||||
# Pin to a stable release: the guest's default `nixpkgs` registry is unstable,
|
||||
# where pipx isn't in the binary cache and builds from source (its test suite
|
||||
# currently fails to build). nixos-24.11 has these cached as substitutes.
|
||||
PREREQ[nixos]="nix profile install nixpkgs/nixos-24.11#python3 nixpkgs/nixos-24.11#git nixpkgs/nixos-24.11#pipx"
|
||||
|
||||
ALL_DISTROS=(ubuntu fedora arch alpine nixos)
|
||||
|
||||
# --- guards ----------------------------------------------------------
|
||||
require_linux() {
|
||||
[ "$(uname -s)" = "Linux" ] \
|
||||
|| { echo "error: this harness is Linux-only (uname is $(uname -s))" >&2; exit 1; }
|
||||
}
|
||||
|
||||
require_kvm() {
|
||||
[ -e /dev/kvm ] && [ -r /dev/kvm ] && [ -w /dev/kvm ] \
|
||||
|| { echo "error: /dev/kvm is missing or not accessible (add yourself to the 'kvm' group)" >&2; exit 1; }
|
||||
}
|
||||
|
||||
require_tools() {
|
||||
local missing=()
|
||||
command -v qemu-system-x86_64 >/dev/null 2>&1 || missing+=(qemu-system-x86_64)
|
||||
command -v qemu-img >/dev/null 2>&1 || missing+=(qemu-img)
|
||||
command -v cloud-localds >/dev/null 2>&1 || missing+=(cloud-localds)
|
||||
command -v ssh >/dev/null 2>&1 || missing+=(ssh)
|
||||
command -v curl >/dev/null 2>&1 || missing+=(curl)
|
||||
if [ "${#missing[@]}" -ne 0 ]; then
|
||||
echo "error: missing required tools: ${missing[*]}" >&2
|
||||
echo " on NixOS: nix shell nixpkgs#qemu nixpkgs#cloud-utils nixpkgs#openssh nixpkgs#curl" >&2
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
known_distro() {
|
||||
[ -n "${IMAGE_URL[$DISTRO]:-}" ] \
|
||||
|| { echo "error: unknown distro '$DISTRO' (known: ${ALL_DISTROS[*]})" >&2; exit 1; }
|
||||
}
|
||||
|
||||
# --- run-dir bookkeeping ---------------------------------------------
|
||||
# The marker lets prereqs/run/status/down in separate invocations find the VM
|
||||
# that `up` started. The test cycles set RUN_DIR themselves and never write it.
|
||||
_marker() { echo "${TMPDIR:-/tmp}/bot-bottle-install-test.$DISTRO.run"; }
|
||||
|
||||
_ensure_run_dir() {
|
||||
if [ -z "$RUN_DIR" ]; then
|
||||
RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/bb-install-test.$DISTRO.XXXXXX")"
|
||||
fi
|
||||
mkdir -p "$RUN_DIR"
|
||||
}
|
||||
|
||||
_load_run_dir() {
|
||||
if [ -z "$RUN_DIR" ] && [ -f "$(_marker)" ]; then
|
||||
RUN_DIR="$(cat "$(_marker)")"
|
||||
fi
|
||||
[ -n "$RUN_DIR" ] && [ -d "$RUN_DIR" ]
|
||||
}
|
||||
|
||||
_ssh_key() { echo "$RUN_DIR/id_ed25519"; }
|
||||
_overlay() { echo "$RUN_DIR/overlay.qcow2"; }
|
||||
_seed() { echo "$RUN_DIR/seed.iso"; }
|
||||
_pidfile() { echo "$RUN_DIR/qemu.pid"; }
|
||||
_serial() { echo "$RUN_DIR/serial.log"; }
|
||||
|
||||
# --- ssh helpers -----------------------------------------------------
|
||||
_ssh_opts() {
|
||||
# No host-key pinning: the guest is thrown away every run.
|
||||
printf '%s\0' \
|
||||
-i "$(_ssh_key)" \
|
||||
-p "$SSH_PORT" \
|
||||
-o StrictHostKeyChecking=no \
|
||||
-o UserKnownHostsFile=/dev/null \
|
||||
-o LogLevel=ERROR \
|
||||
-o ConnectTimeout=8 \
|
||||
-o BatchMode=yes
|
||||
}
|
||||
|
||||
guest() {
|
||||
local -a opts
|
||||
mapfile -d '' -t opts < <(_ssh_opts)
|
||||
ssh "${opts[@]}" "${SSH_USER[$DISTRO]}@127.0.0.1" "$@"
|
||||
}
|
||||
|
||||
wait_for_ssh() {
|
||||
local deadline=$(( SECONDS + BOOT_TIMEOUT ))
|
||||
echo "== waiting for SSH on 127.0.0.1:$SSH_PORT (up to ${BOOT_TIMEOUT}s) =="
|
||||
while [ "$SECONDS" -lt "$deadline" ]; do
|
||||
if guest true 2>/dev/null; then
|
||||
echo " guest is up"
|
||||
return 0
|
||||
fi
|
||||
# Bail early if QEMU has died — no point waiting out the timeout.
|
||||
if [ -f "$(_pidfile)" ] && ! kill -0 "$(cat "$(_pidfile)")" 2>/dev/null; then
|
||||
echo "error: QEMU exited before SSH came up; see $(_serial)" >&2
|
||||
return 1
|
||||
fi
|
||||
sleep 3
|
||||
done
|
||||
echo "error: timed out waiting for SSH; see $(_serial)" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
# --- image cache -----------------------------------------------------
|
||||
_base_image() {
|
||||
# One cached file per distro. NixOS is built (not downloaded), so it has a
|
||||
# fixed cache name; the rest are keyed by the image's basename so a URL bump
|
||||
# lands as a new cache entry rather than a stale hit.
|
||||
if [ "$DISTRO" = nixos ]; then
|
||||
echo "$CACHE_DIR/nixos-built.qcow2"
|
||||
return
|
||||
fi
|
||||
local url="${IMAGE_URL[$DISTRO]}"
|
||||
echo "$CACHE_DIR/$DISTRO-$(basename "$url")"
|
||||
}
|
||||
|
||||
# NixOS has no upstream cloud qcow2; build one with nixos-generators and copy it
|
||||
# out of the (immutable, GC-able) store into the cache.
|
||||
build_nixos_image() {
|
||||
local out; out="$(_base_image)"
|
||||
[ -f "$out" ] && { echo "== nixos base image cached: $out =="; return 0; }
|
||||
command -v nixos-generate >/dev/null 2>&1 || {
|
||||
echo "error: 'nixos-generate' is required to build the NixOS image" >&2
|
||||
echo " run inside: nix shell nixpkgs#nixos-generators nixpkgs#qemu nixpkgs#cloud-utils" >&2
|
||||
return 1
|
||||
}
|
||||
echo "== building NixOS cloud image with nixos-generators (first run is slow) =="
|
||||
local link="$CACHE_DIR/nixos-result"
|
||||
nixos-generate -f qcow --system x86_64-linux \
|
||||
-c "$_SCRIPT_DIR/linux-install-test-nixos.nix" -o "$link"
|
||||
cp -L "$link"/*.qcow2 "$out"
|
||||
rm -f "$link"
|
||||
echo " built + cached: $out"
|
||||
}
|
||||
|
||||
verify_checksum() {
|
||||
local file="$1" sums_url="${SUM_URL[$DISTRO]}" base
|
||||
base="$(basename "${IMAGE_URL[$DISTRO]}")"
|
||||
if [ "${BB_TEST_SKIP_VERIFY:-0}" = "1" ] || [ -z "$sums_url" ]; then
|
||||
echo " checksum: SKIPPED (${sums_url:+set BB_TEST_SKIP_VERIFY=0 to enable}${sums_url:-no sums URL for $DISTRO})" >&2
|
||||
return 0
|
||||
fi
|
||||
local want
|
||||
# Vendors publish either "HASH filename" tables or a bare "HASH" (or a
|
||||
# "SHA256 (file) = HASH" BSD line, e.g. Fedora). Cover all three.
|
||||
local sums; sums="$(curl -fsSL "$sums_url")"
|
||||
want="$(printf '%s\n' "$sums" | awk -v f="$base" '
|
||||
$0 ~ f && $1 ~ /^[0-9a-fA-F]{64}$/ { print $1; exit } # GNU "hash file"
|
||||
$1=="SHA256" && $0 ~ f { gsub(/[()]/,""); print $NF; exit } # BSD "SHA256 (file) = hash"
|
||||
')"
|
||||
[ -z "$want" ] && want="$(printf '%s\n' "$sums" | awk '/^[0-9a-fA-F]{64}$/ {print $1; exit}')"
|
||||
[ -n "$want" ] || { echo "error: could not find a sha256 for $base in $sums_url" >&2; return 1; }
|
||||
local got; got="$(sha256sum "$file" | awk '{print $1}')"
|
||||
if [ "$want" != "$got" ]; then
|
||||
echo "error: checksum mismatch for $base" >&2
|
||||
echo " want $want" >&2
|
||||
echo " got $got" >&2
|
||||
return 1
|
||||
fi
|
||||
echo " checksum: OK"
|
||||
}
|
||||
|
||||
ensure_base_image() {
|
||||
mkdir -p "$CACHE_DIR"
|
||||
if [ "$DISTRO" = nixos ]; then
|
||||
build_nixos_image
|
||||
return
|
||||
fi
|
||||
local base; base="$(_base_image)"
|
||||
if [ -f "$base" ]; then
|
||||
echo "== base image cached: $base =="
|
||||
return 0
|
||||
fi
|
||||
echo "== downloading $DISTRO cloud image =="
|
||||
echo " ${IMAGE_URL[$DISTRO]}"
|
||||
# Download to a temp name and rename on success so an interrupted pull
|
||||
# never poisons the cache with a truncated image.
|
||||
local tmp="$base.partial"
|
||||
curl -fSL --retry 3 -o "$tmp" "${IMAGE_URL[$DISTRO]}"
|
||||
verify_checksum "$tmp"
|
||||
mv "$tmp" "$base"
|
||||
echo " cached: $base"
|
||||
}
|
||||
|
||||
# --- cloud-init seed -------------------------------------------------
|
||||
make_seed() {
|
||||
ssh-keygen -t ed25519 -N '' -f "$(_ssh_key)" -q
|
||||
local pub; pub="$(cat "$(_ssh_key).pub")"
|
||||
local user="${SSH_USER[$DISTRO]}"
|
||||
local user_data="$RUN_DIR/user-data"
|
||||
|
||||
if [ "$user" = "root" ]; then
|
||||
# NixOS' cloud-init lands the key straight on root; no sudo needed.
|
||||
cat > "$user_data" <<EOF
|
||||
#cloud-config
|
||||
ssh_authorized_keys:
|
||||
- $pub
|
||||
EOF
|
||||
elif [ "$DISTRO" = alpine ]; then
|
||||
# Alpine needs three things the systemd distros don't: cloud-init locks
|
||||
# the account (lock_passwd), but Alpine's non-PAM sshd then refuses
|
||||
# pubkey auth for a locked account — so give it a throwaway password;
|
||||
# and OpenRC does not auto-start sshd after the key is injected, so
|
||||
# start it via runcmd.
|
||||
cat > "$user_data" <<EOF
|
||||
#cloud-config
|
||||
users:
|
||||
- name: $user
|
||||
sudo: ALL=(ALL) NOPASSWD:ALL
|
||||
shell: /bin/sh
|
||||
lock_passwd: false
|
||||
ssh_authorized_keys:
|
||||
- $pub
|
||||
chpasswd:
|
||||
expire: false
|
||||
list: |
|
||||
$user:bbtest
|
||||
packages:
|
||||
- sudo
|
||||
runcmd:
|
||||
- [ sh, -c, "rc-service sshd start 2>/dev/null || true" ]
|
||||
EOF
|
||||
else
|
||||
cat > "$user_data" <<EOF
|
||||
#cloud-config
|
||||
users:
|
||||
- name: $user
|
||||
sudo: ALL=(ALL) NOPASSWD:ALL
|
||||
shell: /bin/sh
|
||||
lock_passwd: true
|
||||
ssh_authorized_keys:
|
||||
- $pub
|
||||
EOF
|
||||
fi
|
||||
# NoCloud wants a meta-data with an instance-id, or cloud-init may not treat
|
||||
# the seed as a new instance (the Alpine nocloud image is strict about this).
|
||||
printf 'instance-id: bbtest-%s\nlocal-hostname: bbtest-%s\n' "$DISTRO" "$DISTRO" \
|
||||
> "$RUN_DIR/meta-data"
|
||||
cloud-localds "$(_seed)" "$user_data" "$RUN_DIR/meta-data"
|
||||
}
|
||||
|
||||
# --- commands --------------------------------------------------------
|
||||
cmd_up() {
|
||||
require_linux; require_kvm; require_tools; known_distro
|
||||
_ensure_run_dir
|
||||
ensure_base_image
|
||||
|
||||
# Throwaway copy-on-write overlay: the cached base is read-only backing,
|
||||
# all guest writes land in the overlay, and `down` deletes it. Resize so
|
||||
# pipx + a git build have headroom (cloud-init grows the rootfs to fit).
|
||||
qemu-img create -q -f qcow2 -F qcow2 -b "$(_base_image)" "$(_overlay)" "$DISK"
|
||||
make_seed
|
||||
|
||||
echo "== booting $DISTRO VM (mem=${MEM_MB}M cpus=$CPUS, ssh -> :$SSH_PORT) =="
|
||||
qemu-system-x86_64 \
|
||||
-machine accel=kvm -cpu host -smp "$CPUS" -m "$MEM_MB" \
|
||||
-display none -daemonize \
|
||||
-pidfile "$(_pidfile)" \
|
||||
-serial "file:$(_serial)" \
|
||||
-drive "file=$(_overlay),if=virtio,format=qcow2" \
|
||||
-drive "file=$(_seed),if=virtio,format=raw" \
|
||||
-netdev "user,id=n0,hostfwd=tcp:127.0.0.1:$SSH_PORT-:22" \
|
||||
-device virtio-net-pci,netdev=n0
|
||||
|
||||
# Only publish the marker (so a later prereqs/run/down finds this VM) when
|
||||
# we aren't inside a test cycle, which manages its own RUN_DIR + teardown.
|
||||
[ "$IN_TEST" = 1 ] || echo "$RUN_DIR" > "$(_marker)"
|
||||
|
||||
wait_for_ssh
|
||||
if [ "$IN_TEST" != 1 ]; then
|
||||
echo "== VM is up. Prepare it with: BB_TEST_DISTRO=$DISTRO $0 prereqs (or go straight to 'run') =="
|
||||
fi
|
||||
}
|
||||
|
||||
# Install install.sh's toolchain prerequisites (python3 + git + pipx) into the
|
||||
# running VM. This is what separates `test-ready` from `test`, and it is a
|
||||
# distinct sub-command so a manual up/prereqs/run/down cycle is possible.
|
||||
cmd_prereqs() {
|
||||
require_linux
|
||||
_load_run_dir || { echo "error: no running VM for $DISTRO; run '$0 up' first" >&2; return 1; }
|
||||
echo "== installing prerequisites (python3 + git + pipx) on $DISTRO =="
|
||||
# Runs via the guest login shell; PREREQ is a client-side table value.
|
||||
guest "${PREREQ[$DISTRO]}"
|
||||
}
|
||||
|
||||
cmd_run() {
|
||||
require_linux
|
||||
_load_run_dir || { echo "error: no running VM for $DISTRO; run '$0 up' first" >&2; return 1; }
|
||||
|
||||
local spec_env=""
|
||||
[ -n "${BOT_BOTTLE_INSTALL_SPEC:-}" ] \
|
||||
&& spec_env="BOT_BOTTLE_INSTALL_SPEC='$BOT_BOTTLE_INSTALL_SPEC' "
|
||||
|
||||
echo "== installing bot-bottle as ${SSH_USER[$DISTRO]} =="
|
||||
# Capture install.sh's exit code and full output rather than aborting on
|
||||
# non-zero: on a bare host a clean prerequisite *decline* is sound, so the
|
||||
# verdict step — not set -e — decides the outcome.
|
||||
local rc
|
||||
set +e
|
||||
if [ -n "${BB_TEST_INSTALL_URL:-}" ]; then
|
||||
guest "curl -fsSL '$BB_TEST_INSTALL_URL' | ${spec_env}sh" 2>&1 | tee "$RUN_DIR/install.log"
|
||||
else
|
||||
# Feed THIS checkout's install.sh in over stdin — the same `curl … | sh`
|
||||
# shape a real user runs, and nothing is staged in the guest to leak.
|
||||
guest "${spec_env}sh -s" < "$_REPO_ROOT/install.sh" 2>&1 | tee "$RUN_DIR/install.log"
|
||||
fi
|
||||
rc="${PIPESTATUS[0]}"
|
||||
set -e
|
||||
printf '%s\n' "$rc" > "$RUN_DIR/install.rc"
|
||||
echo "== install.sh exited $rc =="
|
||||
[ "$IN_TEST" = 1 ] \
|
||||
|| echo "== verdict anytime with: BB_TEST_DISTRO=$DISTRO $0 status =="
|
||||
}
|
||||
|
||||
# Quietly report whether a runnable bot-bottle entry point exists for the
|
||||
# guest user, checking the pipx/pip locations install.sh may leave off PATH.
|
||||
entry_point_runnable() {
|
||||
# shellcheck disable=SC2016 # expand in the GUEST shell.
|
||||
guest '
|
||||
for bb in "$HOME/.local/bin/bot-bottle" "$HOME/.bot-bottle/venv/bin/bot-bottle" "$(command -v bot-bottle 2>/dev/null)"; do
|
||||
[ -n "$bb" ] && [ -x "$bb" ] || continue
|
||||
# --help exits 0 before any DB/migration/network work; it is the
|
||||
# cheapest proof the package imports and the shim runs. (bot-bottle
|
||||
# has no --version: an unknown arg would die non-zero.)
|
||||
"$bb" --help >/dev/null 2>&1 && exit 0
|
||||
done
|
||||
exit 1
|
||||
' >/dev/null 2>&1
|
||||
}
|
||||
|
||||
# `bot-bottle doctor` in the guest, classified. doctor's own exit code
|
||||
# conflates "is the install sound" with "is a backend ready to run a bottle" —
|
||||
# and inside a plain VM no backend can be ready (no nested KVM/Docker), so the
|
||||
# raw exit code is non-zero by design. This separates the two: an unhandled
|
||||
# traceback, or a missing python/config line, is an install defect and fails;
|
||||
# a not-ready backend is reported, not fatal (unless BB_TEST_REQUIRE_BACKEND=1,
|
||||
# for a nested-virt host that can actually satisfy it).
|
||||
doctor_in_guest() {
|
||||
local out rc=0 bad=0
|
||||
out="$(mktemp "${TMPDIR:-/tmp}/bb-doctor.XXXXXX")"
|
||||
# shellcheck disable=SC2016 # $HOME/$bb must expand in the GUEST shell.
|
||||
guest '
|
||||
for bb in bot-bottle "$HOME/.local/bin/bot-bottle" "$HOME/.bot-bottle/venv/bin/bot-bottle"; do
|
||||
if command -v "$bb" >/dev/null 2>&1; then
|
||||
case "$bb" in
|
||||
bot-bottle) : ;;
|
||||
*) echo " (not on PATH — running $bb directly, as install.sh advises)" ;;
|
||||
esac
|
||||
exec "$bb" doctor
|
||||
fi
|
||||
done
|
||||
echo " no bot-bottle entry point found for this user" >&2
|
||||
exit 1
|
||||
' >"$out" 2>&1 || rc=$?
|
||||
cat "$out"
|
||||
|
||||
# An unhandled exception is always an install/product defect, never an
|
||||
# environment fact — doctor's non-zero exit alone would not distinguish it.
|
||||
if grep -q 'Traceback (most recent call last)' "$out"; then
|
||||
echo " doctor crashed (traceback above) — a defect, not a missing prerequisite" >&2
|
||||
bad=1
|
||||
fi
|
||||
grep -qE '^ok: +python:' "$out" \
|
||||
|| { echo " doctor never reported a usable python" >&2; bad=1; }
|
||||
grep -qE '^ok: +config:' "$out" \
|
||||
|| { echo " doctor never reported a usable config dir" >&2; bad=1; }
|
||||
if [ "$rc" -ne 0 ] && ! grep -qE '^(fail|warn): +backend' "$out"; then
|
||||
echo " doctor failed for something other than backend readiness" >&2
|
||||
bad=1
|
||||
fi
|
||||
|
||||
local backend_ready=1
|
||||
grep -qE '^fail: +backend' "$out" && backend_ready=0
|
||||
rm -f "$out"
|
||||
|
||||
[ "$bad" -eq 0 ] || return 1
|
||||
if [ "${BB_TEST_REQUIRE_BACKEND:-0}" = "1" ] && [ "$backend_ready" -eq 0 ]; then
|
||||
echo " backend is not ready and BB_TEST_REQUIRE_BACKEND=1 — failing" >&2
|
||||
return 1
|
||||
fi
|
||||
if [ "$backend_ready" -eq 1 ]; then
|
||||
echo " doctor: install sound; a backend is ready"
|
||||
else
|
||||
echo " doctor: install sound; no backend ready (expected in a plain VM — install gate only)"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# The prerequisite-decline messages install.sh prints via die(). On a bare host
|
||||
# ANY of these means install.sh correctly refused rather than half-installing —
|
||||
# a sound outcome for `test`.
|
||||
PREREQ_ERR_RE='is required but was not found|or newer is required|git is required to install from|neither pipx nor a usable|externally managed \(PEP 668\)|is not on PATH'
|
||||
|
||||
# The verdict: is the install sound? Reads install.sh's captured exit code and
|
||||
# output (from cmd_run) plus the guest's resulting state.
|
||||
# - installed & runnable -> the verdict is doctor's soundness classification.
|
||||
# - no entry point, but a recognized prerequisite decline, and declines are
|
||||
# allowed (bare `test`, _REQUIRE_INSTALL=0) -> sound.
|
||||
# - anything else -> not sound.
|
||||
install_verdict() {
|
||||
local rc=""
|
||||
[ -f "$RUN_DIR/install.rc" ] && rc="$(cat "$RUN_DIR/install.rc")"
|
||||
|
||||
if [ "${rc:-1}" = 0 ] && entry_point_runnable; then
|
||||
echo "doctor (in guest):"
|
||||
doctor_in_guest
|
||||
return $?
|
||||
fi
|
||||
|
||||
if [ "${_REQUIRE_INSTALL:-0}" != "1" ] \
|
||||
&& [ -n "$rc" ] && [ "$rc" != 0 ] \
|
||||
&& [ -f "$RUN_DIR/install.log" ] \
|
||||
&& grep -Eiq "$PREREQ_ERR_RE" "$RUN_DIR/install.log"; then
|
||||
echo " install.sh declined with an actionable prerequisite error (rc=$rc)"
|
||||
echo " — sound on a bare host; run 'test-ready' (or 'prereqs') to install."
|
||||
return 0
|
||||
fi
|
||||
|
||||
if [ "${_REQUIRE_INSTALL:-0}" = "1" ]; then
|
||||
echo " prerequisites were provisioned, but install.sh left no runnable entry point (rc=${rc:-?})" >&2
|
||||
else
|
||||
echo " install.sh neither installed nor gave a recognized prerequisite error (rc=${rc:-?})" >&2
|
||||
fi
|
||||
return 1
|
||||
}
|
||||
|
||||
cmd_status() {
|
||||
require_linux
|
||||
if ! _load_run_dir; then
|
||||
echo "vm: no running VM for $DISTRO"
|
||||
return 0
|
||||
fi
|
||||
if [ -f "$(_pidfile)" ] && kill -0 "$(cat "$(_pidfile)")" 2>/dev/null; then
|
||||
echo "vm: $DISTRO running (pid $(cat "$(_pidfile)"), ssh :$SSH_PORT)"
|
||||
else
|
||||
echo "vm: $DISTRO run-dir present but QEMU not alive"
|
||||
return 1
|
||||
fi
|
||||
if install_verdict; then
|
||||
echo "OK[$DISTRO]: the install is sound"
|
||||
return 0
|
||||
fi
|
||||
echo "FAIL[$DISTRO]: the install is not sound (see above)" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
cmd_down() {
|
||||
require_linux
|
||||
if ! _load_run_dir; then
|
||||
echo "$DISTRO: nothing running"
|
||||
return 0
|
||||
fi
|
||||
if [ -f "$(_pidfile)" ]; then
|
||||
local pid; pid="$(cat "$(_pidfile)")"
|
||||
if kill -0 "$pid" 2>/dev/null; then
|
||||
kill "$pid" 2>/dev/null || true
|
||||
for _ in 1 2 3 4 5; do kill -0 "$pid" 2>/dev/null || break; sleep 1; done
|
||||
kill -9 "$pid" 2>/dev/null || true
|
||||
fi
|
||||
fi
|
||||
# Deleting the overlay + seed is the reset; the read-only base stays cached.
|
||||
rm -f "$(_overlay)" "$(_seed)" "$(_ssh_key)" "$(_ssh_key).pub" \
|
||||
"$RUN_DIR/user-data" "$RUN_DIR/meta-data" \
|
||||
"$RUN_DIR/install.rc" "$RUN_DIR/install.log" "$(_serial)"
|
||||
# Only remove a scratch dir we created (leave a user-provided one alone).
|
||||
[ -n "${BB_TEST_RUN_DIR:-}" ] || rmdir "$RUN_DIR" 2>/dev/null || true
|
||||
rm -f "$(_marker)"
|
||||
echo "removed $DISTRO VM and its overlay — install surface is clean."
|
||||
}
|
||||
|
||||
cmd_ssh() {
|
||||
require_linux
|
||||
_load_run_dir || { echo "error: no running VM for $DISTRO" >&2; return 1; }
|
||||
local -a opts
|
||||
mapfile -d '' -t opts < <(_ssh_opts)
|
||||
exec ssh -t "${opts[@]}" "${SSH_USER[$DISTRO]}@127.0.0.1"
|
||||
}
|
||||
|
||||
# Teardown half of the test cycles, armed the moment the VM exists so a failure
|
||||
# or a Ctrl-C still leaves nothing running.
|
||||
_test_teardown() {
|
||||
local rc=$?
|
||||
trap - EXIT INT TERM
|
||||
if [ "${BB_TEST_KEEP:-0}" = "1" ]; then
|
||||
echo
|
||||
echo "== [$_STEPS/$_STEPS] down: SKIPPED (BB_TEST_KEEP=1) =="
|
||||
echo " VM still up; remove with: BB_TEST_DISTRO=$DISTRO $0 down"
|
||||
echo " ssh in with: BB_TEST_RUN_DIR=$RUN_DIR BB_TEST_DISTRO=$DISTRO $0 ssh"
|
||||
exit "$rc"
|
||||
fi
|
||||
echo
|
||||
echo "== [$_STEPS/$_STEPS] down =="
|
||||
cmd_down || rc=1
|
||||
if [ "$rc" -eq 0 ]; then
|
||||
echo
|
||||
echo "PASS[$DISTRO]: $_PASS_CLAIM"
|
||||
else
|
||||
echo
|
||||
echo "FAIL[$DISTRO]: see above (the VM was torn down regardless)." >&2
|
||||
fi
|
||||
exit "$rc"
|
||||
}
|
||||
|
||||
# The two variants differ only in whether the prerequisites get installed
|
||||
# before install.sh runs, which is exactly the question each one asks:
|
||||
#
|
||||
# test a bare host — assert install.sh is SOUND (installs cleanly, or
|
||||
# declines with an actionable prerequisite error).
|
||||
# test-ready prerequisites satisfied — assert install.sh actually LANDS.
|
||||
_test_cycle() {
|
||||
local with_prereqs="$1"
|
||||
require_linux; require_kvm; require_tools; known_distro
|
||||
IN_TEST=1
|
||||
RUN_DIR="$(mktemp -d "${TMPDIR:-/tmp}/bb-install-test.$DISTRO.XXXXXX")"
|
||||
|
||||
# Arm teardown BEFORE cmd_up: its wait_for_ssh can fail after QEMU is
|
||||
# already running (e.g. a guest that never opens SSH), and without the trap
|
||||
# in place that would leak the VM.
|
||||
trap _test_teardown EXIT INT TERM
|
||||
echo "== [1/$_STEPS] up ($DISTRO) =="
|
||||
cmd_up
|
||||
|
||||
local step=2
|
||||
if [ "$with_prereqs" = 1 ]; then
|
||||
echo
|
||||
echo "== [$step/$_STEPS] prereqs =="
|
||||
cmd_prereqs || { echo "error: could not install prerequisites (see above)." >&2; return 1; }
|
||||
step=$(( step + 1 ))
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "== [$step/$_STEPS] run =="
|
||||
cmd_run
|
||||
step=$(( step + 1 ))
|
||||
|
||||
echo
|
||||
echo "== [$step/$_STEPS] verdict =="
|
||||
# install.sh exits 0 even when doctor reports unmet prerequisites, so the
|
||||
# install succeeding is not the verdict — this is.
|
||||
cmd_status || {
|
||||
echo "error: the install is not sound for $DISTRO (see above)." >&2
|
||||
echo " re-run with BB_TEST_KEEP=1 to keep the VM and dig in." >&2
|
||||
return 1
|
||||
}
|
||||
}
|
||||
|
||||
cmd_test() {
|
||||
_STEPS=4
|
||||
_PASS_CLAIM="on a bare $DISTRO host, install.sh behaves soundly."
|
||||
_test_cycle 0
|
||||
}
|
||||
|
||||
cmd_test_ready() {
|
||||
_STEPS=5
|
||||
_PASS_CLAIM="a $DISTRO host with prerequisites satisfied installs bot-bottle cleanly."
|
||||
# Prerequisites are provisioned, so a graceful decline is no longer an
|
||||
# acceptable outcome — the install must actually land.
|
||||
_REQUIRE_INSTALL=1
|
||||
_test_cycle 1
|
||||
}
|
||||
|
||||
cmd_test_all() {
|
||||
require_linux; require_kvm; require_tools
|
||||
local -a passed=() failed=()
|
||||
local port="$SSH_PORT"
|
||||
local script
|
||||
script="$_SCRIPT_DIR/$(basename "${BASH_SOURCE[0]}")"
|
||||
# Every distro × both variants. Each cell gets its own forwarded port so a
|
||||
# leftover from a prior cell can't collide, and its own subshell so one
|
||||
# cell's failure (or teardown trap) doesn't abort the matrix.
|
||||
for sub in test test-ready; do
|
||||
for d in "${ALL_DISTROS[@]}"; do
|
||||
echo
|
||||
echo "########################################################"
|
||||
echo "# $d ($sub)"
|
||||
echo "########################################################"
|
||||
if ( DISTRO="$d" SSH_PORT="$port" \
|
||||
BB_TEST_DISTRO="$d" BB_TEST_SSH_PORT="$port" \
|
||||
bash "$script" "$sub" ); then
|
||||
passed+=("$d/$sub")
|
||||
else
|
||||
failed+=("$d/$sub")
|
||||
fi
|
||||
port=$(( port + 1 ))
|
||||
done
|
||||
done
|
||||
echo
|
||||
echo "== matrix summary =="
|
||||
echo " PASS: ${passed[*]:-(none)}"
|
||||
echo " FAIL: ${failed[*]:-(none)}"
|
||||
[ "${#failed[@]}" -eq 0 ]
|
||||
}
|
||||
|
||||
case "${1:-}" in
|
||||
test) cmd_test ;;
|
||||
test-ready) cmd_test_ready ;;
|
||||
test-all) cmd_test_all ;;
|
||||
up) cmd_up ;;
|
||||
prereqs) cmd_prereqs ;;
|
||||
run) cmd_run ;;
|
||||
status) cmd_status ;;
|
||||
down) cmd_down ;;
|
||||
ssh) cmd_ssh ;;
|
||||
*) echo "usage: $0 {test|test-ready|test-all|up|prereqs|run|status|down|ssh} (distro via BB_TEST_DISTRO)" >&2; exit 2 ;;
|
||||
esac
|
||||
@@ -7,6 +7,7 @@ the smoke-test no-op — the logic that must hold without a VM.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
@@ -91,6 +92,106 @@ class TestBuildAgentRootfsDir(unittest.TestCase):
|
||||
self.assertNotEqual(first, second)
|
||||
|
||||
|
||||
class TestBuildContext(unittest.TestCase):
|
||||
"""The COPY-source parsing + context shipping that lets a Dockerfile pin an
|
||||
input by COPYing a committed file (codex checksum list, claude/pi npm
|
||||
lockfiles) through the otherwise-empty VM-side build context."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.root = Path(self._tmp.name)
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
codex = self.root / "bot_bottle" / "contrib" / "codex"
|
||||
codex.mkdir(parents=True)
|
||||
self.sums = codex / "codex-package_SHA256SUMS"
|
||||
self.sums.write_text("aaaa codex-package-x86_64-unknown-linux-musl.tar.gz\n")
|
||||
self.dockerfile = self.root / "Dockerfile"
|
||||
self.dockerfile.write_text(
|
||||
"FROM node:22-slim\n"
|
||||
"COPY --chown=node:node "
|
||||
"bot_bottle/contrib/codex/codex-package_SHA256SUMS /tmp/x\n"
|
||||
)
|
||||
|
||||
def _patch_root(self):
|
||||
return patch.object(
|
||||
image_builder.resources, "build_root", return_value=self.root)
|
||||
|
||||
def test_copy_sources_parses_flags_and_continuations(self):
|
||||
df = self.root / "Multi"
|
||||
df.write_text(
|
||||
"FROM x\n"
|
||||
"COPY a/one.json \\\n a/two.json /dest/\n"
|
||||
"COPY --from=builder /built /built\n" # excluded: build stage
|
||||
"COPY --chown=n:n b/three /dest\n"
|
||||
)
|
||||
self.assertEqual(
|
||||
image_builder._context_copy_sources(df),
|
||||
["a/one.json", "a/two.json", "b/three"],
|
||||
)
|
||||
|
||||
def test_context_files_resolves_existing_under_root(self):
|
||||
with self._patch_root():
|
||||
files = image_builder._context_files(self.dockerfile)
|
||||
self.assertEqual(
|
||||
[rel for rel, _ in files],
|
||||
["bot_bottle/contrib/codex/codex-package_SHA256SUMS"],
|
||||
)
|
||||
|
||||
def test_context_files_recurses_into_directory_sources(self):
|
||||
nested = self.root / "assets" / "nested"
|
||||
nested.mkdir(parents=True)
|
||||
(self.root / "assets" / "one").write_text("one")
|
||||
(nested / "two").write_text("two")
|
||||
df = self.root / "Directory"
|
||||
df.write_text("FROM x\nCOPY assets /opt/assets\n")
|
||||
with self._patch_root():
|
||||
files = image_builder._context_files(df)
|
||||
self.assertEqual(
|
||||
[rel for rel, _ in files],
|
||||
["assets/nested/two", "assets/one"],
|
||||
)
|
||||
|
||||
def test_context_files_drops_missing_and_traversal(self):
|
||||
df = self.root / "Bad"
|
||||
df.write_text("FROM x\nCOPY ../escape /d\nCOPY does/not/exist /d\n")
|
||||
with self._patch_root():
|
||||
self.assertEqual(image_builder._context_files(df), [])
|
||||
|
||||
def test_rootfs_digest_tracks_context_file_content(self):
|
||||
with self._patch_root(), \
|
||||
patch.object(image_builder.util, "cache_dir", return_value=self.root):
|
||||
first = image_builder._rootfs_digest(self.dockerfile)
|
||||
self.sums.write_text("bbbb codex-package-x86_64-unknown-linux-musl.tar.gz\n")
|
||||
second = image_builder._rootfs_digest(self.dockerfile)
|
||||
self.assertNotEqual(first, second)
|
||||
|
||||
def test_send_build_context_noop_without_copy(self):
|
||||
df = self.root / "None"
|
||||
df.write_text("FROM x\n")
|
||||
with self._patch_root(), \
|
||||
patch.object(image_builder.subprocess, "Popen") as popen:
|
||||
image_builder._send_build_context(Path("/k"), "10.0.0.1", df, "/tmp/c")
|
||||
popen.assert_not_called()
|
||||
|
||||
def test_send_build_context_streams_copied_files_to_vm(self):
|
||||
completed = subprocess.CompletedProcess([], 0, stdout=b"", stderr=b"")
|
||||
with self._patch_root(), \
|
||||
patch.object(image_builder.subprocess, "Popen") as popen, \
|
||||
patch.object(image_builder.subprocess, "run",
|
||||
return_value=completed) as run:
|
||||
popen.return_value.stdout = None
|
||||
popen.return_value.returncode = 0
|
||||
popen.return_value.wait.return_value = 0
|
||||
image_builder._send_build_context(
|
||||
Path("/k"), "10.0.0.1", self.dockerfile, "/tmp/c")
|
||||
tar_argv = popen.call_args.args[0]
|
||||
self.assertEqual(tar_argv[:3], ["tar", "-C", str(self.root)])
|
||||
self.assertIn("--", tar_argv)
|
||||
self.assertIn(
|
||||
"bot_bottle/contrib/codex/codex-package_SHA256SUMS", tar_argv)
|
||||
self.assertIn("tar -C /tmp/c/ctx -xf -", run.call_args.args[0][-1])
|
||||
|
||||
|
||||
class TestSmokeTest(unittest.TestCase):
|
||||
def test_buildah_receives_centralized_image_build_args(self):
|
||||
with patch.object(
|
||||
@@ -114,7 +215,6 @@ class TestSmokeTest(unittest.TestCase):
|
||||
ssh.assert_not_called()
|
||||
|
||||
def test_failed_smoke_dies(self):
|
||||
import subprocess
|
||||
result = subprocess.CompletedProcess([], 1, stdout="broken", stderr="")
|
||||
with patch.object(image_builder, "_ssh", return_value=result), \
|
||||
self.assertRaises(SystemExit):
|
||||
|
||||
@@ -0,0 +1,98 @@
|
||||
"""Unit: how the CLI tells users to re-run it under sudo.
|
||||
|
||||
`sudo bot-bottle …` is wrong for the users who followed the documented
|
||||
install: sudo's secure_path excludes ~/.local/bin, where both pipx and
|
||||
install.sh put the entry point. These lock in the absolute-path form.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import io
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
from bot_bottle import invocation
|
||||
|
||||
|
||||
class TestSelfPath(unittest.TestCase):
|
||||
def test_absolute_argv0_is_used_as_is(self):
|
||||
with mock.patch.object(invocation.sys, "argv", ["/opt/venv/bin/bot-bottle"]):
|
||||
self.assertEqual("/opt/venv/bin/bot-bottle", invocation.self_path())
|
||||
|
||||
def test_bare_name_is_resolved_through_path(self):
|
||||
# The case that matters: invoked as `bot-bottle`, installed in a
|
||||
# directory sudo would drop.
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
entry = Path(d, "bot-bottle")
|
||||
entry.write_text("#!/bin/sh\n")
|
||||
entry.chmod(0o755)
|
||||
with mock.patch.object(invocation.sys, "argv", ["bot-bottle"]), \
|
||||
mock.patch.dict(os.environ, {"PATH": d}):
|
||||
self.assertEqual(str(entry), invocation.self_path())
|
||||
|
||||
def test_relative_path_is_made_absolute(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
entry = Path(d, "bot-bottle")
|
||||
entry.write_text("#!/bin/sh\n")
|
||||
entry.chmod(0o755)
|
||||
cwd = os.getcwd()
|
||||
try:
|
||||
os.chdir(d)
|
||||
with mock.patch.object(invocation.sys, "argv", ["./bot-bottle"]):
|
||||
self.assertTrue(os.path.isabs(invocation.self_path()))
|
||||
finally:
|
||||
os.chdir(cwd)
|
||||
|
||||
def test_unresolvable_entry_point_falls_back_to_the_name(self):
|
||||
# `python -m`-style invocation, or an argv[0] that no longer exists.
|
||||
# A slightly wrong hint beats a traceback raised while reporting some
|
||||
# unrelated problem.
|
||||
with mock.patch.object(invocation.sys, "argv", ["/nonexistent/gone"]), \
|
||||
mock.patch.object(invocation.shutil, "which", return_value=None):
|
||||
self.assertEqual("/nonexistent/gone", invocation.self_path())
|
||||
with mock.patch.object(invocation.sys, "argv", [""]), \
|
||||
mock.patch.object(invocation.shutil, "which", return_value=None):
|
||||
self.assertEqual("bot-bottle", invocation.self_path())
|
||||
|
||||
|
||||
class TestSudoCommand(unittest.TestCase):
|
||||
def test_names_an_absolute_path_not_the_bare_command(self):
|
||||
with mock.patch.object(invocation, "self_path",
|
||||
return_value="/home/u/.local/bin/bot-bottle"):
|
||||
cmd = invocation.sudo_command("backend", "setup", "--backend=firecracker")
|
||||
self.assertEqual(
|
||||
"sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker",
|
||||
cmd,
|
||||
)
|
||||
# The regression this exists to prevent.
|
||||
self.assertNotIn("sudo bot-bottle", cmd)
|
||||
|
||||
|
||||
class TestFirecrackerSetupUsesIt(unittest.TestCase):
|
||||
def test_root_reinvocation_hint_names_an_absolute_path(self):
|
||||
# The message that prompted all this. Drive the real code path rather
|
||||
# than scanning the source, which would also match the comment
|
||||
# explaining why the bare form is wrong.
|
||||
from bot_bottle.backend.firecracker import setup as fc_setup
|
||||
|
||||
err, out = io.StringIO(), io.StringIO()
|
||||
with mock.patch.object(fc_setup.os, "geteuid", return_value=501), \
|
||||
mock.patch.object(fc_setup.invocation, "self_path",
|
||||
return_value="/home/u/.local/bin/bot-bottle"), \
|
||||
contextlib.redirect_stderr(err), contextlib.redirect_stdout(out):
|
||||
fc_setup._setup_systemd()
|
||||
|
||||
printed = err.getvalue()
|
||||
self.assertIn(
|
||||
"sudo /home/u/.local/bin/bot-bottle backend setup --backend=firecracker",
|
||||
printed,
|
||||
)
|
||||
self.assertNotIn("sudo bot-bottle", printed)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -151,31 +151,6 @@ class TestReprovisionGateway(unittest.TestCase):
|
||||
self.c.reprovision_gateway("b1", "key")
|
||||
|
||||
|
||||
class TestUpdateAgentSecret(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.c = OrchestratorClient("http://orch:8080")
|
||||
|
||||
def test_success_posts_single_secret(self) -> None:
|
||||
with patch(_URLOPEN, return_value=_resp(200, {"updated": True})) as opened:
|
||||
self.assertTrue(self.c.update_agent_secret("b1", "A", "fresh", "key"))
|
||||
request = opened.call_args.args[0]
|
||||
self.assertEqual("POST", request.get_method())
|
||||
self.assertTrue(request.full_url.endswith("/bottles/b1/secret"))
|
||||
self.assertEqual(
|
||||
{"name": "A", "value": "fresh", "env_var_secret": "key"},
|
||||
json.loads(request.data),
|
||||
)
|
||||
|
||||
def test_unknown_bottle_is_false(self) -> None:
|
||||
with patch(_URLOPEN, side_effect=_http_error(404)):
|
||||
self.assertFalse(self.c.update_agent_secret("b1", "A", "v", "key"))
|
||||
|
||||
def test_other_status_raises(self) -> None:
|
||||
with patch(_URLOPEN, side_effect=_http_error(400)):
|
||||
with self.assertRaises(OrchestratorClientError):
|
||||
self.c.update_agent_secret("b1", "A", "v", "key")
|
||||
|
||||
|
||||
class TestHealthAndPolicy(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.c = OrchestratorClient("http://orch:8080")
|
||||
|
||||
@@ -199,20 +199,6 @@ class TestAgentSecrets(unittest.TestCase):
|
||||
got = self.store.get_agent_secrets("bottle-1")
|
||||
self.assertEqual({"K": "new", "K2": "v2"}, got)
|
||||
|
||||
def test_store_agent_secret_upserts_one_key_leaving_others(self) -> None:
|
||||
self.store.store_agent_secrets("bottle-1", {"K": "v1", "K2": "v2"})
|
||||
self.store.store_agent_secret("bottle-1", "K", "v1-new") # update existing
|
||||
self.store.store_agent_secret("bottle-1", "K3", "v3") # insert new
|
||||
self.assertEqual(
|
||||
{"K": "v1-new", "K2": "v2", "K3": "v3"},
|
||||
self.store.get_agent_secrets("bottle-1"),
|
||||
)
|
||||
|
||||
def test_store_agent_secret_does_not_duplicate_rows(self) -> None:
|
||||
self.store.store_agent_secret("bottle-1", "K", "v1")
|
||||
self.store.store_agent_secret("bottle-1", "K", "v2")
|
||||
self.assertEqual({"K": "v2"}, self.store.get_agent_secrets("bottle-1"))
|
||||
|
||||
def test_delete_removes_secrets(self) -> None:
|
||||
self.store.store_agent_secrets("bottle-1", {"K": "v"})
|
||||
self.store.delete_agent_secrets("bottle-1")
|
||||
|
||||
@@ -125,47 +125,6 @@ class TestDispatch(unittest.TestCase):
|
||||
{"EGRESS_TOKEN_0": "upstream-secret"}, self.orch.tokens_for(bottle_id),
|
||||
)
|
||||
|
||||
def test_update_agent_secret_in_place(self) -> None:
|
||||
key = base64.urlsafe_b64encode(b"unit-test-key").rstrip(b"=").decode()
|
||||
status, payload = dispatch(
|
||||
self.orch, "POST", "/bottles", _body({
|
||||
"source_ip": "10.243.0.21",
|
||||
"tokens": {"A": "old", "B": "keep"},
|
||||
"env_var_secret": key,
|
||||
}),
|
||||
)
|
||||
self.assertEqual(201, status)
|
||||
bottle_id = payload["bottle_id"]
|
||||
assert isinstance(bottle_id, str)
|
||||
status, response = dispatch(
|
||||
self.orch, "POST", f"/bottles/{bottle_id}/secret",
|
||||
_body({"name": "A", "value": "fresh", "env_var_secret": key}),
|
||||
)
|
||||
self.assertEqual((200, {"updated": True}), (status, response))
|
||||
self.assertEqual({"A": "fresh", "B": "keep"}, self.orch.tokens_for(bottle_id))
|
||||
|
||||
def test_update_agent_secret_validates_and_unknown_bottle(self) -> None:
|
||||
status, _ = dispatch(self.orch, "POST", "/bottles/b1/secret", b"not-json")
|
||||
self.assertEqual(400, status)
|
||||
status, _ = dispatch(
|
||||
self.orch, "POST", "/bottles/b1/secret", _body({"name": "A"}),
|
||||
)
|
||||
self.assertEqual(400, status) # value + env_var_secret missing
|
||||
status, _ = dispatch(
|
||||
self.orch, "POST", "/bottles/ghost/secret",
|
||||
_body({"name": "A", "value": "v", "env_var_secret": "k"}),
|
||||
)
|
||||
self.assertEqual(404, status)
|
||||
|
||||
def test_update_agent_secret_is_cli_only(self) -> None:
|
||||
# A data-plane (`gateway`) caller must not set a bottle's tokens.
|
||||
status, _ = dispatch(
|
||||
self.orch, "POST", "/bottles/b1/secret",
|
||||
_body({"name": "A", "value": "v", "env_var_secret": "k"}),
|
||||
role=ROLE_GATEWAY,
|
||||
)
|
||||
self.assertEqual(403, status)
|
||||
|
||||
def test_reprovision_validates_request_and_missing_rows(self) -> None:
|
||||
status, _ = dispatch(
|
||||
self.orch, "POST", "/bottles/b1/reprovision_gateway", b"not-json",
|
||||
|
||||
@@ -103,25 +103,6 @@ class TestOrchestrator(unittest.TestCase):
|
||||
self.assertTrue(self.orch.reprovision_from_secret(rec.bottle_id, key))
|
||||
self.assertEqual({"EGRESS_TOKEN_0": "secret"}, self.orch.tokens_for(rec.bottle_id))
|
||||
|
||||
def test_update_agent_secret_refreshes_one_token_in_place(self) -> None:
|
||||
key = new_env_var_secret()
|
||||
rec = self.orch.launch_bottle(
|
||||
"10.243.0.20", tokens={"A": "old-a", "B": "keep-b"}, env_var_secret=key,
|
||||
)
|
||||
self.assertTrue(self.orch.update_agent_secret(rec.bottle_id, "A", "new-a", key))
|
||||
# In memory: A refreshed, B left untouched.
|
||||
self.assertEqual({"A": "new-a", "B": "keep-b"}, self.orch.tokens_for(rec.bottle_id))
|
||||
# At rest: the whole set stays decryptable with the SAME env_var_secret,
|
||||
# so a later reprovision restores the refreshed value (not the launch one).
|
||||
self.orch._tokens.clear()
|
||||
self.assertTrue(self.orch.reprovision_from_secret(rec.bottle_id, key))
|
||||
self.assertEqual({"A": "new-a", "B": "keep-b"}, self.orch.tokens_for(rec.bottle_id))
|
||||
|
||||
def test_update_agent_secret_unknown_bottle_is_false(self) -> None:
|
||||
self.assertFalse(
|
||||
self.orch.update_agent_secret("ghost", "A", "v", new_env_var_secret())
|
||||
)
|
||||
|
||||
def test_reprovision_rejects_missing_rows_and_wrong_key(self) -> None:
|
||||
self.assertFalse(self.orch.reprovision_from_secret("missing", new_env_var_secret()))
|
||||
key = "AQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQE"
|
||||
|
||||
Reference in New Issue
Block a user