Compare commits
31 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 016e59e029 | |||
| f3664dea9f | |||
| 88b82a169e | |||
| 5a9428cc86 | |||
| 60039f2eb3 | |||
| bc42836327 | |||
| 0fc5457e41 | |||
| c0493f0b01 | |||
| 5828f5e900 | |||
| be025ff8fb | |||
| 9537c96586 | |||
| 6b43fe73c1 | |||
| d3370a88bb | |||
| 39167528db | |||
| b25ace4c00 | |||
| 9c06702b32 | |||
| cd0983d943 | |||
| 99176b1edf | |||
| 652f14dcb1 | |||
| 38c13708c7 | |||
| 82669b22d5 | |||
| 1a4b390e8a | |||
| 955cb3bcbd | |||
| 605146d287 | |||
| 27dea58ae1 | |||
| 5401f036a9 | |||
| 238f5f7614 | |||
| 2644759b0d | |||
| e53104d5c1 | |||
| 8f6148d571 | |||
| 45f3cefbc5 |
@@ -0,0 +1,26 @@
|
||||
name: prd-number-check
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, reopened, synchronize]
|
||||
|
||||
jobs:
|
||||
require-numbered-prds:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Reject unnumbered PRDs
|
||||
run: |
|
||||
unnumbered=$(find docs/prds -maxdepth 1 -type f \
|
||||
-name 'prd-new-*.md' -print | sort)
|
||||
|
||||
if [ -n "$unnumbered" ]; then
|
||||
echo "::error::Assign every new PRD its final sequential number before merge."
|
||||
echo "Unnumbered PRDs:"
|
||||
echo "$unnumbered"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "All PRDs have final numbers."
|
||||
@@ -1,122 +0,0 @@
|
||||
# Assign sequential numbers to prd-new-*.md files on merge to main.
|
||||
#
|
||||
# When a PR merges to main and includes prd-new-*.md files this workflow:
|
||||
# 1. Finds the next available NNNN number by scanning existing PRDs.
|
||||
# 2. Renames each prd-new-*.md to NNNN-<slug>.md.
|
||||
# 3. Updates the title header (# PRD prd-new: → # PRD NNNN:).
|
||||
# 4. Flips Status: Draft → Active when the push touched files outside
|
||||
# docs/prds/ anywhere in its commit range (i.e. the implementation
|
||||
# shipped together with the PRD).
|
||||
# 5. Commits the renaming back to main.
|
||||
#
|
||||
# No-op if the working tree contains no prd-new-*.md files.
|
||||
#
|
||||
# NOTE: The workflow scans the working tree (not just HEAD~1..HEAD) because
|
||||
# PRs land as multi-commit pushes and the prd-new file is often added in an
|
||||
# earlier commit on the branch, not in the final squash/merge commit.
|
||||
|
||||
name: prd-number
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
paths:
|
||||
- 'docs/prds/prd-new-*.md'
|
||||
|
||||
jobs:
|
||||
assign-numbers:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
# No actions/setup-python: the inline script is stdlib-only on the
|
||||
# image's system Python 3.12 (older act_runner mishandles its PATH).
|
||||
- name: Configure git
|
||||
run: |
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "github-actions[bot]@users.noreply.github.com"
|
||||
|
||||
- name: Assign PRD numbers
|
||||
run: |
|
||||
python3 - <<'EOF'
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
prds_dir = Path("docs/prds")
|
||||
|
||||
# Scan the working tree — prd-new files may have landed in any
|
||||
# commit of a multi-commit push, not just HEAD.
|
||||
new_prds = sorted(prds_dir.glob("prd-new-*.md"))
|
||||
|
||||
if not new_prds:
|
||||
print("No prd-new-*.md files found — nothing to do.")
|
||||
sys.exit(0)
|
||||
|
||||
# Determine whether non-PRD files were also changed anywhere in
|
||||
# the push range (BEFORE_SHA → HEAD). Falls back to HEAD~1 when
|
||||
# the env var isn't set (e.g. local act runs).
|
||||
before_sha = os.environ.get("GITHUB_EVENT_BEFORE", "HEAD~1")
|
||||
all_changed = subprocess.run(
|
||||
["git", "diff", "--name-only", before_sha, "HEAD"],
|
||||
capture_output=True, text=True, check=True,
|
||||
).stdout.splitlines()
|
||||
non_prd_changed = any(
|
||||
not f.startswith("docs/prds/") for f in all_changed
|
||||
)
|
||||
|
||||
# Find next available number.
|
||||
existing = sorted(
|
||||
int(m.group(1))
|
||||
for p in prds_dir.glob("*.md")
|
||||
if (m := re.match(r"^(\d{4})-", p.name))
|
||||
)
|
||||
next_num = (max(existing) + 1) if existing else 1
|
||||
|
||||
for prd_path in sorted(new_prds):
|
||||
slug = re.sub(r"^prd-new-", "", prd_path.stem)
|
||||
new_name = f"{next_num:04d}-{slug}.md"
|
||||
new_path = prds_dir / new_name
|
||||
print(f" {prd_path.name} → {new_name}")
|
||||
|
||||
content = prd_path.read_text()
|
||||
|
||||
# Update title header.
|
||||
content = re.sub(
|
||||
r"^(#\s+PRD\s+)prd-new(:)",
|
||||
rf"\g<1>{next_num:04d}\2",
|
||||
content,
|
||||
count=1,
|
||||
flags=re.MULTILINE,
|
||||
)
|
||||
|
||||
# Conditionally flip Status.
|
||||
if non_prd_changed:
|
||||
content = re.sub(
|
||||
r"(\*\*Status:\*\*\s*)Draft",
|
||||
r"\g<1>Active",
|
||||
content,
|
||||
count=1,
|
||||
)
|
||||
|
||||
new_path.write_text(content)
|
||||
subprocess.run(["git", "rm", str(prd_path)], check=True)
|
||||
subprocess.run(["git", "add", str(new_path)], check=True)
|
||||
next_num += 1
|
||||
|
||||
subprocess.run(
|
||||
["git", "commit", "-m", "ci(prd): assign sequential numbers to new PRDs"],
|
||||
check=True,
|
||||
)
|
||||
subprocess.run(["git", "push"], check=True)
|
||||
EOF
|
||||
@@ -0,0 +1,337 @@
|
||||
# Run the complete backend test suite before a release. This workflow is
|
||||
# intentionally manual because Firecracker and macOS use privileged,
|
||||
# self-hosted runners.
|
||||
#
|
||||
# The suite uses stdlib `unittest` discovery — no external Python
|
||||
# dependencies are required to execute it. Tests are split by directory:
|
||||
#
|
||||
# tests/unit/ — pure unit tests; always run
|
||||
# tests/integration/ — need a reachable backend; skip cleanly when
|
||||
# the backend isn't available on the runner
|
||||
# tests/canaries/ — upstream regression canaries; run on a separate
|
||||
# schedule (see canaries.yml), not here
|
||||
#
|
||||
# Unit, Docker, and Firecracker run once under coverage and upload a small
|
||||
# .coverage.* artifact for the combined coverage job. macOS reports coverage
|
||||
# in place because it is an advisory host-mode runner.
|
||||
|
||||
name: pre-release-test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
unit:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# No actions/setup-python: the runner image already ships Python 3.12,
|
||||
# and older act_runner engines mishandle setup-python's PATH (coverage
|
||||
# lands in one interpreter, `python3` resolves to another). Install
|
||||
# straight into the ephemeral job container's system Python —
|
||||
# --break-system-packages is safe because the container is disposable.
|
||||
- name: Install dev requirements
|
||||
run: python3 -m pip install --break-system-packages -r requirements-dev.txt
|
||||
|
||||
- name: Run unit tests with coverage
|
||||
env:
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.unit
|
||||
run: python3 -m coverage run -m unittest discover -t . -s tests/unit -v
|
||||
|
||||
- name: Report unit coverage
|
||||
env:
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.unit
|
||||
run: python3 -m coverage report -m
|
||||
|
||||
# upload-artifact@v3's glob skips dotfiles, so a bare `.coverage.unit`
|
||||
# silently uploads nothing ("No files were found"). Stage it under a
|
||||
# non-dot name; the coverage job renames it back before `coverage
|
||||
# combine`. `cp` also fails loudly if coverage never wrote the file.
|
||||
- name: Stage unit coverage for upload
|
||||
run: cp .coverage.unit coverage-unit.dat
|
||||
|
||||
- name: Upload unit coverage artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: coverage-unit
|
||||
path: coverage-unit.dat
|
||||
|
||||
integration-docker:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# No actions/setup-python (see the note in the `unit` job); the
|
||||
# container's system Python 3.12 runs the stdlib test suite directly.
|
||||
- name: Install coverage
|
||||
run: python3 -m pip install --break-system-packages coverage
|
||||
|
||||
# Fail loudly if the backend this job promises isn't actually usable,
|
||||
# rather than letting every test silently `unittest.skip` and the job
|
||||
# go green on zero coverage. `backend status` prints a clear per-check
|
||||
# summary (docker on PATH, daemon reachable) and exits non-zero when a
|
||||
# prerequisite is missing — the same readiness check the skip guards
|
||||
# gate on via `has_backend`.
|
||||
- name: Preflight — Docker backend is ready
|
||||
run: |
|
||||
python3 --version
|
||||
python3 cli.py backend status --backend=docker
|
||||
|
||||
- name: Run integration tests (docker) with coverage
|
||||
env:
|
||||
BOT_BOTTLE_BACKEND: docker
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.docker
|
||||
run: python3 -m coverage run -m unittest discover -t . -s tests/integration -v
|
||||
|
||||
# Non-dot name so upload-artifact's dotfile-skipping glob picks it up.
|
||||
- name: Stage docker coverage for upload
|
||||
run: cp .coverage.docker coverage-docker.dat
|
||||
|
||||
- name: Upload docker coverage artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: coverage-docker
|
||||
path: coverage-docker.dat
|
||||
|
||||
# Integration tests against the Firecracker backend. Runs on a self-hosted
|
||||
# KVM runner (label `kvm`) where /dev/kvm and the TAP/nft pool are available.
|
||||
#
|
||||
# Manual only: the privileged KVM runner does not execute proposed changes
|
||||
# unattended.
|
||||
#
|
||||
# Runner prerequisites (provision once; see README "Firecracker on Linux"):
|
||||
# `firecracker` on PATH, `/dev/kvm` accessible, cached kernel +
|
||||
# static dropbear at /var/cache/bot-bottle-fc/dropbear, and the pool as a
|
||||
# persistent systemd unit.
|
||||
#
|
||||
# The infra candidate is built here directly (no artifact download) to
|
||||
# eliminate the ~70 s ubuntu-latest upload + ~83 s combined download that
|
||||
# the old build-infra → integration-firecracker + coverage chain incurred.
|
||||
integration-firecracker:
|
||||
runs-on: [self-hosted, kvm]
|
||||
if: github.event_name == 'workflow_dispatch'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Preflight — Firecracker host is ready
|
||||
run: |
|
||||
command -v firecracker >/dev/null || {
|
||||
echo "firecracker not on PATH — provision the runner (README: Firecracker on Linux)"; exit 1; }
|
||||
test -e /dev/kvm || { echo "/dev/kvm missing — KVM not available on this runner"; exit 1; }
|
||||
# `backend status` exits non-zero unless the TAP pool is up + no
|
||||
# range overlap; it prints the exact `backend setup` fix.
|
||||
python3 cli.py backend status --backend=firecracker
|
||||
|
||||
- name: Build infra candidate from this checkout
|
||||
env:
|
||||
BOT_BOTTLE_FC_DROPBEAR: /var/cache/bot-bottle-fc/dropbear
|
||||
run: python3 -m bot_bottle.backend.firecracker.publish_infra --output infra-candidate --reuse-published
|
||||
|
||||
- name: Replace the persistent infra VM with the candidate
|
||||
run: python3 -c 'from bot_bottle.backend.firecracker import infra_vm; infra_vm.stop()'
|
||||
|
||||
# No dev-requirements install: `coverage` is already provided by the
|
||||
# self-hosted runner's Nix python env, and that env has no `pip`
|
||||
# module to install into anyway.
|
||||
- name: Run integration tests (firecracker) with coverage
|
||||
env:
|
||||
BOT_BOTTLE_BACKEND: firecracker
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_DIR: ${{ github.workspace }}/infra-candidate
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.firecracker
|
||||
run: python3 -m coverage run -m unittest discover -t . -s tests/integration -v
|
||||
|
||||
- name: Stage firecracker coverage for upload
|
||||
run: cp .coverage.firecracker coverage-firecracker.dat
|
||||
|
||||
- name: Upload firecracker coverage artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: coverage-firecracker
|
||||
path: coverage-firecracker.dat
|
||||
|
||||
- name: Upload tested rootfs
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate/
|
||||
|
||||
- name: Upload dropbear for publish verification
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: /var/cache/bot-bottle-fc/dropbear
|
||||
|
||||
# Integration tests against the macOS Apple Container backend. Runs on a
|
||||
# self-hosted macOS runner (label `macos`) registered in HOST mode — Apple
|
||||
# Container needs the host `container` CLI + virtualization framework and
|
||||
# cannot run inside a Linux container, so this cannot reuse the KVM runner.
|
||||
#
|
||||
# Advisory only: workflow_dispatch (manual) exclusively — never push or
|
||||
# pull_request. A single non-redundant laptop that sleeps/roams must not run
|
||||
# unattended on every push to main, let alone block a PR merge, so this job is
|
||||
# deliberately NOT in the `coverage` job's `needs` and its coverage never
|
||||
# feeds the diff-coverage gate. Dispatch-only also means no fork PR (or any
|
||||
# push) ever executes on the host-mode runner.
|
||||
#
|
||||
# The infra container is a singleton (`bot-bottle-mac-infra`); the
|
||||
# `concurrency` group serializes runs so two never collide on it (#425), and
|
||||
# the always-run teardown removes it so a crashed run can't wedge the next.
|
||||
#
|
||||
# Runner prerequisites (provision once; see README "macOS Apple Container"):
|
||||
# the `container` CLI on PATH with `container system status` running, and a
|
||||
# Python >=3.11 with `coverage` importable on the launchd service PATH.
|
||||
integration-macos:
|
||||
runs-on: [self-hosted, macos]
|
||||
if: github.event_name == 'workflow_dispatch'
|
||||
concurrency:
|
||||
group: integration-macos-infra
|
||||
cancel-in-progress: false
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# Fail loudly if the backend this job promises isn't actually usable,
|
||||
# rather than letting every test silently `unittest.skip` and the job go
|
||||
# green on zero coverage. `backend status` exits non-zero (and prints the
|
||||
# per-check summary) when the `container` CLI or its system service is
|
||||
# missing — the same readiness check the skip guards gate on.
|
||||
- name: Preflight — Apple Container backend is ready
|
||||
run: |
|
||||
command -v container >/dev/null || {
|
||||
echo "container CLI not on PATH — provision the runner (README: macOS Apple Container)"; exit 1; }
|
||||
container system status || {
|
||||
echo "container system service not running — run 'container system start'"; exit 1; }
|
||||
python3 cli.py backend status --backend=macos-container
|
||||
|
||||
# `coverage` comes from the runner's provisioned Python (no pip install
|
||||
# into the host interpreter). Advisory job: report coverage in-line for
|
||||
# visibility but don't upload — it never feeds the combined gate.
|
||||
- name: Run integration tests (macos-container) with coverage
|
||||
env:
|
||||
BOT_BOTTLE_BACKEND: macos-container
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.macos
|
||||
run: python3 -m coverage run -m unittest discover -t . -s tests/integration -v
|
||||
|
||||
- name: Report macos coverage
|
||||
env:
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.macos
|
||||
run: python3 -m coverage report -m
|
||||
|
||||
# On failure, capture the infra containers' state and logs BEFORE the
|
||||
# teardown below removes them — otherwise a control-plane crash is
|
||||
# undiagnosable from CI, since `stop()` deletes the orchestrator (and its
|
||||
# logs) on every run. Best-effort: never let the diagnostics themselves
|
||||
# fail the job, and keep going if a container is already gone.
|
||||
- name: Dump infra diagnostics (on failure)
|
||||
if: failure()
|
||||
run: |
|
||||
set +e
|
||||
echo "=== containers ==="
|
||||
container ls -a | grep bot-bottle-mac || echo "(no bot-bottle-mac containers)"
|
||||
echo "=== networks ==="
|
||||
container network ls | grep bot-bottle-mac || echo "(no bot-bottle-mac networks)"
|
||||
for c in bot-bottle-mac-orchestrator bot-bottle-mac-infra; do
|
||||
echo "=== inspect $c ==="
|
||||
container inspect "$c" || echo "($c not found)"
|
||||
echo "=== logs $c ==="
|
||||
container logs "$c" || echo "($c logs unavailable)"
|
||||
done
|
||||
exit 0
|
||||
|
||||
# Remove the singleton infra container so a crashed or cancelled run
|
||||
# cannot leave `bot-bottle-mac-infra` wedged for the next job.
|
||||
- name: Teardown infra singleton
|
||||
if: always()
|
||||
run: python3 -c 'from bot_bottle.backend.macos_container.infra import MacosInfraService; MacosInfraService().stop()'
|
||||
|
||||
# Combined coverage gate: aggregates .coverage.* artifacts uploaded by each
|
||||
# test job, then runs the diff-coverage gate (new/changed lines >= 90%).
|
||||
#
|
||||
# Runs on ubuntu-latest — no KVM needed, no test reruns. Coverage files use
|
||||
# relative_files = True (.coveragerc) so they combine cleanly across runners.
|
||||
# Each test job sets COVERAGE_FILE to an absolute path so coverage.py writes
|
||||
# to a known location that upload-artifact can find regardless of runner env.
|
||||
#
|
||||
coverage:
|
||||
needs: [unit, integration-docker, integration-firecracker]
|
||||
timeout-minutes: 15
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Install coverage
|
||||
run: python3 -m pip install --break-system-packages coverage
|
||||
|
||||
- name: Download unit coverage artifact
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: coverage-unit
|
||||
path: ${{ github.workspace }}
|
||||
|
||||
- name: Download docker coverage artifact
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: coverage-docker
|
||||
path: ${{ github.workspace }}
|
||||
|
||||
- name: Download firecracker coverage artifact
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: coverage-firecracker
|
||||
path: ${{ github.workspace }}
|
||||
|
||||
# Rename the non-dot upload names back to the .coverage.* files that
|
||||
# `coverage combine` discovers (see the staging steps in each test job).
|
||||
- name: Reassemble coverage data files
|
||||
run: |
|
||||
mv coverage-unit.dat .coverage.unit
|
||||
mv coverage-docker.dat .coverage.docker
|
||||
mv coverage-firecracker.dat .coverage.firecracker
|
||||
|
||||
- name: Combined coverage (unit + integration, incl. firecracker)
|
||||
run: PYTHON=python3 bash scripts/coverage.sh aggregate critical
|
||||
|
||||
- name: Diff-coverage gate (changed lines >= 90%)
|
||||
run: |
|
||||
git fetch --no-tags origin main:refs/remotes/origin/main
|
||||
python3 scripts/diff_coverage.py --base origin/main --min 90
|
||||
|
||||
publish-infra:
|
||||
needs:
|
||||
- unit
|
||||
- integration-docker
|
||||
- integration-firecracker
|
||||
- integration-macos
|
||||
- coverage
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout the tested revision
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Download the tested rootfs
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate
|
||||
|
||||
# publish_infra re-derives the version from the checkout to confirm the
|
||||
# bundle matches before uploading, and the version hashes the dropbear
|
||||
# bytes. Download the same dropbear integration-firecracker used.
|
||||
- name: Download the staged dropbear
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: firecracker-inputs
|
||||
|
||||
- name: Publish the tested candidate
|
||||
env:
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_TOKEN: ${{ secrets.BOT_BOTTLE_INFRA_ARTIFACT_TOKEN }}
|
||||
BOT_BOTTLE_FC_DROPBEAR: ${{ github.workspace }}/firecracker-inputs/dropbear
|
||||
run: python3 -m bot_bottle.backend.firecracker.publish_infra --publish-dir infra-candidate
|
||||
+7
-176
@@ -1,21 +1,6 @@
|
||||
# Run the project's test suite when package or runtime inputs change on a PR
|
||||
# or on push to main.
|
||||
#
|
||||
# The suite uses stdlib `unittest` discovery — no external Python
|
||||
# dependencies are required to execute it. Tests are split by directory:
|
||||
#
|
||||
# tests/unit/ — pure unit tests; always run
|
||||
# tests/integration/ — need a reachable backend; skip cleanly when
|
||||
# the backend isn't available on the runner
|
||||
# tests/canaries/ — upstream regression canaries; run on a separate
|
||||
# schedule (see canaries.yml), not here
|
||||
#
|
||||
# Each test job runs once under coverage and uploads a small .coverage.*
|
||||
# artifact. The `coverage` job combines them — no test reruns, no KVM
|
||||
# dependency on that job. For main-branch pushes only, the tested rootfs
|
||||
# and matching dropbear are uploaded so `publish-infra` can publish the
|
||||
# byte-identical artifact that was tested. PRs avoid the ~194 MB rootfs
|
||||
# transfer entirely.
|
||||
# Run the automated test gate when package or runtime inputs change on a PR
|
||||
# or on push to main. Privileged self-hosted backends live in the manually
|
||||
# dispatched pre-release-test workflow.
|
||||
|
||||
name: test
|
||||
|
||||
@@ -37,6 +22,7 @@ on:
|
||||
- 'requirements-dev.txt'
|
||||
- '.coveragerc'
|
||||
- '.dockerignore'
|
||||
- '.gitea/workflows/test.yml'
|
||||
pull_request:
|
||||
paths:
|
||||
- 'bot_bottle/**'
|
||||
@@ -52,7 +38,7 @@ on:
|
||||
- 'requirements-dev.txt'
|
||||
- '.coveragerc'
|
||||
- '.dockerignore'
|
||||
workflow_dispatch:
|
||||
- '.gitea/workflows/test.yml'
|
||||
|
||||
jobs:
|
||||
unit:
|
||||
@@ -61,11 +47,6 @@ jobs:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# No actions/setup-python: the runner image already ships Python 3.12,
|
||||
# and older act_runner engines mishandle setup-python's PATH (coverage
|
||||
# lands in one interpreter, `python3` resolves to another). Install
|
||||
# straight into the ephemeral job container's system Python —
|
||||
# --break-system-packages is safe because the container is disposable.
|
||||
- name: Install dev requirements
|
||||
run: python3 -m pip install --break-system-packages -r requirements-dev.txt
|
||||
|
||||
@@ -79,10 +60,6 @@ jobs:
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.unit
|
||||
run: python3 -m coverage report -m
|
||||
|
||||
# upload-artifact@v3's glob skips dotfiles, so a bare `.coverage.unit`
|
||||
# silently uploads nothing ("No files were found"). Stage it under a
|
||||
# non-dot name; the coverage job renames it back before `coverage
|
||||
# combine`. `cp` also fails loudly if coverage never wrote the file.
|
||||
- name: Stage unit coverage for upload
|
||||
run: cp .coverage.unit coverage-unit.dat
|
||||
|
||||
@@ -98,17 +75,9 @@ jobs:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
# No actions/setup-python (see the note in the `unit` job); the
|
||||
# container's system Python 3.12 runs the stdlib test suite directly.
|
||||
- name: Install coverage
|
||||
run: python3 -m pip install --break-system-packages coverage
|
||||
|
||||
# Fail loudly if the backend this job promises isn't actually usable,
|
||||
# rather than letting every test silently `unittest.skip` and the job
|
||||
# go green on zero coverage. `backend status` prints a clear per-check
|
||||
# summary (docker on PATH, daemon reachable) and exits non-zero when a
|
||||
# prerequisite is missing — the same readiness check the skip guards
|
||||
# gate on via `has_backend`.
|
||||
- name: Preflight — Docker backend is ready
|
||||
run: |
|
||||
python3 --version
|
||||
@@ -120,7 +89,6 @@ jobs:
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.docker
|
||||
run: python3 -m coverage run -m unittest discover -t . -s tests/integration -v
|
||||
|
||||
# Non-dot name so upload-artifact's dotfile-skipping glob picks it up.
|
||||
- name: Stage docker coverage for upload
|
||||
run: cp .coverage.docker coverage-docker.dat
|
||||
|
||||
@@ -130,107 +98,10 @@ jobs:
|
||||
name: coverage-docker
|
||||
path: coverage-docker.dat
|
||||
|
||||
# Integration tests against the Firecracker backend. Runs on a self-hosted
|
||||
# KVM runner (label `kvm`) where /dev/kvm and the TAP/nft pool are available.
|
||||
#
|
||||
# Restricted to same-repo PRs, push to main, and workflow_dispatch — fork
|
||||
# PRs don't execute untrusted code on the privileged runner.
|
||||
#
|
||||
# Runner prerequisites (provision once; see README "Firecracker on Linux"):
|
||||
# `firecracker` on PATH, `/dev/kvm` accessible, cached kernel +
|
||||
# static dropbear at /var/cache/bot-bottle-fc/dropbear, and the pool as a
|
||||
# persistent systemd unit.
|
||||
#
|
||||
# The infra candidate is built here directly (no artifact download) to
|
||||
# eliminate the ~70 s ubuntu-latest upload + ~83 s combined download that
|
||||
# the old build-infra → integration-firecracker + coverage chain incurred.
|
||||
# For main-branch pushes the tested rootfs and matching dropbear are
|
||||
# uploaded so publish-infra can publish the byte-identical artifact; PRs
|
||||
# skip those uploads entirely.
|
||||
integration-firecracker:
|
||||
runs-on: [self-hosted, kvm]
|
||||
if: >-
|
||||
github.event_name == 'push' ||
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'pull_request' &&
|
||||
github.event.pull_request.head.repo.full_name == github.repository)
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Preflight — Firecracker host is ready
|
||||
run: |
|
||||
command -v firecracker >/dev/null || {
|
||||
echo "firecracker not on PATH — provision the runner (README: Firecracker on Linux)"; exit 1; }
|
||||
test -e /dev/kvm || { echo "/dev/kvm missing — KVM not available on this runner"; exit 1; }
|
||||
# `backend status` exits non-zero unless the TAP pool is up + no
|
||||
# range overlap; it prints the exact `backend setup` fix.
|
||||
python3 cli.py backend status --backend=firecracker
|
||||
|
||||
- name: Build infra candidate from this checkout
|
||||
env:
|
||||
BOT_BOTTLE_FC_DROPBEAR: /var/cache/bot-bottle-fc/dropbear
|
||||
run: python3 -m bot_bottle.backend.firecracker.publish_infra --output infra-candidate --reuse-published
|
||||
|
||||
- name: Replace the persistent infra VM with the candidate
|
||||
run: python3 -c 'from bot_bottle.backend.firecracker import infra_vm; infra_vm.stop()'
|
||||
|
||||
# No dev-requirements install: `coverage` is already provided by the
|
||||
# self-hosted runner's Nix python env, and that env has no `pip`
|
||||
# module to install into anyway.
|
||||
- name: Run integration tests (firecracker) with coverage
|
||||
env:
|
||||
BOT_BOTTLE_BACKEND: firecracker
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_DIR: ${{ github.workspace }}/infra-candidate
|
||||
COVERAGE_FILE: ${{ github.workspace }}/.coverage.firecracker
|
||||
run: python3 -m coverage run -m unittest discover -t . -s tests/integration -v
|
||||
|
||||
# Non-dot name so upload-artifact's dotfile-skipping glob picks it up.
|
||||
- name: Stage firecracker coverage for upload
|
||||
run: cp .coverage.firecracker coverage-firecracker.dat
|
||||
|
||||
- name: Upload firecracker coverage artifact
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: coverage-firecracker
|
||||
path: coverage-firecracker.dat
|
||||
|
||||
# Only upload the large rootfs artifact on main-branch pushes;
|
||||
# PRs avoid the ~194 MB transfer. publish-infra only runs on main
|
||||
# and downloads these to publish the byte-identical tested rootfs.
|
||||
- name: Upload tested rootfs (main branch only)
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate/
|
||||
|
||||
- name: Upload dropbear for publish verification (main branch only)
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: /var/cache/bot-bottle-fc/dropbear
|
||||
|
||||
# Combined coverage gate: aggregates .coverage.* artifacts uploaded by each
|
||||
# test job, then runs the diff-coverage gate (new/changed lines >= 90%).
|
||||
#
|
||||
# Runs on ubuntu-latest — no KVM needed, no test reruns. Coverage files use
|
||||
# relative_files = True (.coveragerc) so they combine cleanly across runners.
|
||||
# Each test job sets COVERAGE_FILE to an absolute path so coverage.py writes
|
||||
# to a known location that upload-artifact can find regardless of runner env.
|
||||
#
|
||||
# Restricted to the same events as integration-firecracker: it depends on
|
||||
# that job's coverage artifact and skips for fork PRs alongside it.
|
||||
coverage:
|
||||
needs: [unit, integration-docker, integration-firecracker]
|
||||
needs: [unit, integration-docker]
|
||||
timeout-minutes: 15
|
||||
runs-on: ubuntu-latest
|
||||
if: >-
|
||||
github.event_name == 'push' ||
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
(github.event_name == 'pull_request' &&
|
||||
github.event.pull_request.head.repo.full_name == github.repository)
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
@@ -252,55 +123,15 @@ jobs:
|
||||
name: coverage-docker
|
||||
path: ${{ github.workspace }}
|
||||
|
||||
- name: Download firecracker coverage artifact
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: coverage-firecracker
|
||||
path: ${{ github.workspace }}
|
||||
|
||||
# Rename the non-dot upload names back to the .coverage.* files that
|
||||
# `coverage combine` discovers (see the staging steps in each test job).
|
||||
- name: Reassemble coverage data files
|
||||
run: |
|
||||
mv coverage-unit.dat .coverage.unit
|
||||
mv coverage-docker.dat .coverage.docker
|
||||
mv coverage-firecracker.dat .coverage.firecracker
|
||||
|
||||
- name: Combined coverage (unit + integration, incl. firecracker)
|
||||
- name: Combined coverage (unit + docker integration)
|
||||
run: PYTHON=python3 bash scripts/coverage.sh aggregate critical
|
||||
|
||||
- name: Diff-coverage gate (changed lines >= 90%)
|
||||
run: |
|
||||
git fetch --no-tags origin main:refs/remotes/origin/main
|
||||
python3 scripts/diff_coverage.py --base origin/main --min 90
|
||||
|
||||
publish-infra:
|
||||
needs: [unit, integration-docker, integration-firecracker, coverage]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
steps:
|
||||
- name: Checkout the tested revision
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Download the tested rootfs
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: infra-candidate
|
||||
path: infra-candidate
|
||||
|
||||
# publish_infra re-derives the version from the checkout to confirm the
|
||||
# bundle matches before uploading, and the version hashes the dropbear
|
||||
# bytes. Download the SAME dropbear integration-firecracker used, or
|
||||
# the recheck computes a "<missing>"-dropbear version and rejects the
|
||||
# candidate.
|
||||
- name: Download the staged dropbear (matches build's version)
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
name: firecracker-inputs
|
||||
path: firecracker-inputs
|
||||
|
||||
- name: Publish the tested candidate
|
||||
env:
|
||||
BOT_BOTTLE_INFRA_ARTIFACT_TOKEN: ${{ secrets.BOT_BOTTLE_INFRA_ARTIFACT_TOKEN }}
|
||||
BOT_BOTTLE_FC_DROPBEAR: ${{ github.workspace }}/firecracker-inputs/dropbear
|
||||
run: python3 -m bot_bottle.backend.firecracker.publish_infra --publish-dir infra-candidate
|
||||
|
||||
@@ -44,10 +44,10 @@ backend remains available with `BOT_BOTTLE_BACKEND=docker` or
|
||||
|
||||
- Three kinds of doc, each with its own conventions in-folder; see
|
||||
`docs/README.md` for when to write which:
|
||||
- **PRDs** (`docs/prds/`) — one feature per file. While a PR is open
|
||||
the file is named `prd-new-<kebab>.md`; CI assigns a sequential
|
||||
number on merge to `main` and renames it. A `Status:` line tracks
|
||||
lifecycle: Draft → Active (shipped to `main`) →
|
||||
- **PRDs** (`docs/prds/`) — one feature per file. A draft may initially
|
||||
use `prd-new-<kebab>.md`, but its author must assign the next
|
||||
sequential number before merge; CI rejects unnumbered PRDs. A
|
||||
`Status:` line tracks lifecycle: Draft → Active (shipped to `main`) →
|
||||
Superseded/Retargeted. Format in `docs/prds/README.md`.
|
||||
- **Research notes** (`docs/research/`) — opinionated investigations;
|
||||
unnumbered kebab-case, freeform and verdict-first. See
|
||||
|
||||
@@ -30,7 +30,7 @@
|
||||
|
||||
## Architecture
|
||||
|
||||
On the default macOS Apple Container backend, a bottle is an agent container on a host-only internal network plus a gateway attached to both that internal network and a NAT egress network. The agent gets HTTP(S)_PROXY and CA bundle env vars pointing at the gateway's internal-network IP, so HTTP/HTTPS traffic flows through the gateway instead of direct egress. `bottle.git` / git-gate is intentionally deferred on this backend until a safe Apple Container key-delivery path exists.
|
||||
On the default macOS Apple Container backend, a bottle is an agent container on a host-only internal network plus a gateway attached to both that internal network and a NAT egress network. The agent gets HTTP(S)_PROXY and CA bundle env vars pointing at the gateway's internal-network IP, so HTTP/HTTPS traffic flows through the gateway instead of direct egress. git-gate runs over the gateway's consolidated `git-http` daemon (the legacy per-bottle `git://` daemon is not used on this backend); keys are provisioned dynamically at launch and revoked on teardown.
|
||||
|
||||
On the Firecracker backend, a bottle is an agent microVM plus a Docker gateway for egress, git-gate, and supervise. The VM reaches the gateway over a per-bottle point-to-point TAP link; a dedicated fail-closed `nftables` table (`inet bot_bottle_fc`) confines the guest to that link, so nothing leaves the box except through the gateway. The TAP pool and nft table are provisioned once (root); per-launch needs no privilege.
|
||||
|
||||
@@ -75,6 +75,8 @@ On compatible macOS hosts, the default backend requires Apple's `container` CLI
|
||||
|
||||
Use `BOT_BOTTLE_BACKEND=docker ./cli.py start <agent>` on hosts where neither Apple Container nor KVM is available and Docker is the desired backend.
|
||||
|
||||
> **CI (macOS Apple Container):** the `integration-macos` job (`.gitea/workflows/test.yml`) runs the integration suite against `BOT_BOTTLE_BACKEND=macos-container` on a self-hosted macOS runner labelled `macos`, because Apple Container needs the host virtualization framework and cannot run in a Linux container (so it can't reuse the `kvm` runner). Provision an Apple Silicon host with the `container` CLI on `PATH` and `container system status` running, then register the runner in **host mode** (not docker mode) with the `macos` label — `brew install gitea-runner` (the `act_runner` rename). Give it a Python ≥ 3.11 with `coverage` importable on the launchd service's `PATH` (a launchd service doesn't inherit your shell profile, so pin `node` and the Python env explicitly). The job is **advisory** — `workflow_dispatch` (manual) only, never triggered by push or PR — since a single laptop that sleeps/roams must not block merges or churn on every push to main; its coverage doesn't feed the gate. The infra container is a singleton (`bot-bottle-mac-infra`), so keep runner concurrency at 1.
|
||||
|
||||
### Containers inside a bottle
|
||||
|
||||
A bottle may set `nested_containers: true`. On the macOS backend this starts a
|
||||
|
||||
@@ -20,7 +20,6 @@ from .util import run_docker
|
||||
from ...paths import (
|
||||
ORCHESTRATOR_TOKEN_ENV,
|
||||
bot_bottle_root,
|
||||
host_orchestrator_token,
|
||||
)
|
||||
from ...gateway import GatewayError
|
||||
from ...orchestrator.lifecycle import (
|
||||
@@ -172,7 +171,9 @@ class DockerOrchestrator(Orchestrator):
|
||||
fixed-name container first)."""
|
||||
self._ensure_control_network()
|
||||
run_docker(["docker", "rm", "--force", self.name])
|
||||
_signing_key = host_orchestrator_token()
|
||||
# The signing key comes through the shared provisioning contract (#476),
|
||||
# which fail-closes rather than yield an empty key that would run OPEN.
|
||||
_signing_key = self.control_plane_key()
|
||||
proc = run_docker([
|
||||
"docker", "run", "--detach",
|
||||
"--name", self.name,
|
||||
|
||||
@@ -21,7 +21,6 @@ import time
|
||||
from pathlib import Path
|
||||
|
||||
from ...log import die, info
|
||||
from ...paths import host_orchestrator_token
|
||||
from ...orchestrator.lifecycle import (
|
||||
DEFAULT_STARTUP_TIMEOUT_SECONDS,
|
||||
Orchestrator,
|
||||
@@ -108,11 +107,13 @@ class FirecrackerOrchestrator(Orchestrator):
|
||||
data_drive=self._ensure_registry_volume(),
|
||||
)
|
||||
# Push the host-canonical signing key (the init waits for it before
|
||||
# starting the control plane). The host token file stays the single
|
||||
# source of truth, so a co-running docker/macOS control plane keeps
|
||||
# working; the guest verifies tokens with the same key the CLI signs from.
|
||||
# starting the control plane). It comes through the shared provisioning
|
||||
# contract (#476) — the same host token file every backend uses, so a
|
||||
# co-running docker/macOS control plane keeps working and the guest
|
||||
# verifies tokens with the same key the CLI signs from; fail-closed, so
|
||||
# the guest is never handed an empty key that would run it OPEN.
|
||||
infra_vm.push_secret(
|
||||
vm, host_orchestrator_token(), infra_vm._GUEST_SIGNING_KEY_PATH,
|
||||
vm, self.control_plane_key(), infra_vm._GUEST_SIGNING_KEY_PATH,
|
||||
"the control-plane signing key to the orchestrator VM "
|
||||
"(its control plane will not start)",
|
||||
)
|
||||
|
||||
@@ -19,10 +19,7 @@ from pathlib import Path
|
||||
|
||||
from ... import log
|
||||
from ... import resources
|
||||
from ...paths import (
|
||||
ORCHESTRATOR_TOKEN_ENV,
|
||||
host_orchestrator_token,
|
||||
)
|
||||
from ...paths import ORCHESTRATOR_TOKEN_ENV
|
||||
from ...orchestrator.lifecycle import (
|
||||
DEFAULT_HEALTH_TIMEOUT_SECONDS,
|
||||
DEFAULT_PORT,
|
||||
@@ -136,7 +133,9 @@ class MacosOrchestrator(Orchestrator):
|
||||
|
||||
def _run_container(self, current_hash: str) -> None:
|
||||
container_mod.force_remove_container(self.name)
|
||||
_signing_key = host_orchestrator_token()
|
||||
# The signing key comes through the shared provisioning contract (#476),
|
||||
# which fail-closes rather than yield an empty key that would run OPEN.
|
||||
_signing_key = self.control_plane_key()
|
||||
argv = [
|
||||
"container", "run", "--detach",
|
||||
"--name", self.name,
|
||||
|
||||
@@ -2,7 +2,8 @@
|
||||
bot-bottle and report what's ready.
|
||||
|
||||
Fails (non-zero exit) only on the two hard requirements: a new-enough
|
||||
Python and at least one usable backend. The config directory is a soft
|
||||
Python and at least one backend that is *ready* (passes its full status
|
||||
checks, so `start` can actually work). The config directory is a soft
|
||||
check — `install.sh` creates it, but a missing one only warrants a note,
|
||||
not a failure, since `start` provisions what it needs on first run.
|
||||
"""
|
||||
@@ -13,7 +14,7 @@ import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from ...backend import is_backend_available, known_backend_names
|
||||
from ...backend import is_backend_ready, known_backend_names
|
||||
from ..constants import PROG
|
||||
|
||||
MIN_PYTHON = (3, 11)
|
||||
@@ -43,19 +44,25 @@ def _check_python() -> bool:
|
||||
|
||||
|
||||
def _check_backends() -> bool:
|
||||
"""At least one backend must be available on this host. Report each
|
||||
known backend so the operator sees why a missing one is missing."""
|
||||
available = []
|
||||
"""At least one backend must be *ready* to run a bottle — i.e. pass its
|
||||
full status() checks (daemon reachable, network pool present, KVM usable),
|
||||
not merely have a binary on PATH. A binary-only check would report `ok`
|
||||
on a host with a stopped Docker daemon or a half-configured Firecracker,
|
||||
where `start` still can't work. Each not-ready backend prints its own
|
||||
diagnostics (quiet=False) so the operator sees exactly what's missing."""
|
||||
ready = []
|
||||
for name in known_backend_names():
|
||||
if is_backend_available(name):
|
||||
available.append(name)
|
||||
if available:
|
||||
_ok("backend", f"available: {', '.join(available)}")
|
||||
if is_backend_ready(name, quiet=False):
|
||||
_ok("backend", f"{name}: ready")
|
||||
ready.append(name)
|
||||
else:
|
||||
_warn("backend", f"{name}: not ready (see diagnostics above)")
|
||||
if ready:
|
||||
return True
|
||||
_fail(
|
||||
"backend",
|
||||
"no backend available; install Apple Container (macOS), "
|
||||
"Firecracker (Linux), or Docker",
|
||||
"no backend is ready to run a bottle; start Docker, or finish "
|
||||
"Apple Container (macOS) / Firecracker (Linux) setup",
|
||||
)
|
||||
return False
|
||||
|
||||
|
||||
@@ -17,7 +17,7 @@ Each queued proposal tool call:
|
||||
4. On a decision within the window, returns the operator's
|
||||
`{status, notes}`. On timeout, returns `status: pending` **with the
|
||||
proposal id** and leaves the proposal queued — the flow is
|
||||
non-blocking past the grace window (PRD prd-new / issue #412).
|
||||
non-blocking past the grace window (PRD 0072 / issue #412).
|
||||
|
||||
`check-proposal` is the non-blocking companion: given a `proposal_id`
|
||||
returned by a `pending` response, it reports the current decision
|
||||
|
||||
@@ -18,8 +18,8 @@ import urllib.request
|
||||
from collections.abc import Iterable
|
||||
from dataclasses import dataclass
|
||||
|
||||
from ..orchestrator_auth import ROLE_CLI, mint
|
||||
from ..paths import host_orchestrator_token
|
||||
from ..orchestrator_auth import ROLE_CLI
|
||||
from ..trust_domain import CONTROL_PLANE
|
||||
from .server import ORCHESTRATOR_AUTH_HEADER
|
||||
|
||||
DEFAULT_TIMEOUT_SECONDS = 5.0
|
||||
@@ -32,7 +32,7 @@ def _host_auth_token() -> str:
|
||||
"" means 'send no auth header' — correct against an open (unconfigured)
|
||||
control plane, and harmlessly rejected by a secured one."""
|
||||
try:
|
||||
return mint(ROLE_CLI, host_orchestrator_token())
|
||||
return CONTROL_PLANE.mint(ROLE_CLI)
|
||||
except (OSError, ValueError):
|
||||
return ""
|
||||
|
||||
|
||||
@@ -21,8 +21,7 @@ import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
from ..orchestrator_auth import ROLE_GATEWAY, mint
|
||||
from ..paths import host_orchestrator_token
|
||||
from ..trust_domain import ControlPlaneProvisioning
|
||||
|
||||
DEFAULT_PORT = 8099
|
||||
DEFAULT_STARTUP_TIMEOUT_SECONDS = 45.0
|
||||
@@ -58,6 +57,12 @@ class Orchestrator(abc.ABC):
|
||||
so it — not the gateway — mints the gateway's role-scoped token.
|
||||
Backend-neutral."""
|
||||
|
||||
# The shared control-plane auth provisioning contract (#476). Every backend
|
||||
# gets its signing key + gateway token through this one seam rather than
|
||||
# re-deriving the wiring; it is fail-closed for every backend — the
|
||||
# orchestrator never starts without its signing key.
|
||||
provisioning: ControlPlaneProvisioning = ControlPlaneProvisioning()
|
||||
|
||||
def ensure_built(self) -> None:
|
||||
"""Ensure the orchestrator's image / rootfs exists, building it if
|
||||
needed. Default: nothing to build (e.g. a pre-pulled image). Call before
|
||||
@@ -105,9 +110,17 @@ class Orchestrator(abc.ABC):
|
||||
def mint_gateway_token(self) -> str:
|
||||
"""Mint a role-scoped `gateway` JWT from the host signing key for the
|
||||
gateway to present. The orchestrator holds the key; the gateway never
|
||||
does (#469). Backend-neutral — the same host token file is the single
|
||||
source of truth across backends."""
|
||||
return mint(ROLE_GATEWAY, host_orchestrator_token())
|
||||
does (#469). Routed through the shared provisioning contract (#476), so
|
||||
the same host token file is the single source of truth across backends."""
|
||||
return self.provisioning.gateway_token()
|
||||
|
||||
def control_plane_key(self) -> str:
|
||||
"""The raw signing key the control-plane *process* must receive — the ONE
|
||||
place a backend obtains it (docker/macOS inject it as `key_env`;
|
||||
firecracker pushes it to the guest). Fail-closed via the provisioning
|
||||
contract: it raises rather than yield an empty key that would run the
|
||||
server OPEN (#476)."""
|
||||
return self.provisioning.orchestrator_key()
|
||||
|
||||
|
||||
__all__ = [
|
||||
@@ -117,4 +130,5 @@ __all__ = [
|
||||
"OrchestratorStartError",
|
||||
"source_hash",
|
||||
"Orchestrator",
|
||||
"ControlPlaneProvisioning",
|
||||
]
|
||||
|
||||
@@ -63,8 +63,8 @@ import sys
|
||||
import typing
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from ..orchestrator_auth import ROLE_CLI, ROLES, verify
|
||||
from ..paths import ORCHESTRATOR_TOKEN_ENV
|
||||
from ..orchestrator_auth import ROLE_CLI, ROLES
|
||||
from ..trust_domain import CONTROL_PLANE
|
||||
from ..supervisor.types import TOOLS
|
||||
from .service import OrchestratorCore
|
||||
|
||||
@@ -413,11 +413,13 @@ class OrchestratorServer(socketserver.ThreadingMixIn, http.server.HTTPServer):
|
||||
|
||||
def __init__(self, address: tuple[str, int], orchestrator: OrchestratorCore) -> None:
|
||||
self.orchestrator = orchestrator
|
||||
self._signing_key = os.environ.get(ORCHESTRATOR_TOKEN_ENV, "").strip()
|
||||
# The control-plane trust domain's signing key, as injected into THIS
|
||||
# (the owning) process by the launcher (#476). Unset → open mode below.
|
||||
self._signing_key = CONTROL_PLANE.key_from_env()
|
||||
if not self._signing_key:
|
||||
sys.stderr.write(
|
||||
"orchestrator: WARNING — no control-plane signing key "
|
||||
f"(${ORCHESTRATOR_TOKEN_ENV}); running WITHOUT caller "
|
||||
f"(${CONTROL_PLANE.key_env}); running WITHOUT caller "
|
||||
"authentication. Any client that can reach this port can drive "
|
||||
"it. Backends that put the control plane on an agent-reachable "
|
||||
"network MUST set this.\n"
|
||||
@@ -433,7 +435,7 @@ class OrchestratorServer(socketserver.ThreadingMixIn, http.server.HTTPServer):
|
||||
role (→ per-route 401/403 in `dispatch`)."""
|
||||
if not self._signing_key:
|
||||
return ROLE_CLI
|
||||
return verify(presented, self._signing_key)
|
||||
return CONTROL_PLANE.verify(presented, self._signing_key)
|
||||
|
||||
|
||||
def make_server(
|
||||
|
||||
@@ -113,7 +113,7 @@ _MIGRATIONS = TableMigrations(
|
||||
# egress allowlist / routes / git config selected by source IP. The
|
||||
# multi-tenant gateway resolves it per request via `attribute`.
|
||||
"ALTER TABLE orchestrator_bottles ADD COLUMN policy TEXT NOT NULL DEFAULT ''",
|
||||
# v4 — per-bottle encrypted egress secrets (PRD prd-new-secret-provider).
|
||||
# v4 — per-bottle encrypted egress secrets (PRD 0080).
|
||||
# One row per env-var: key (env-var name) is plaintext for auditing;
|
||||
# value is the encrypted token string. The encryption key (ENV_VAR_SECRET)
|
||||
# lives only in the agent's environment — a row alone cannot recover the
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
"""Symmetric encryption for per-bottle egress secrets (PRD prd-new-secret-provider).
|
||||
"""Symmetric encryption for per-bottle egress secrets (PRD 0080).
|
||||
|
||||
Each agent receives a random ENV_VAR_SECRET at startup — passed as an env var,
|
||||
never logged or persisted. The host uses this key to encrypt each egress auth
|
||||
|
||||
@@ -59,12 +59,17 @@ _HEADER_SEGMENT = _b64url_encode(
|
||||
)
|
||||
|
||||
|
||||
def mint(role: str, secret: str) -> str:
|
||||
def mint(role: str, secret: str, *, roles: frozenset[str] = ROLES) -> str:
|
||||
"""A compact HS256 token asserting `role`, signed with `secret`.
|
||||
|
||||
Raises ValueError for an unknown role (mint only what the control plane will
|
||||
accept) or an empty signing key (an unsigned credential is never valid)."""
|
||||
if role not in ROLES:
|
||||
`roles` is the set the signing key is allowed to sign (default: the
|
||||
orchestrator's `{gateway, cli}`). A separate service (e.g. the host
|
||||
controller) passes its own key + role set so its tokens can't be forged with
|
||||
the orchestrator's key — see `trust_domain.py`, issues #476/#468.
|
||||
|
||||
Raises ValueError for a role outside `roles`, or an empty signing key (an
|
||||
unsigned credential is never valid)."""
|
||||
if role not in roles:
|
||||
raise ValueError(f"unknown control-plane role {role!r}")
|
||||
if not secret:
|
||||
raise ValueError("cannot mint a control-plane token without a signing key")
|
||||
@@ -73,10 +78,11 @@ def mint(role: str, secret: str) -> str:
|
||||
return f"{signing_input}.{_sign(secret, signing_input)}"
|
||||
|
||||
|
||||
def verify(token: str, secret: str) -> str | None:
|
||||
def verify(token: str, secret: str, *, roles: frozenset[str] = ROLES) -> str | None:
|
||||
"""The role a valid `token` carries, or None if it is malformed, wrongly
|
||||
signed, or names an unknown role. Constant-time signature check; rejects any
|
||||
header whose alg isn't HS256 (no alg-confusion / `none`)."""
|
||||
signed, or names a role outside `roles` (the verifying trust domain's set —
|
||||
default `{gateway, cli}`). Constant-time signature check; rejects any header
|
||||
whose alg isn't HS256 (no alg-confusion / `none`)."""
|
||||
if not token or not secret:
|
||||
return None
|
||||
parts = token.split(".")
|
||||
@@ -94,7 +100,7 @@ def verify(token: str, secret: str) -> str | None:
|
||||
if not isinstance(header, dict) or header.get("alg") != _ALG:
|
||||
return None
|
||||
role = payload.get("role") if isinstance(payload, dict) else None
|
||||
return role if isinstance(role, str) and role in ROLES else None
|
||||
return role if isinstance(role, str) and role in roles else None
|
||||
|
||||
|
||||
__all__ = ["ROLE_GATEWAY", "ROLE_CLI", "ROLES", "mint", "verify"]
|
||||
|
||||
+19
-9
@@ -97,16 +97,17 @@ def host_gateway_ca_dir() -> Path:
|
||||
return ca_dir
|
||||
|
||||
|
||||
def host_orchestrator_token() -> str:
|
||||
"""The per-host control-plane secret, minted (256-bit, url-safe) and
|
||||
persisted 0600 on first use, then reused.
|
||||
def host_signing_key(filename: str) -> str:
|
||||
"""A per-host signing key at `<root>/<filename>`, minted (256-bit, url-safe)
|
||||
and persisted 0600 on first use, then reused.
|
||||
|
||||
This is the shared secret the launchers inject into the control-plane and
|
||||
gateway containers and that the host CLI presents on every call. It is a
|
||||
*host* artifact — the file lives under the root the agent never mounts, and
|
||||
the env var is set only on the trusted containers — so reading it here is
|
||||
safe on the host launch path but the value never reaches a bottle."""
|
||||
path = bot_bottle_root() / ORCHESTRATOR_TOKEN_FILENAME
|
||||
The generic form of `host_orchestrator_token()`: each service names its own
|
||||
key file (`trust_domain.py`), so the orchestrator and a separate service like
|
||||
the host controller (#468) get distinct keys neither can read. It is a *host*
|
||||
artifact — the file lives under the root the agent never mounts, and its value
|
||||
is injected only into the trusted control-plane process — so reading it here
|
||||
is safe on the launch path but the value never reaches a bottle."""
|
||||
path = bot_bottle_root() / filename
|
||||
try:
|
||||
existing = path.read_text().strip()
|
||||
if existing:
|
||||
@@ -128,6 +129,14 @@ def host_orchestrator_token() -> str:
|
||||
return token
|
||||
|
||||
|
||||
def host_orchestrator_token() -> str:
|
||||
"""The per-host control-plane signing key — the host-canonical key the
|
||||
launchers inject into the control-plane process and the host CLI mints its
|
||||
own `cli` token from. The `control-plane` trust domain's specialization of
|
||||
`host_signing_key()`."""
|
||||
return host_signing_key(ORCHESTRATOR_TOKEN_FILENAME)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"HOST_DB_FILENAME",
|
||||
"ORCHESTRATOR_TOKEN_FILENAME",
|
||||
@@ -138,5 +147,6 @@ __all__ = [
|
||||
"host_db_path",
|
||||
"host_db_dir",
|
||||
"host_gateway_ca_dir",
|
||||
"host_signing_key",
|
||||
"host_orchestrator_token",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
"""Per-service control-plane signing keys (issue #476).
|
||||
|
||||
A `TrustDomain` is one service's signing material: its host-canonical key file,
|
||||
the roles that key may sign, and the env vars its key and a pre-minted token ride
|
||||
in. Scoping `mint`/`verify` to a domain's roles keeps one service's key from
|
||||
signing (or accepting) another service's tokens.
|
||||
|
||||
Today there is one domain, `CONTROL_PLANE` — the orchestrator's key (roles
|
||||
`{gateway, cli}`): the orchestrator holds it and mints the gateway's and CLI's
|
||||
tokens. The host controller (#468) will add a **second** domain with its own key
|
||||
the orchestrator never holds. That is the point: the host controller starts and
|
||||
stops the orchestrator, so the orchestrator must not be able to mint the
|
||||
credentials it uses to talk to it. Adding a `host` role to `CONTROL_PLANE`
|
||||
instead would defeat that — the orchestrator holds that key, so it could forge
|
||||
`host` tokens.
|
||||
|
||||
`ControlPlaneProvisioning` is the one seam every backend launcher uses to get the
|
||||
orchestrator its key and the gateway its token, instead of re-deriving that
|
||||
wiring per backend (the bug class behind PR #471 — see
|
||||
`docs/prds/0079-control-plane-auth-provisioning.md`).
|
||||
|
||||
Stdlib-only: the HMAC lives in `orchestrator_auth`, the key file in `paths`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
|
||||
from . import orchestrator_auth
|
||||
from .orchestrator_auth import ROLE_GATEWAY
|
||||
from .paths import (
|
||||
ORCHESTRATOR_AUTH_JWT_ENV,
|
||||
ORCHESTRATOR_TOKEN_ENV,
|
||||
ORCHESTRATOR_TOKEN_FILENAME,
|
||||
host_signing_key,
|
||||
)
|
||||
|
||||
|
||||
class ProvisioningError(RuntimeError):
|
||||
"""A control-plane auth invariant would be violated (e.g. starting the
|
||||
orchestrator without its signing key — which would run OPEN)."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TrustDomain:
|
||||
"""One service's signing material: a host-canonical key file, the roles that
|
||||
key may sign, and the env vars its key and a minted token ride in.
|
||||
|
||||
The service that *owns* the domain (e.g. the orchestrator) receives the raw
|
||||
key via `key_env`; a delegate (e.g. the gateway) receives only a pre-minted,
|
||||
role-scoped token via `token_env` it cannot rewrite. `mint`/`verify` are
|
||||
scoped to `roles`, so this service's key can neither sign nor accept another
|
||||
service's role."""
|
||||
|
||||
name: str
|
||||
key_filename: str
|
||||
roles: frozenset[str]
|
||||
key_env: str
|
||||
token_env: str
|
||||
|
||||
def signing_key(self) -> str:
|
||||
"""This service's host-canonical signing key (minted 0600 on first use).
|
||||
Host-side only — the value is injected into the owning process."""
|
||||
return host_signing_key(self.key_filename)
|
||||
|
||||
def key_from_env(self, environ: Mapping[str, str] | None = None) -> str:
|
||||
"""The signing key as the owning process sees it — read from `key_env`
|
||||
(default `os.environ`). "" when unset; the caller decides whether that is
|
||||
fatal (`ControlPlaneProvisioning`) or the open-mode fallback
|
||||
(`OrchestratorServer`)."""
|
||||
env = os.environ if environ is None else environ
|
||||
return env.get(self.key_env, "").strip()
|
||||
|
||||
def mint(self, role: str) -> str:
|
||||
"""A role-scoped token for a delegate, signed with this service's key.
|
||||
Raises ValueError for a role this service doesn't sign."""
|
||||
if role not in self.roles:
|
||||
raise ValueError(f"role {role!r} is not in trust domain {self.name!r}")
|
||||
return orchestrator_auth.mint(role, self.signing_key(), roles=self.roles)
|
||||
|
||||
def verify(self, token: str, key: str) -> str | None:
|
||||
"""The role `token` carries under `key`, or None. `key` is passed in
|
||||
(not read from disk) because the verifier — the control-plane process —
|
||||
holds it in `key_env`, not on disk in its guest."""
|
||||
return orchestrator_auth.verify(token, key, roles=self.roles)
|
||||
|
||||
|
||||
# The orchestrator's domain: the key the orchestrator (and host CLI) holds, the
|
||||
# `gateway` token it mints for the data plane, and the `cli` token the CLI mints
|
||||
# for itself. #468's host controller will add a second, separate domain.
|
||||
CONTROL_PLANE = TrustDomain(
|
||||
name="control-plane",
|
||||
key_filename=ORCHESTRATOR_TOKEN_FILENAME,
|
||||
roles=orchestrator_auth.ROLES,
|
||||
key_env=ORCHESTRATOR_TOKEN_ENV,
|
||||
token_env=ORCHESTRATOR_AUTH_JWT_ENV,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ControlPlaneProvisioning:
|
||||
"""The one seam every backend launcher uses to provision control-plane auth,
|
||||
instead of re-deriving the four invariants that each cost a PR #471 review
|
||||
round: the orchestrator gets the raw key (`orchestrator_key`), the gateway
|
||||
gets a minted `gateway` token (`gateway_token`), the host CLI mints its own
|
||||
`cli` token from the same host-canonical key, and the orchestrator never
|
||||
starts open."""
|
||||
|
||||
domain: TrustDomain = CONTROL_PLANE
|
||||
|
||||
def orchestrator_key(self) -> str:
|
||||
"""The raw signing key the orchestrator process must receive (carry it in
|
||||
`domain.key_env`). Fail-closed: raises rather than return "", since an
|
||||
empty key runs the server open — and being on a separate host does not
|
||||
stop the gateway from reaching the control plane (it must, for
|
||||
`/resolve`), so an open orchestrator would treat that gateway as `cli`."""
|
||||
key = self.domain.signing_key()
|
||||
if not key:
|
||||
raise ProvisioningError(
|
||||
f"refusing to start the {self.domain.name} orchestrator without "
|
||||
"a signing key: an open orchestrator authenticates no one and "
|
||||
"grants every caller that reaches it full `cli` (#476)"
|
||||
)
|
||||
return key
|
||||
|
||||
def gateway_token(self) -> str:
|
||||
"""The `gateway`-role token the gateway receives (carry it in
|
||||
`domain.token_env`) — minted from the key, never the key itself, so a
|
||||
compromised gateway cannot forge a `cli` token."""
|
||||
return self.domain.mint(ROLE_GATEWAY)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"ProvisioningError",
|
||||
"TrustDomain",
|
||||
"CONTROL_PLANE",
|
||||
"ControlPlaneProvisioning",
|
||||
]
|
||||
@@ -7,6 +7,7 @@ picking the right document for what you're capturing.
|
||||
|
||||
| Artifact | For |
|
||||
|---|---|
|
||||
| **Design workflow** (`docs/design-workflow.md`) | How discussion becomes canonical design, how dependencies are recorded, and when implementation may begin. |
|
||||
| **Glossary** (`docs/glossary.md`) | Canonical term definitions — what words mean in this project. |
|
||||
| **PRD** (`docs/prds/`) | A feature: what to build, scope, success criteria. |
|
||||
| **Research note** (`docs/research/`) | A landscape/tradeoff investigation. |
|
||||
|
||||
+12
-1
@@ -2,11 +2,22 @@
|
||||
|
||||
The test workflow lives at [`.gitea/workflows/test.yml`](../.gitea/workflows/test.yml).
|
||||
It runs the unit suite plus one integration job per backend
|
||||
(`integration-docker`, `integration-firecracker`) on:
|
||||
(`integration-docker`, `integration-firecracker`, `integration-macos`) on:
|
||||
|
||||
- every push to a branch with an open pull request, and
|
||||
- every push to `main`.
|
||||
|
||||
`integration-macos` is the exception: it is **advisory**, running only on
|
||||
`workflow_dispatch` (manual dispatch), never on push or pull requests. It targets the
|
||||
Apple Container backend on a self-hosted macOS runner (label `macos`,
|
||||
registered in host mode — Apple Container can't run in a Linux container, so it
|
||||
can't reuse the `kvm` runner). A single non-redundant laptop must not be able
|
||||
to block a PR merge, so the job stays out of the `coverage` job's `needs` and
|
||||
its coverage never feeds the diff-coverage gate. Because the infra container is
|
||||
a singleton (`bot-bottle-mac-infra`), the job declares a `concurrency` group
|
||||
and tears the container down on exit; keep runner concurrency at 1. See the
|
||||
README "macOS Apple Container" CI note for runner provisioning.
|
||||
|
||||
Each integration job selects its backend via `BOT_BOTTLE_BACKEND` and
|
||||
runs a **preflight** (`./cli.py backend status --backend=<name>`) that
|
||||
prints a clear per-check readiness summary and fails the job when the
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
# ADR 0005: Keep tracker metadata on issues
|
||||
# ADR 0005: Keep tracker metadata on one tracker object
|
||||
|
||||
- **Status:** Accepted
|
||||
- **Date:** 2026-07-18
|
||||
- **Deciders:** didericis
|
||||
|
||||
> **Amended 2026-07-26.** A pull request may carry labels directly instead of
|
||||
> linking a tracking issue. When a PR does link an issue, the issue remains the
|
||||
> canonical owner of planning metadata and the reference is validated.
|
||||
|
||||
## Context
|
||||
|
||||
Gitea exposes labels on both issues and pull requests. Applying the same labels
|
||||
@@ -20,19 +24,29 @@ would make the issue history less truthful.
|
||||
|
||||
## Decision
|
||||
|
||||
Issues are the canonical tracker records and own labels. Every issue has at
|
||||
least one label. An issue opened or left without labels receives
|
||||
`Status/Needs Triage` automatically until it is classified.
|
||||
Issues are the canonical tracker records and own labels when a separate work
|
||||
item exists. Every issue has at least one label. An issue opened or left
|
||||
without labels receives `Status/Needs Triage` automatically until it is
|
||||
classified.
|
||||
|
||||
Pull requests carry no labels. Every new PR deliberately references at least
|
||||
one existing issue in its title or description with one of these forms:
|
||||
Every new pull request is tracked in exactly one of two mutually exclusive
|
||||
ways:
|
||||
|
||||
1. It deliberately references at least one existing issue in its title or
|
||||
description. Tracker metadata stays on that issue and the PR remains
|
||||
unlabelled.
|
||||
2. It carries at least one label directly when a separate issue would add no
|
||||
useful planning context.
|
||||
|
||||
Issue references use one of these forms:
|
||||
|
||||
- `Closes #123`, `Fixes #123`, or `Resolves #123` when merging completes it.
|
||||
- `Part of #123`, `Related to #123`, `Refs #123`, or `References #123` when it
|
||||
contributes without completing it.
|
||||
|
||||
Gitea Actions enforces both PR rules as a status check and repairs the empty
|
||||
issue-label state. Branch protection makes the PR policy check required.
|
||||
Gitea Actions enforces the exclusive either/or PR rule, validates any issue
|
||||
references, and repairs the empty issue-label state. Branch protection makes
|
||||
the PR policy check required.
|
||||
|
||||
The policy applies from 2026-07-18 onward. Existing issues may be labelled as
|
||||
they are encountered, but closed PRs are grandfathered: no retrospective
|
||||
@@ -40,9 +54,13 @@ issues or PR labels are created solely to make history conform.
|
||||
|
||||
## Consequences
|
||||
|
||||
- Classification, priority, and workflow metadata have one source of truth.
|
||||
- A PR's issue link is the navigation path to its planning metadata.
|
||||
- Classification, priority, and workflow metadata have one source of truth for
|
||||
each change: the linked issue when one exists, otherwise the PR.
|
||||
- For issue-backed changes, the PR's issue link is the navigation path to its
|
||||
planning metadata.
|
||||
- Multi-PR issues do not require copied or synchronized labels.
|
||||
- Small standalone changes do not require a tracking issue created solely to
|
||||
satisfy automation.
|
||||
- `Status/Needs Triage` is an intentional fallback, not a final
|
||||
classification.
|
||||
- Direct issue creation remains convenient; automation repairs a missing label
|
||||
@@ -53,5 +71,6 @@ issues or PR labels are created solely to make history conform.
|
||||
## Links
|
||||
|
||||
- Issue #405.
|
||||
- `.gitea/workflows/tracker-policy.yml`.
|
||||
- `.gitea/workflows/tracker-policy-pr.yml`.
|
||||
- `.gitea/workflows/tracker-policy-issues.yml`.
|
||||
- `scripts/tracker_policy.py`.
|
||||
|
||||
@@ -0,0 +1,218 @@
|
||||
# Design workflow
|
||||
|
||||
How bot-bottle turns discussion into canonical design and then into
|
||||
implementation without leaving the repository's architecture scattered across
|
||||
issue and review threads.
|
||||
|
||||
The goal is not more documentation. The goal is one discoverable current answer
|
||||
for every load-bearing design question.
|
||||
|
||||
## Sources of truth
|
||||
|
||||
Design artifacts have different jobs:
|
||||
|
||||
| Artifact | Authority |
|
||||
|---|---|
|
||||
| Decision records | Stable system-wide boundaries, policies, and invariants |
|
||||
| PRDs | The current design for a feature |
|
||||
| Research notes | Evidence and tradeoff analysis; informative, not normative |
|
||||
| Issues | Work tracking, open questions, and discussion |
|
||||
| Pull-request comments | Review history; never the final home of a design decision |
|
||||
|
||||
When a discussion changes the design, update the relevant PRD or decision
|
||||
record before treating the discussion as resolved. A comment may explain why a
|
||||
decision changed, but future implementers must not need to reconstruct the
|
||||
decision from a thread.
|
||||
|
||||
Avoid duplicating the same rule in several canonical documents. Prefer one
|
||||
canonical statement and links from dependent documents.
|
||||
|
||||
## Choosing the canonical artifact
|
||||
|
||||
Use a PRD when the decision describes a feature: its behavior, scope, success
|
||||
criteria, trust model, implementation slices, and tests.
|
||||
|
||||
Use a decision record when the choice is broader than one feature or will
|
||||
constrain several future features. Examples include state ownership, credential
|
||||
boundaries, compatibility policy, and what the project does or does not claim
|
||||
as a security guarantee.
|
||||
|
||||
Use a research note when the conclusion depends on comparing external systems,
|
||||
protocols, or approaches. Promote any resulting project decision into a PRD or
|
||||
decision record.
|
||||
|
||||
## From discussion to implementation
|
||||
|
||||
### 1. Open the design discussion
|
||||
|
||||
An issue may start with incomplete requirements. Record:
|
||||
|
||||
- the problem and desired outcome;
|
||||
- known security or compatibility constraints;
|
||||
- the current owner of affected state and credentials;
|
||||
- related PRDs, decisions, issues, and pull requests;
|
||||
- open questions that would materially change the implementation.
|
||||
|
||||
Do not disguise an unresolved trust-boundary or state-ownership decision as an
|
||||
implementation detail.
|
||||
|
||||
### 2. Draft or update the canonical design
|
||||
|
||||
Before substantial implementation, write the feature PRD and update any
|
||||
system-wide decision it changes.
|
||||
|
||||
An active design should make these relationships visible near its top:
|
||||
|
||||
```markdown
|
||||
Status: Draft | Active | Superseded | Retargeted
|
||||
Depends on: #...
|
||||
Supersedes: ...
|
||||
```
|
||||
|
||||
Record dependencies only on the dependent document. Do not maintain reverse
|
||||
`Blocks` lists that can drift as dependent work changes.
|
||||
|
||||
For security-sensitive work, state:
|
||||
|
||||
- the exact guarantee and explicit non-guarantees;
|
||||
- trusted and untrusted components;
|
||||
- who creates each identity or attribution field;
|
||||
- who owns durable state;
|
||||
- failure and recovery behavior;
|
||||
- how the design is tested at its boundaries.
|
||||
|
||||
### 3. Resolve review into the repository
|
||||
|
||||
When review settles a design-changing question:
|
||||
|
||||
1. Update the canonical document in the same pull request.
|
||||
2. Mark conflicting documents Superseded or Retargeted, or update them.
|
||||
3. Add or adjust dependency links.
|
||||
4. Leave a concise resolution comment linking to the canonical change.
|
||||
|
||||
A useful resolution comment is:
|
||||
|
||||
```text
|
||||
Resolution: <what was decided>
|
||||
Canonicalized in: <document/section/commit>
|
||||
Supersedes: <older statement, if any>
|
||||
Follow-up: <remaining implementation or question>
|
||||
```
|
||||
|
||||
The resolution is incomplete until the repository reflects it.
|
||||
|
||||
### 4. Check design readiness
|
||||
|
||||
Implementation may begin when:
|
||||
|
||||
- the PRD's material trust, ownership, and compatibility questions are settled;
|
||||
- dependencies and blockers are explicit;
|
||||
- the design agrees with current architecture and decision records;
|
||||
- superseded documents are marked or updated;
|
||||
- success criteria and boundary tests are concrete;
|
||||
- remaining open questions can be answered during implementation without
|
||||
changing the feature's guarantee or component ownership.
|
||||
|
||||
Small exploratory spikes may happen earlier. A spike proves feasibility; it does
|
||||
not establish a production contract or silently settle the design.
|
||||
|
||||
### 5. Implement in ordered slices
|
||||
|
||||
Prefer small, independently reviewable slices after the parent design is
|
||||
accepted. Record the dependency chain explicitly.
|
||||
|
||||
Parallel work is safe when slices do not compete for the same unsettled
|
||||
interface or ownership boundary. If a foundational change will alter the
|
||||
transport, schema, state owner, or trust domain used by another slice, land the
|
||||
foundation first.
|
||||
|
||||
An implementation pull request should identify:
|
||||
|
||||
- the PRD or decision it implements;
|
||||
- the implementation chunk;
|
||||
- its base and blockers;
|
||||
- any design deviation discovered during implementation.
|
||||
|
||||
If implementation reveals a load-bearing design change, pause that slice and
|
||||
update the canonical design. Do not let the code and review thread become an
|
||||
undocumented replacement for the PRD.
|
||||
|
||||
## Dependency and staleness management
|
||||
|
||||
### Dependency direction
|
||||
|
||||
Write dependencies in terms of contracts, not chronology:
|
||||
|
||||
```text
|
||||
credential provisioning contract
|
||||
-> host-controller authentication
|
||||
-> privileged host operations
|
||||
```
|
||||
|
||||
If only part of a feature is blocked, say so. For example, a manifest parser may
|
||||
proceed while that feature's durable audit-storage chunk waits for the canonical
|
||||
audit schema.
|
||||
|
||||
### Superseding documents
|
||||
|
||||
Do not silently edit history to make an old design appear to have always said
|
||||
the new thing. Preserve the rationale, but make current status unmistakable:
|
||||
|
||||
```markdown
|
||||
Status: Superseded
|
||||
Superseded by: <document>
|
||||
Reason: <one paragraph>
|
||||
```
|
||||
|
||||
If part of a PRD remains valid, mark it Retargeted and identify which scope moved
|
||||
elsewhere.
|
||||
|
||||
Add a short supersession note near the top explaining what changed, why the old
|
||||
design is no longer current, and where the current design lives. For a research
|
||||
note whose original analysis remains useful, preserve that analysis and append
|
||||
a dated addendum with the newer finding instead of rewriting the note as though
|
||||
it had always reached the new conclusion.
|
||||
|
||||
### Architecture sweeps
|
||||
|
||||
After a foundational change, do a targeted architecture sweep before building
|
||||
more features on it:
|
||||
|
||||
1. Identify the concepts the change affects, such as `bot-bottle.db`, host
|
||||
controller, orchestrator, audit ownership, or signing key.
|
||||
2. Search active PRDs, decisions, and open issues for those concepts.
|
||||
3. Update or supersede contradictory statements.
|
||||
4. Refresh dependency links and the current architecture summary.
|
||||
5. Confirm stacked implementation branches still have the correct base.
|
||||
|
||||
This is a milestone activity, not a recurring documentation ceremony.
|
||||
|
||||
## Pull-request checklist
|
||||
|
||||
Use the relevant items in design and implementation pull requests:
|
||||
|
||||
- [ ] The canonical PRD or decision is linked.
|
||||
- [ ] Design-changing review decisions are reflected in-repo.
|
||||
- [ ] Dependencies and blockers are explicit.
|
||||
- [ ] State, credential, and trust ownership agree with current architecture.
|
||||
- [ ] Superseded or retargeted documents are marked.
|
||||
- [ ] Security guarantees and non-guarantees are precise.
|
||||
- [ ] Open questions do not change the promised guarantee or ownership model.
|
||||
- [ ] Implementation deviations updated the canonical design.
|
||||
|
||||
## Lightweight maintenance
|
||||
|
||||
Automation should enforce document shape, not pretend to understand
|
||||
architecture. Useful checks include:
|
||||
|
||||
- active PRDs contain status and dependency metadata;
|
||||
- superseded PRDs link to their replacement;
|
||||
- referenced documents and issues exist;
|
||||
- implementation pull requests identify their PRD and chunk;
|
||||
- document filenames and lifecycle states follow repository conventions.
|
||||
|
||||
Human review remains responsible for detecting conflicting guarantees or
|
||||
ownership claims.
|
||||
|
||||
The durable rule is simple: **discussion discovers the decision; the repository
|
||||
records it; implementation follows it.**
|
||||
@@ -1,9 +1,14 @@
|
||||
# PRD 0001: Per-agent egress proxy via pipelock
|
||||
|
||||
- **Status:** Active
|
||||
- **Status:** Superseded by [PRD 0017](0017-egress-proxy-via-mitmproxy.md)
|
||||
and [PRD 0052](0052-egress-dlp-addon.md)
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-05-08
|
||||
|
||||
> **Superseded.** Pipelock was removed in issue #193. PRD 0017 moved
|
||||
> egress enforcement and credential injection to mitmproxy; PRD 0052 moved DLP
|
||||
> enforcement into the egress addon. The design below is retained as history.
|
||||
|
||||
## Summary
|
||||
|
||||
Run pipelock as a sidecar container on each bot-bottle agent's only
|
||||
|
||||
@@ -1,9 +1,14 @@
|
||||
# PRD 0006: pipelock native TLS interception
|
||||
|
||||
- **Status:** Active
|
||||
- **Status:** Superseded by [PRD 0017](0017-egress-proxy-via-mitmproxy.md)
|
||||
and [PRD 0052](0052-egress-dlp-addon.md)
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-05-12
|
||||
|
||||
> **Superseded.** Pipelock was removed in issue #193. TLS interception now
|
||||
> belongs to the mitmproxy egress design in PRD 0017, with DLP implemented by
|
||||
> the egress addon in PRD 0052. The design below is retained as history.
|
||||
|
||||
## Summary
|
||||
|
||||
Turn on pipelock's built-in `tls_interception` so its DLP / URL /
|
||||
|
||||
@@ -1,11 +1,17 @@
|
||||
# PRD 0015: pipelock block remediation
|
||||
|
||||
- **Status:** Active
|
||||
- **Status:** Superseded by [PRD 0017](0017-egress-proxy-via-mitmproxy.md)
|
||||
and [PRD 0052](0052-egress-dlp-addon.md)
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-05-25
|
||||
- **Parent:** PRD 0012
|
||||
- **Depends on:** PRD 0013
|
||||
|
||||
> **Superseded.** Pipelock and its restart-based allowlist remediation path
|
||||
> were removed in issue #193. Current egress enforcement is the mitmproxy
|
||||
> design from PRD 0017 with DLP in PRD 0052. The design below is retained as
|
||||
> history.
|
||||
|
||||
## Summary
|
||||
|
||||
Wires the **pipelock block** path (PRD 0012 *Stuck categories*) end-to-end. The supervisor, on approval of a `pipelock-block` proposal, writes the new pipelock allowlist to the host and restarts pipelock; the agent's in-flight outbound calls may drop and rely on retry. The TUI gains a proactive `pipelock edit <bottle>` verb for operator-initiated edits unrelated to a tool call. The pipelock audit log (format defined in PRD 0013) is filled in with real entries on every edit.
|
||||
|
||||
@@ -4,6 +4,11 @@
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-05-26
|
||||
|
||||
> **Superseded.** PRD 0049 removed the active-agents pane and agent-scoped
|
||||
> operator edit verbs when the dashboard was narrowed back to a proposal-only
|
||||
> supervise TUI. A future agent-management surface was deferred rather than
|
||||
> carried forward from this design. The design below is retained as history.
|
||||
|
||||
## Summary
|
||||
|
||||
The dashboard today is proposal-centric: it lists every pending
|
||||
|
||||
@@ -4,6 +4,11 @@
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-05-26
|
||||
|
||||
> **Superseded.** PRD 0049 removed start, re-attach, and stop actions from the
|
||||
> dashboard when it became the proposal-only supervise TUI. Bottle lifecycle
|
||||
> remains in the dedicated CLI commands; no dashboard replacement from this
|
||||
> design remains active. The design below is retained as history.
|
||||
|
||||
## Summary
|
||||
|
||||
Today the dashboard is read-only: it surfaces pending proposals
|
||||
|
||||
@@ -4,6 +4,11 @@
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-05-26
|
||||
|
||||
> **Superseded.** PRD 0049 removed agent handoff and tmux pane management when
|
||||
> the dashboard was reduced to the proposal-only supervise TUI. The split-pane
|
||||
> interaction described below has no active replacement and is retained only
|
||||
> as design history.
|
||||
|
||||
## Summary
|
||||
|
||||
When the dashboard runs inside tmux, lay it out as the **left
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
# PRD 0024: Consolidate per-bottle sidecars into a single bundle
|
||||
|
||||
- **Status:** Active
|
||||
- **Status:** Superseded by [PRD 0070](0070-per-host-orchestrator.md)
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-05-26
|
||||
|
||||
> **Superseded.** PRD 0070 replaced the per-bottle sidecar bundle with a
|
||||
> persistent per-host gateway and separate orchestrator control plane. The
|
||||
> design below is retained as history.
|
||||
|
||||
## Summary
|
||||
|
||||
Replace the four per-bottle sidecar containers in the Docker
|
||||
|
||||
@@ -1,10 +1,16 @@
|
||||
# PRD 0037: Pipelock YAML Render Contract
|
||||
|
||||
- **Status:** Active
|
||||
- **Status:** Superseded by [PRD 0017](0017-egress-proxy-via-mitmproxy.md)
|
||||
and [PRD 0052](0052-egress-dlp-addon.md)
|
||||
- **Author:** didericis-codex
|
||||
- **Created:** 2026-06-02
|
||||
- **Issue:** #130
|
||||
|
||||
> **Superseded.** Pipelock and its YAML renderer were removed in issue #193.
|
||||
> Current egress configuration is consumed by the mitmproxy design from PRD
|
||||
> 0017 and its DLP addon from PRD 0052. The contract below is retained as
|
||||
> history.
|
||||
|
||||
## Summary
|
||||
|
||||
Lock down the contract between `pipelock_build_config` and
|
||||
|
||||
@@ -1,10 +1,17 @@
|
||||
# PRD 0067: SQLite local storage
|
||||
|
||||
- **Status:** Active
|
||||
- **Status:** Retargeted by [PRD 0070](0070-per-host-orchestrator.md) and
|
||||
issues #469/#471
|
||||
- **Author:** codex
|
||||
- **Created:** 2026-07-01
|
||||
- **Issue:** #319
|
||||
|
||||
> **Retargeted.** The SQLite storage and migration foundation remains in use,
|
||||
> but the writable data-plane database mount described below is no longer the
|
||||
> active ownership model. Issues #469/#471 removed `bot-bottle.db` from the
|
||||
> data plane; under PRD 0070 only the orchestrator control plane opens the
|
||||
> operational database, and gateway components reach state through RPC.
|
||||
|
||||
## Summary
|
||||
|
||||
Add a small stdlib SQLite storage layer for bot-bottle host runtime state,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0069: Firecracker-native, Docker-free backend
|
||||
|
||||
- **Status:** Draft (partially superseded)
|
||||
- **Status:** Retargeted by [PRD 0070](0070-per-host-orchestrator.md)
|
||||
- **Author:** Claude
|
||||
- **Created:** 2026-07-12
|
||||
- **Issue:** #348
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD 0070: Per-host orchestrator service
|
||||
|
||||
- **Status:** Draft
|
||||
- **Status:** Active
|
||||
- **Author:** Claude
|
||||
- **Created:** 2026-07-12
|
||||
- **Issue:** #351
|
||||
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
# PRD prd-new: Claude forward_host_credentials
|
||||
# PRD 0071: Claude forward_host_credentials
|
||||
|
||||
- **Status:** Draft
|
||||
- **Status:** Active
|
||||
- **Author:** claude
|
||||
- **Created:** 2026-07-01
|
||||
- **Issue:** #325
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
# PRD prd-new: Non-blocking supervise (async approval + proposal polling)
|
||||
# PRD 0072: Non-blocking supervise (async approval + proposal polling)
|
||||
|
||||
- **Status:** Draft
|
||||
- **Status:** Active
|
||||
- **Author:** didericis
|
||||
- **Created:** 2026-07-18
|
||||
- **Issue:** #412
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
# PRD prd-new: Consolidate infra backend for Docker
|
||||
# PRD 0073: Consolidate infra backend for Docker
|
||||
|
||||
- **Status:** Active
|
||||
- **Author:** Claude
|
||||
@@ -1,4 +1,4 @@
|
||||
# PRD prd-new: CI artifact-based coverage and local Firecracker candidate flow
|
||||
# PRD 0074: CI artifact-based coverage and local Firecracker candidate flow
|
||||
|
||||
- **Status:** Active
|
||||
- **Author:** Claude
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD prd-new: Containers inside a bottle
|
||||
# PRD 0075: Containers inside a bottle
|
||||
|
||||
- **Status:** Draft
|
||||
- **Status:** Active
|
||||
- **Author:** Claude
|
||||
- **Created:** 2026-07-21
|
||||
- **Issue:** #392
|
||||
+2
-2
@@ -1,6 +1,6 @@
|
||||
# PRD prd-new: Modernize built-in agent images
|
||||
# PRD 0076: Modernize built-in agent images
|
||||
|
||||
- **Status:** Draft
|
||||
- **Status:** Active
|
||||
- **Author:** Codex
|
||||
- **Created:** 2026-07-21
|
||||
- **Issue:** #451
|
||||
@@ -0,0 +1,142 @@
|
||||
# PRD 0077: macOS (Apple Container) CI runner
|
||||
|
||||
- **Status:** Active
|
||||
- **Author:** Claude
|
||||
- **Created:** 2026-07-25
|
||||
- **Issue:** #426
|
||||
|
||||
## Summary
|
||||
|
||||
CI has no runner for the `macos-container` (Apple Container) backend.
|
||||
`.gitea/workflows/test.yml` exercises Docker (`ubuntu-latest`) and
|
||||
Firecracker (self-hosted `kvm`) but never the macOS backend. This PRD adds a
|
||||
self-hosted macOS runner (label `macos`) and an advisory `integration-macos`
|
||||
job that runs the integration suite against `BOT_BOTTLE_BACKEND=macos-container`,
|
||||
so the backend that is the default on macOS stops shipping unexercised.
|
||||
|
||||
## Problem
|
||||
|
||||
The gap is not theoretical. `5ad3449` moved `bot_bottle` from flat files under
|
||||
`/app` into a pip-installed package but left init scripts spawning the
|
||||
supervisor as `python3 /app/gateway_init.py`, which no longer exists. Both the
|
||||
Firecracker and macOS backends carried the identical bug:
|
||||
|
||||
- **firecracker** — caught and fixed in `127ba49` because the KVM runner
|
||||
(added in `c193b04`, PR #349) runs that backend's integration suite.
|
||||
- **macos-container** — survived on `main` and only surfaced when a human ran
|
||||
`bot-bottle start` by hand.
|
||||
|
||||
The failure mode is expensive to debug: the supervisor never starts, so
|
||||
mitmdump never generates its CA, and launch dies downstream with
|
||||
`GatewayError: gateway CA not available`, which points at TLS rather than at
|
||||
the supervisor. Unit tests did not help — `test_macos_infra` asserted the
|
||||
substring `"gateway_init.py"`, which the *broken* path satisfies. (That
|
||||
specific assertion has since been tightened to the module form
|
||||
`bot_bottle.gateway_init`, matching its Firecracker twin, so the exact
|
||||
regression is now covered on `ubuntu-latest`. What remains missing is the
|
||||
end-to-end runner that would catch the *next* macOS-only launch regression.)
|
||||
|
||||
PR #470 (#414) already made the integration suite backend-agnostic:
|
||||
`skip_unless_selected_backend_available()` gates on the *selected* backend's
|
||||
own `is_backend_ready()` rather than `docker_available()`, and each
|
||||
integration job runs `./cli.py backend status --backend=<name>` as a preflight
|
||||
that fails loudly when the backend is missing. That is the machinery this job
|
||||
plugs into; this PRD supplies the runner and the job.
|
||||
|
||||
## Goals / Success criteria
|
||||
|
||||
- A macOS runner is registered and picks up jobs by the `macos` label.
|
||||
- An `integration-macos` job runs the integration suite against
|
||||
`BOT_BOTTLE_BACKEND=macos-container`.
|
||||
- The job **fails, not skips**, when the backend is unavailable on the runner
|
||||
(via the `backend status` preflight).
|
||||
- Reverting the `macos_container/infra.py` supervisor fix makes the job fail:
|
||||
the broken supervisor path throws `GatewayError` at bottle launch, which is
|
||||
`TestSandboxEscape.setUpClass`, failing the whole class before any individual
|
||||
attack runs.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Making `integration-macos` a **required** PR check. It runs on
|
||||
`workflow_dispatch` (manual dispatch) only — never on push or PRs. A single
|
||||
non-redundant laptop that sleeps and roams must never be able to block a PR
|
||||
merge or churn unattended on every push to main, and it is deliberately kept
|
||||
out of the `coverage` job's `needs` so the diff-coverage gate never depends on
|
||||
it.
|
||||
- Multi-machine or hosted macOS runners. Apple Container needs the host
|
||||
virtualization framework, so the runner must be a physical/VM macOS host on
|
||||
Apple Silicon — it cannot reuse the KVM runner or run in a Linux container.
|
||||
- Coverage aggregation from the macOS job into the combined gate (would couple
|
||||
the gate to the laptop).
|
||||
|
||||
## Design
|
||||
|
||||
### Runner (operational, provisioned once)
|
||||
|
||||
- Apple Silicon macOS host with Apple's `container` CLI installed and
|
||||
`container system status` reporting `running`.
|
||||
- Install the runner: `brew install gitea-runner` (the `act_runner` rename),
|
||||
registered in **host mode** with label `macos` — not docker mode, because
|
||||
Apple Container needs the host `container` CLI and virtualization framework,
|
||||
not a nested container.
|
||||
- A Python ≥ 3.11 with `coverage` importable on the runner's `PATH`. Because a
|
||||
launchd service does not inherit an interactive shell's `PATH`, pin `node`
|
||||
(for the JS `actions/*`) and the Python env explicitly in the service
|
||||
environment rather than relying on `nvm`/shell profile.
|
||||
- Concurrency 1. The infra container is a singleton (`bot-bottle-mac-infra`),
|
||||
so two simultaneous runs on one host collide (#425). The job also declares a
|
||||
`concurrency` group as belt-and-suspenders and tears the singleton down after
|
||||
each run.
|
||||
|
||||
### `integration-macos` job
|
||||
|
||||
Modeled on `integration-firecracker`:
|
||||
|
||||
- `runs-on: [self-hosted, macos]`.
|
||||
- `if:` `workflow_dispatch` only (advisory, manual dispatch; never push or PRs,
|
||||
so no fork-PR exposure, no merge-blocking, and no unattended runs on push).
|
||||
- `concurrency: { group: integration-macos-infra, cancel-in-progress: false }`
|
||||
to serialize runs against the singleton.
|
||||
- **Preflight** — `command -v container`, `container system status`, then
|
||||
`./cli.py backend status --backend=macos-container`; any failure exits
|
||||
non-zero so a misprovisioned runner fails loudly instead of silently
|
||||
skipping.
|
||||
- Run the integration suite under coverage with
|
||||
`BOT_BOTTLE_BACKEND=macos-container` and print a `coverage report -m` for
|
||||
visibility (no upload, not in the gate).
|
||||
- **Teardown** (`if: always()`) — `MacosInfraService().stop()` removes the
|
||||
singleton so a crashed run cannot wedge the next one.
|
||||
|
||||
### The `test_sandbox_escape` CI guard (the trap #470 left)
|
||||
|
||||
`TestSandboxEscape` is the only backend-agnostic integration test that boots a
|
||||
real bottle, so it is the one that would catch a macOS launch regression. It
|
||||
still carries a second guard that skips under `GITEA_ACTIONS` for every backend
|
||||
except `firecracker`:
|
||||
|
||||
```python
|
||||
@unittest.skipIf(
|
||||
os.environ.get("GITEA_ACTIONS") == "true"
|
||||
and os.environ.get("BOT_BOTTLE_BACKEND") != "firecracker",
|
||||
...,
|
||||
)
|
||||
```
|
||||
|
||||
The skip exists because the *containerized* `act_runner` (docker on
|
||||
`ubuntu-latest`) can't see a host bind mount and hides sibling-gateway network
|
||||
topology. Those constraints do not apply to a **host-mode** runner — neither
|
||||
the KVM host runner nor a macOS host runner is containerized. This PRD relaxes
|
||||
the guard to allow both host-mode backends (`firecracker`, `macos-container`)
|
||||
through while still skipping on the containerized Docker job. Without this
|
||||
change the macOS job would run green while skipping the exact test that proves
|
||||
the backend launches — the very false-green this issue is about.
|
||||
|
||||
## Open questions
|
||||
|
||||
- None known that block the job. git-gate is fully implemented on the macOS
|
||||
backend (the gateway's consolidated `git-http` daemon plus dynamic key
|
||||
provisioning/revocation), so `TestSandboxEscape` attack 5 — secret exfil
|
||||
pushed through git-gate, rejected by the gitleaks hook before the upstream
|
||||
push — runs the same as on the other backends. Any genuinely
|
||||
macOS-specific test adjustment would surface at first runner bring-up, but
|
||||
none is anticipated from the current backend implementation.
|
||||
@@ -1,6 +1,6 @@
|
||||
# PRD prd-new: Quick install script
|
||||
# PRD 0078: Quick install script
|
||||
|
||||
- **Status:** Draft
|
||||
- **Status:** Active
|
||||
- **Author:** claude
|
||||
- **Created:** 2026-07-25
|
||||
- **Issue:** #197
|
||||
@@ -33,7 +33,7 @@ user whether their host is actually ready to run a bottle.
|
||||
`bot-bottle doctor`. It never installs Docker or a VM backend silently
|
||||
and never uses `sudo`.
|
||||
- `install.sh` is idempotent — safe to re-run.
|
||||
- `bot-bottle doctor` reports Python version, backend availability, and
|
||||
- `bot-bottle doctor` reports Python version, backend *readiness*, and
|
||||
config-dir presence, exiting non-zero when a hard prerequisite is unmet.
|
||||
- The package keeps **zero runtime pip dependencies** (stdlib-only,
|
||||
matching the existing constraint in `AGENTS.md`).
|
||||
@@ -129,8 +129,11 @@ A POSIX `sh` bootstrapper that:
|
||||
4. Installs via `pipx` if available, else `python3 -m pip install --user`.
|
||||
The spec defaults to the git URL and is overridable via
|
||||
`BOT_BOTTLE_INSTALL_SPEC` (used by tests / local installs).
|
||||
4. Locates the `bot-bottle` entry point (PATH or `~/.local/bin`).
|
||||
5. Runs `bot-bottle doctor` and reports the result.
|
||||
5. Locates the `bot-bottle` entry point: PATH first, else the
|
||||
interpreter's own user-scheme scripts dir resolved via `sysconfig`
|
||||
(`~/.local/bin` on Linux, `~/Library/Python/<X.Y>/bin` on a python.org
|
||||
macOS interpreter — not hardcoded).
|
||||
6. Runs `bot-bottle doctor` and reports the result.
|
||||
|
||||
It is idempotent and never calls `sudo`.
|
||||
|
||||
@@ -140,10 +143,12 @@ A new store-free subcommand (no DB migration required) that checks and
|
||||
reports:
|
||||
|
||||
- **python** — interpreter version (hard requirement: ≥ 3.11).
|
||||
- **backend** — at least one backend available on this host
|
||||
(macos-container / firecracker / docker), reusing
|
||||
`is_backend_available()` rather than hardcoding Docker, since the
|
||||
default backend is now a VM backend. Hard requirement.
|
||||
- **backend** — at least one backend *ready* on this host
|
||||
(macos-container / firecracker / docker), via `is_backend_ready()` — a
|
||||
full backend `status()` probe (daemon reachable, network pool present,
|
||||
KVM usable), not a PATH-only check: a stopped daemon or half-configured
|
||||
backend must not report `ok` when `start` can't work. Each not-ready
|
||||
backend prints its own diagnostics. Hard requirement.
|
||||
- **config** — whether `~/.bot-bottle/` exists (advisory only; `start`
|
||||
provisions on first run).
|
||||
|
||||
@@ -152,7 +157,8 @@ Exits 0 when both hard requirements pass, non-zero otherwise.
|
||||
## Testing strategy
|
||||
|
||||
- Unit test `bot-bottle doctor` success/failure paths with backend
|
||||
availability and Python version mocked.
|
||||
readiness (`is_backend_ready`) and Python version mocked, including the
|
||||
available-but-not-ready → fail case.
|
||||
- Unit test that `pyproject.toml` parses, declares the entry point and an
|
||||
empty `dependencies` list, and that every `package-data` glob resolves
|
||||
to a file that exists on disk (guards against drift).
|
||||
@@ -0,0 +1,89 @@
|
||||
# PRD 0079: Per-service signing keys for control-plane auth
|
||||
|
||||
- **Status:** Active
|
||||
- **Author:** claude
|
||||
- **Created:** 2026-07-26
|
||||
- **Issue:** #476
|
||||
|
||||
## Summary
|
||||
|
||||
Provision control-plane signing keys **per service**, through one shared seam, so
|
||||
no service can mint another's credentials. Concretely: the orchestrator holds the
|
||||
control-plane key and mints the gateway's and CLI's tokens; the host controller
|
||||
(#468, next) gets a **separate** key the orchestrator never holds — so the
|
||||
orchestrator cannot forge the credentials it uses to talk to the host controller
|
||||
that starts and stops it. Landing this seam also retires the per-backend
|
||||
provisioning duplication that made PR #471 take three review rounds.
|
||||
|
||||
## Problem
|
||||
|
||||
**1. The orchestrator could forge host-controller credentials.** The
|
||||
orchestrator's key signs roles `{gateway, cli}`. The tempting way to add the host
|
||||
controller (#468) is a third role, `host`, on that same key. But then the
|
||||
orchestrator — which holds the key — can mint `host` tokens, and the host
|
||||
controller, which owns the orchestrator's lifecycle, must not trust anything the
|
||||
orchestrator can mint. The two services need separate keys.
|
||||
|
||||
**2. Every backend provisioned auth by hand.** Each launcher (docker
|
||||
gateway/infra, macOS infra, firecracker infra) re-derived how to generate the
|
||||
signing key, scope it to the orchestrator, mint the gateway JWT, and keep the
|
||||
host key file canonical. All three PR #471 High-severity findings were this one
|
||||
integration bug in different launchers: the data plane got the full `cli` token;
|
||||
the firecracker control plane ran open; the firecracker guest clobbered the host
|
||||
key.
|
||||
|
||||
## Goals / Success Criteria
|
||||
|
||||
- The orchestrator and the host controller sign with **different** keys; neither
|
||||
can mint the other's tokens. (This PR provisions the orchestrator's key and
|
||||
leaves a drop-in seam for the host controller's.)
|
||||
- One shared provisioning seam every backend uses — a new backend or daemon
|
||||
implements it instead of rediscovering these four invariants:
|
||||
1. the signing key is host-canonical: a guest is handed it, never generates or
|
||||
overwrites it;
|
||||
2. only the orchestrator process gets the raw key; the gateway gets a
|
||||
pre-minted `gateway` token it can't rewrite into `cli`;
|
||||
3. the host CLI's `cli` token is minted from the same key, so it stays valid
|
||||
across co-running backends;
|
||||
4. the orchestrator never runs open — the signing key is mandatory, with no
|
||||
topology opt-out (a separate host does not stop a caller from reaching the
|
||||
control-plane listener, so it cannot make open mode safe).
|
||||
|
||||
## Non-goals
|
||||
|
||||
- The host controller itself (#468) — this only provisions the orchestrator's
|
||||
key and the seam #468 plugs into.
|
||||
- Rewriting the HMAC primitive: `orchestrator_auth.mint/verify` gain an optional
|
||||
`roles=` arg (default unchanged) so a key can carry a different role set;
|
||||
nothing else changes.
|
||||
- Network topology, the plane split (#469), or the server's open-mode fallback
|
||||
for tests.
|
||||
|
||||
## Design
|
||||
|
||||
A **`TrustDomain`** is one service's signing material: its host-canonical key
|
||||
file, the roles that key may sign, and the env vars its key and a pre-minted
|
||||
token ride in. `mint`/`verify` are scoped to that domain's roles, so a token
|
||||
signed by one service's key neither carries nor verifies another service's role.
|
||||
|
||||
- `CONTROL_PLANE` — the orchestrator's domain: key `orchestrator-token`, roles
|
||||
`{gateway, cli}`. The orchestrator process holds the key; the gateway holds
|
||||
only a minted `gateway` token; the host CLI mints its own `cli` token.
|
||||
- The host controller (#468) will add a second `TrustDomain` — its own key file
|
||||
and role(s) — that the orchestrator never holds.
|
||||
|
||||
**`ControlPlaneProvisioning`** is the seam the backends call.
|
||||
`orchestrator_key()` returns the raw key for the control-plane process
|
||||
(fail-closed for every backend: it raises rather than hand back an empty key that
|
||||
would run the server open). `gateway_token()` mints the gateway's token. Each backend applies these through its own transport —
|
||||
docker/macOS inject env vars, firecracker pushes over SSH — but none re-derives
|
||||
*which* key or role.
|
||||
|
||||
`paths.host_signing_key(filename)` generalizes `host_orchestrator_token()` so each
|
||||
domain names its own key file.
|
||||
|
||||
## Open questions
|
||||
|
||||
None blocking. #468 adds its `TrustDomain` and a second
|
||||
`ControlPlaneProvisioning`-shaped consumer; renaming that class to something
|
||||
service-neutral is a cosmetic call to make then.
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
# PRD prd-new: Encrypted at-rest egress secrets (SecretProvider, interim slice)
|
||||
# PRD 0080: Encrypted at-rest egress secrets (SecretProvider, interim slice)
|
||||
|
||||
- **Status:** Draft
|
||||
- **Author:** didericis
|
||||
+8
-5
@@ -7,10 +7,13 @@ document vs. a research note or a decision record).
|
||||
|
||||
## Naming and numbering
|
||||
|
||||
New PRDs use a `prd-new-<kebab-title>.md` placeholder name while the PR
|
||||
is open. On merge to `main` a CI workflow assigns the next sequential
|
||||
number (`0024-…`, `0025-…`), renames the file, and updates the title
|
||||
header. Numbers are never reused; gaps are fine.
|
||||
New PRDs may use a `prd-new-<kebab-title>.md` placeholder name while the
|
||||
design is being drafted. Before merge, assign the next sequential number
|
||||
after the highest-numbered PRD on `main`, rename the file to
|
||||
`NNNN-<kebab-title>.md`, and update the title header. CI blocks merging
|
||||
while any `prd-new-*.md` placeholder remains. If concurrent PRs select the
|
||||
same number, the later PR must take the next available number before it
|
||||
merges. Numbers are never reused; gaps are fine.
|
||||
|
||||
Once numbered, the filename stays fixed for the life of the doc.
|
||||
|
||||
@@ -26,7 +29,7 @@ The `Status:` line near the top tracks the PRD's lifecycle:
|
||||
## Format
|
||||
|
||||
```markdown
|
||||
# PRD prd-new: <short title> ← placeholder; CI fills in the number on merge
|
||||
# PRD prd-new: <short title> ← replace with the final number before merge
|
||||
|
||||
- **Status:** Draft
|
||||
- **Author:** <who>
|
||||
|
||||
@@ -0,0 +1,732 @@
|
||||
# PRD prd-new: Canonical tamper-evident audit-event schema and local query contract
|
||||
|
||||
- **Status:** Draft
|
||||
- **Author:** didericis-claude
|
||||
- **Created:** 2026-07-26
|
||||
- **Issue:** #487
|
||||
|
||||
## Summary
|
||||
|
||||
bot-bottle already emits security- and provenance-relevant events from
|
||||
several producers — supervise operator decisions (PRD 0013's
|
||||
`AuditStore`), egress allow/block enforcement, git-gate push decisions,
|
||||
control-plane token minting, and (next) host-controller lifecycle
|
||||
transitions — but each writes its own shape to its own sink. There is no
|
||||
shared envelope, no tamper-evidence, and no single place to search. Local
|
||||
incident reconstruction means grepping several stores that don't agree on
|
||||
field names, timestamps, or how a bottled agent is identified.
|
||||
|
||||
This PRD defines **one canonical audit-event contract** every producer
|
||||
emits into:
|
||||
|
||||
1. A **versioned envelope** — schema version, event id, event/observed
|
||||
timestamps, host-attributed identity (`bottle`/`bottled_agent`/
|
||||
`activation`), provenance (`manifest_digest` — the manifest is the
|
||||
policy — and `engine` = bot-bottle version/SHA),
|
||||
host-observed `actor`/`action`/`resource`/`outcome`, correlation/
|
||||
causation ids, a sensitivity class, a typed payload, and an explicit
|
||||
**trust boundary** between host-supplied and agent-claimed fields.
|
||||
2. **Canonical JSON serialization + a per-writer hash chain** (with
|
||||
normative test vectors), so any deletion, edit, or reorder of a past
|
||||
record breaks the chain and is detectable offline.
|
||||
3. An **append-only JSONL journal as the source of truth**, with a
|
||||
**rebuildable SQLite index** and a local `audit query`/`verify` surface
|
||||
— no paid platform, no network dependency.
|
||||
4. An **initial event registry** covering lifecycle, host-controller,
|
||||
supervise decision, egress (request/decision/cutoff/anomaly), git-gate
|
||||
and signed-commit (#480), auth/authz, and audit self-events — each with
|
||||
its trusted-vs-claimed fields and redaction rules.
|
||||
5. A **stable export projection** (CloudEvents / OpenTelemetry Logs) and
|
||||
the **#324 delivery contract** (payload, per-chain
|
||||
`(host, epoch, seq)` cursor, dedup,
|
||||
backpressure, retention ordering).
|
||||
|
||||
It is explicitly scheduled to land **immediately after the host
|
||||
controller (#468)** so the host controller's lifecycle transitions are the
|
||||
first producer wired onto the new contract (per the directive on #487).
|
||||
|
||||
## Problem
|
||||
|
||||
Audit infrastructure is fragmented across #468, #324, and #480 with no
|
||||
shared schema. Concretely:
|
||||
|
||||
- **No shared envelope.** `supervise_audit_entries` (PRD 0013) has
|
||||
`timestamp, bottle_slug, component, operator_action, ...`. The egress
|
||||
proxy and git-gate log their own ad-hoc lines. There is no common
|
||||
`event_id`, `event_type`, or version, so cross-producer correlation
|
||||
("what did bottled agent X do between its start and this rejected push?") is
|
||||
manual and lossy.
|
||||
- **No tamper-evidence.** The audit store is a plain SQLite table. Anyone
|
||||
who can write the DB can delete or rewrite a row and leave no trace.
|
||||
Audit that an attacker (or a buggy agent) can silently rewrite is not
|
||||
audit.
|
||||
- **Trusted and untrusted data are mixed.** A bottled agent is attributed by
|
||||
**source IP → slug** at the gateway (host-supplied, trustworthy). An
|
||||
agent can also *claim* things about itself in a tool call
|
||||
(agent-claimed, adversarial). Today nothing in the record marks which is
|
||||
which, so a reader can be misled by an agent-supplied field that looks
|
||||
authoritative.
|
||||
- **No local search.** Reconstructing an incident means reading multiple
|
||||
sinks with different schemas. There is no query contract and no promise
|
||||
that the index can be rebuilt from the journal if it drifts or is lost.
|
||||
- **No redaction rule.** Nothing prohibits a producer from writing a raw
|
||||
token or secret into an audit record, which would turn the audit log
|
||||
itself into a credential store.
|
||||
|
||||
## Goals / Success Criteria
|
||||
|
||||
- A single `AuditEvent` envelope type, versioned, that every producer
|
||||
emits. The trust boundary is **one `untrusted` region**: everything
|
||||
outside it is host-established and trusted (source-IP → bottled-agent
|
||||
attribution, host wall-clock, producer identity, chain metadata);
|
||||
`untrusted` is the sole place anything an agent or a remote claimed may
|
||||
go. The boundary is structural, not a convention.
|
||||
- **Canonical serialization** (`sort_keys`, `(",", ":")` separators,
|
||||
UTF-8, `ensure_ascii=False`) is defined once and reused, so the same
|
||||
logical event always hashes identically across producers and hosts.
|
||||
- Each writer maintains a **hash chain**: `hash = sha256(prev_hash ||
|
||||
canonical(event))`. Deleting or editing any past record breaks every
|
||||
subsequent link; a standalone verifier detects the break offline with no
|
||||
secret material.
|
||||
- The **JSONL journal is the source of truth**; the **SQLite index is
|
||||
fully rebuildable** from it (`audit rebuild` reconstructs the DB and
|
||||
re-verifies the chain).
|
||||
- **Local query** works with no paid platform and no egress: filter by
|
||||
bottled agent, event type, time range, and producer, and follow a bottled
|
||||
agent's events in order.
|
||||
- A **redaction rule** is enforced at the envelope boundary: known
|
||||
credential-shaped fields are rejected/redacted before a record is
|
||||
written; the writer refuses raw secrets rather than storing them.
|
||||
- The envelope **projects onto the OpenTelemetry Logs data model and a
|
||||
CloudEvents JSON envelope** by field re-mapping alone (no reformat),
|
||||
preserving the trust boundary and carrying the integrity fields — per
|
||||
#487's export/interop requirement. (The export adapters are follow-up;
|
||||
the *schema* must make them a re-map.)
|
||||
- The **host controller (#468)** emits `lifecycle.*` events through this
|
||||
contract as the first consumer; existing supervise/egress producers are
|
||||
migrated behind the same envelope without changing operator-facing
|
||||
behavior.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- **Cross-host aggregation / shipping.** This PRD makes each host's journal
|
||||
canonical and correlatable *by construction* (stable ids, hash chain),
|
||||
but the transport that merges multiple hosts into one timeline is a
|
||||
follow-up (#324). The schema is designed so that merge is a later append,
|
||||
not a reformat.
|
||||
- **Cryptographic signing / external anchoring.** Hash-chaining gives
|
||||
tamper-**evidence** (you can detect edits), not tamper-**resistance**
|
||||
against an attacker who can rewrite the whole chain. Per-writer signing
|
||||
keys and periodic external anchoring are a follow-up; the chain-head hash
|
||||
is the seam they attach to.
|
||||
- **Real-time alerting / SIEM rules.** Query is local and pull-based here.
|
||||
- **Retention / rotation policy.** Journal rotation and TTL are operator
|
||||
policy, tracked separately; the format must survive rotation (chain head
|
||||
carried across segments) but this PRD does not set the schedule.
|
||||
- **Replacing PRD 0013's operator queue.** The supervise proposal/response
|
||||
queue is unchanged; only its terminal *audit* record is re-emitted onto
|
||||
the new envelope.
|
||||
|
||||
## Design
|
||||
|
||||
### The envelope
|
||||
|
||||
One dataclass, `AuditEvent`, serialized to a JSON object with a small,
|
||||
stable top level:
|
||||
|
||||
```
|
||||
{
|
||||
// ---- schema + integrity (host-owned) ----
|
||||
"v": 1, // schema version — bumped only on a breaking change
|
||||
"id": "<uuid4>", // globally unique event id; stable across export/replay (dedup key)
|
||||
"type": "egress.decision", // dotted event type from the registry
|
||||
"epoch": 7, // writer-boot counter, bumped once per host-controller (writer) start
|
||||
"seq": 1287, // monotonic sequence within this epoch (gap-detectable)
|
||||
"segment": "20260726T000000Z", // journal segment id (rotation boundary); chain continues across segments
|
||||
"prev": "<hex>", // prior record hash ("" only for the first record of a new chain)
|
||||
"hash": "<hex>", // sha256(prev + canonical(this event with hash=""))
|
||||
|
||||
// ---- timestamps (host-owned) ----
|
||||
"ts_event": "2026-07-26T18:22:04.061Z", // when the underlying event occurred at the boundary
|
||||
"ts_recorded": "2026-07-26T18:22:04.113Z", // when the single writer appended it (authoritative)
|
||||
"ts_mono": 90142.55, // monotonic secs since this epoch's boot (intra-epoch ordering only)
|
||||
|
||||
// ---- attribution + provenance (host-established) ----
|
||||
"producer": "egress", // host component that emitted the event
|
||||
"host": "mac-studio-1",
|
||||
"engine": "bot-bottle/0.1.0+abc1234", // bot-bottle version + git SHA of the enforcing host code
|
||||
"bottle": "amber-fox", // bottle (container/VM) identity
|
||||
"bottled_agent": "amber-fox-12", // bottled-agent slug from source-IP attribution (null for host-level events)
|
||||
"activation": "01J8Z...", // activation id: one run/session of the bottled agent (null if n/a)
|
||||
"manifest_digest": "sha256:9f2…", // digest of the manifest — which IS the policy (egress routes etc.); null if n/a
|
||||
|
||||
// ---- semantics: host-observed facts of what happened ----
|
||||
"actor": "bottled-agent:amber-fox-12", // who acted, as a host-attributed identity
|
||||
"action": "egress.connect", // what was attempted / done
|
||||
"resource": "registry.npmjs.org:443", // what it acted on, as observed at the boundary
|
||||
"outcome": "blocked", // host-decided result: allowed|blocked|deferred|success|failure
|
||||
"sensitivity": "security", // classification: normal|security|restricted (drives redaction + export)
|
||||
|
||||
// ---- correlation (host-assigned) ----
|
||||
"correlation_id": "flow-9c2a…", // groups a related flow (request → decision → cutoff)
|
||||
"causation_id": "<event id>", // the event that directly caused this one ("" if root)
|
||||
|
||||
// ---- typed, trusted, event-specific payload (shape fixed per type in the registry) ----
|
||||
"payload": {
|
||||
"route_id": 4,
|
||||
"detector": "token_patterns"
|
||||
},
|
||||
|
||||
// ---- the ONLY untrusted region: agent- or remote-claimed data ----
|
||||
"untrusted": {
|
||||
"reason": "npm install needs registry.npmjs.org" // the agent's stated justification
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The **trust boundary is a single region, not a split.** Everything outside
|
||||
`untrusted` is trusted by construction — the host established it: schema and
|
||||
chain metadata, both timestamps, the attribution/provenance fields
|
||||
(source-IP → `bottled_agent`/`bottle`/`activation`, `manifest_digest`,
|
||||
`engine`), the host-observed semantics
|
||||
(`actor`/`action`/`resource`/`outcome`), the host-assigned correlation ids,
|
||||
and the typed `payload`. `untrusted` is the **one** place anything an agent
|
||||
or a remote claimed may go (e.g. the agent's free-text `reason`). A reader
|
||||
(or a future policy engine) trusts every field outside `untrusted` for
|
||||
attribution and treats `untrusted.*` — and only `untrusted.*` — as
|
||||
adversarial claims.
|
||||
|
||||
Framing it as "one untrusted region, everything else trusted" removes the
|
||||
mistake where a producer forgets to mark a claimed field: a field is
|
||||
trusted unless it is deliberately placed inside `untrusted`. The
|
||||
construction API enforces this — producers pass trusted fields explicitly
|
||||
and hand all agent/remote-claimed data as the single `untrusted` mapping,
|
||||
so there is no way to emit a top-level field that *looks* authoritative but
|
||||
isn't.
|
||||
|
||||
**Two timestamps** because they answer different questions and can diverge
|
||||
under backpressure: `ts_event` is when the thing happened at the boundary
|
||||
(the proxy saw the connect, the gate saw the push); `ts_recorded` is when
|
||||
the single writer durably appended it. Ordering and the chain use
|
||||
`(epoch, seq)`, never either wall clock. Both are host-set — a bottled
|
||||
agent never supplies a timestamp.
|
||||
|
||||
**The manifest *is* the policy.** bot-bottle has no separate policy
|
||||
artifact — a bottled agent's egress routes and other constraints are
|
||||
declared in its manifest (`bot_bottle/manifest/egress.py`), so
|
||||
`manifest_digest` already pins the ruleset in force; there is no distinct
|
||||
`policy_version`. Given a fixed manifest, the only other thing that can
|
||||
change a decision's outcome is the enforcing code — captured by `engine`
|
||||
(bot-bottle version + git SHA). So two `egress.decision` records with the
|
||||
same `resource` but different `outcome` are explained by exactly one of:
|
||||
different `manifest_digest` (the rules changed) or different `engine` (the
|
||||
enforcer changed). Runtime operator overrides (a supervise `egress-allow`)
|
||||
are themselves audit events, so the effective ruleset at any instant is
|
||||
`manifest_digest` plus the logged, approved deltas — reconstructable from
|
||||
the chain, not from a version stamp.
|
||||
|
||||
**Optionality.** `bottle`/`bottled_agent`/`activation`, `manifest_digest`,
|
||||
and `payload`/`untrusted` are `null`/absent for events that have no such
|
||||
subject (a host-level `hostctl.*` or `audit.*` event has no bottled agent).
|
||||
Absent ≠ empty: a reader distinguishes "no subject" from "unknown". `id`,
|
||||
`type`, the chain fields, both timestamps, `producer`, `host`, `engine`,
|
||||
`actor`, `action`, `outcome`, and `sensitivity` are always present.
|
||||
|
||||
#### Trust provenance of every common field
|
||||
|
||||
| Field | Trust | Set by |
|
||||
|---|---|---|
|
||||
| `v` `id` `type` `epoch` `seq` `segment` `prev` `hash` | trusted | the single writer |
|
||||
| `ts_event` | trusted | emitting host component (boundary) |
|
||||
| `ts_recorded` `ts_mono` | trusted | the single writer |
|
||||
| `producer` `host` `engine` | trusted | the single writer |
|
||||
| `bottle` `bottled_agent` `activation` | trusted | gateway source-IP → slug attribution |
|
||||
| `manifest_digest` | trusted | control plane (the manifest = the policy in force) |
|
||||
| `actor` `action` `resource` `outcome` | trusted | host component that observed/decided it |
|
||||
| `sensitivity` | trusted | registry default for `type`, overridable up (never down) by the producer |
|
||||
| `correlation_id` `causation_id` | trusted | the single writer (assigned as it threads the flow) |
|
||||
| `payload.*` | trusted | emitting host component (shape fixed per `type`) |
|
||||
| `untrusted.*` | **claimed** | copied verbatim from a bottle / gateway / forge / remote |
|
||||
|
||||
Every registry entry (below) restates, per event type, which `payload`
|
||||
keys are required and names any `untrusted` keys it carries — so "trusted
|
||||
vs claimed" is explicit for every event-specific attribute, not just the
|
||||
common ones.
|
||||
|
||||
### Canonical serialization + hash chain
|
||||
|
||||
Serialization is defined once (extends the existing `sha256_hex` /
|
||||
`util.py` helpers):
|
||||
|
||||
```
|
||||
def canonical(event: dict) -> str:
|
||||
return json.dumps(event, sort_keys=True, separators=(",", ":"),
|
||||
ensure_ascii=False)
|
||||
```
|
||||
|
||||
The `hash` field is computed over the canonical form of the event **with
|
||||
`hash` set to `""`**, prefixed by the previous record's hash:
|
||||
|
||||
```
|
||||
digest = sha256_hex(prev_hash + canonical({**event, "hash": ""}))
|
||||
```
|
||||
|
||||
`prev` is the prior record's `hash`; only the first record of a brand-new
|
||||
chain uses `prev = ""`. The first record of a rotated segment carries the
|
||||
preceding segment's head, as specified under *Rotation* below.
|
||||
Two exact rules pin the bytes so the chain is reproducible anywhere:
|
||||
|
||||
1. **Serialize the record with its own `hash` field set to `""`** (present,
|
||||
empty), never omitted — the key set is identical before and after
|
||||
hashing.
|
||||
2. **Digest = `sha256_hex(prev + canonical(record_with_empty_hash))`**,
|
||||
where `prev` is the previous record's `hash` string (`""` at genesis),
|
||||
`+` is string concatenation, and `canonical` is the function above.
|
||||
`ts_mono`, being a float, is serialized by Python's shortest-round-trip
|
||||
`repr` via `json.dumps`; producers therefore emit it as a JSON number
|
||||
they do not post-process. (All other fields are strings/ints/objects,
|
||||
which serialize unambiguously.)
|
||||
|
||||
Editing or deleting record *n* changes its hash, so record *n+1*'s `prev`
|
||||
no longer matches — the break is local and names the tampered record.
|
||||
Verification needs only the journal itself (no keys), so it runs offline
|
||||
and in CI.
|
||||
|
||||
#### Test vectors (normative)
|
||||
|
||||
Two records, reduced to the chain-relevant fields, demonstrate the exact
|
||||
serialization and linkage. An implementation is conformant iff it
|
||||
reproduces these bytes and hashes.
|
||||
|
||||
```
|
||||
# Record 0 — brand-new chain genesis (prev = "")
|
||||
canonical(record0, hash=""):
|
||||
{"hash":"","id":"11111111-1111-4111-8111-111111111111","prev":"","seq":0,"type":"audit.segment_open"}
|
||||
hash0 = sha256("" + canonical) =
|
||||
942ea5729bcac6efdbdea942396bfa574ab0d6ebf5615402595359422f2aeb83
|
||||
|
||||
# Record 1 — chains onto record 0 (prev = hash0)
|
||||
canonical(record1, hash=""):
|
||||
{"hash":"","id":"22222222-2222-4222-8222-222222222222","prev":"942ea5729bcac6efdbdea942396bfa574ab0d6ebf5615402595359422f2aeb83","seq":1,"type":"lifecycle.bottled_agent_start"}
|
||||
hash1 = sha256(hash0 + canonical) =
|
||||
bc082347680405fee50b60a9c304611aa026950b15d869b7e3ae56e1c451b856
|
||||
|
||||
# Tamper check: flip record0.type → recompute →
|
||||
# 4553eda647f33f0c608cfea44be28efbbaca45ed30b873fcbd4405fa5ce737ed
|
||||
# which no longer equals record1.prev (942ea5…) — the break is detected at record1.
|
||||
```
|
||||
|
||||
The implementation PR ships these plus full-envelope vectors (every field
|
||||
populated, and a redaction case) as committed fixtures, so a schema-version
|
||||
bump that changes the bytes fails a golden test loudly.
|
||||
|
||||
### Ordering, idempotency, and duplicate handling
|
||||
|
||||
- **Ordering.** `(epoch, seq)` is a strict total order within one host's
|
||||
native chain and, because
|
||||
the writer is single, a strict order per bottle/activation within that
|
||||
host — satisfying "at least strict causal order per activation/bottle".
|
||||
`causation_id` records the explicit cause edges (a DAG) on top of the
|
||||
total order, so a consumer can reconstruct request → decision → cutoff
|
||||
even if unrelated events interleave between them. There is deliberately
|
||||
no invented total order across imported host chains; a cross-host key is
|
||||
`(host, epoch, seq)`, and consumers use correlation/causation edges where
|
||||
causal ordering across hosts is known.
|
||||
- **Idempotency.** `id` is the idempotency key. A producer that retries an
|
||||
emit (e.g. after a writer restart mid-handoff) **reuses the same `id`**;
|
||||
before appending, the writer checks a durable, host-wide id ledger that
|
||||
spans every segment and survives restart/retention, and drops an id
|
||||
already present. The ledger is operational metadata, not an audit source
|
||||
of truth: after a crash it is reconciled from the journal before appends
|
||||
resume, and retention preserves id tombstones after journal segments are
|
||||
pruned. An in-memory set may cache the ledger but is never the authority.
|
||||
The index additionally has a unique key on `id`; its `UPSERT` is
|
||||
defensive and does not substitute for the pre-append check. Thus a
|
||||
duplicate never enters the source-of-truth journal, double-counts, or
|
||||
forks the chain.
|
||||
- **Deduplication downstream.** Because `id` is stable across export and
|
||||
replay, #324's cursor replay and any cross-host merge dedup on `id` — no
|
||||
consumer needs to invent a second identity.
|
||||
|
||||
### Behavior across rotation, restart, import, truncation
|
||||
|
||||
- **Rotation.** At a segment boundary the writer opens a new segment file,
|
||||
sets its `segment` id, and carries the rotated-out segment's head as the
|
||||
new segment's first `prev` — so the chain is continuous *across* segments
|
||||
(`prev = ""` is reserved for the first record of a brand-new chain) while
|
||||
each file stays independently openable. `verify` walks segments in order
|
||||
and checks the preceding-head-to-first-record link at each seam.
|
||||
- **Restart.** Covered above: read last line → adopt its `hash` as `prev`,
|
||||
bump `epoch`, reset `seq`. The chain never restarts even though the
|
||||
counters do.
|
||||
- **Import.** `audit import <segment>` appends an externally supplied
|
||||
segment (e.g. recovered from another host or a backup). Import verifies
|
||||
the incoming chain in isolation first. A continuation whose first `prev`
|
||||
matches a known head extends that chain. A foreign chain is registered as
|
||||
a separate immutable chain namespace rather than rewriting or grafting
|
||||
its records (which would invalidate their hashes). Imported records keep
|
||||
their original `host`, `id`, `epoch`, and `seq`; the index keys their
|
||||
native order by `(host, epoch, seq)` so attribution is not laundered and
|
||||
tuples from different hosts cannot collide.
|
||||
- **Truncation.** A crash can leave a partial final line; `verify` reports
|
||||
it as `truncated-tail` (recoverable — replay resumes from the last intact
|
||||
record). A chain that ends before a persisted head, or a missing interior
|
||||
`seq`, is reported as `gap`/`missing-suffix` (evidence of deletion, not a
|
||||
clean crash). The two are distinguished so an operator can tell "power
|
||||
loss" from "someone trimmed the log".
|
||||
|
||||
### Single writer; ordering across restarts
|
||||
|
||||
**Decided: one writer per host** (reviewed — the host controller owns it).
|
||||
Producers hand events to the host controller, which is the sole appender,
|
||||
so the chain has one well-defined total order and one `seq`/`epoch`
|
||||
counter. This ties audit availability to the host controller being up,
|
||||
which is acceptable because the host controller already gates every
|
||||
lifecycle transition; per-producer chains are noted only as a future
|
||||
scaling path, not built now.
|
||||
|
||||
**Restarts** are handled by the chain, not the clock. `ts_mono` resets to
|
||||
~0 on every writer start, so it orders events only *within* one boot. On
|
||||
start the writer:
|
||||
|
||||
1. reads the last line of the journal, adopts its `hash` as the next
|
||||
record's `prev` (the chain is continuous across the restart), and
|
||||
2. bumps `epoch` (persisted alongside the chain head) and resets `seq` to
|
||||
0 for the new boot.
|
||||
|
||||
Total order within this host's native chain is therefore `(epoch, seq)` —
|
||||
monotonic across restarts by construction — with
|
||||
`ts_event`/`ts_recorded` for human reading and `ts_mono` for sub-second
|
||||
ordering inside an epoch. Imported chains retain their own
|
||||
`(host, epoch, seq)` order and do not acquire a fictional order relative to
|
||||
the local chain. A crash mid-append truncates at most the last (partial)
|
||||
line; the verifier flags it and replay resumes from the last intact record.
|
||||
|
||||
### Journal (source of truth) + SQLite index (rebuildable)
|
||||
|
||||
- **Journal:** one append-only JSONL file per host (path from `paths.py`,
|
||||
alongside `host_db_path()`), one canonical event per line, opened
|
||||
`O_APPEND`. This is authoritative.
|
||||
- **Index:** a new `audit_events` table via the existing `DbStore` /
|
||||
`TableMigrations` machinery. It is a **derived cache, not a second source
|
||||
of truth**: `audit rebuild` truncates and replays the journal,
|
||||
re-verifying the chain as it goes, so a deleted or drifted DB is
|
||||
regenerated from the journal with no data loss. On a `verify` failure
|
||||
during rebuild it stops and reports rather than indexing past a break.
|
||||
(This supersedes the free-standing `supervise_audit_entries` table, which
|
||||
becomes a producer onto the new index.)
|
||||
|
||||
**Indexable fields** (columns + indices): `ts_event`, `ts_recorded`,
|
||||
`type`, `host`, `bottle`, `bottled_agent`, `activation`, `actor`,
|
||||
`outcome`, `sensitivity`, `correlation_id`, `causation_id`, plus two
|
||||
event-specific projections promoted out of `payload` for query —
|
||||
`repository` and `commit_sha` (populated for `forge.*`/`commit.*`, null
|
||||
otherwise). The unique event id and native-order index are respectively
|
||||
`id` and `(host, epoch, seq)`; a local-only `ingest_seq` provides stable
|
||||
display order when a query intentionally mixes chains without pretending
|
||||
that it is causal order. The full canonical record is stored verbatim in a
|
||||
`raw` column so the index never loses fidelity to the journal.
|
||||
|
||||
**Local query surface** — `audit query`, no egress, no paid platform:
|
||||
|
||||
```
|
||||
audit query \
|
||||
[--since T] [--until T] [--type egress.*] [--host H] [--bottle B] \
|
||||
[--activation A] [--agent SLUG] [--actor ID] [--outcome blocked] \
|
||||
[--repository R] [--correlation-id C] [--commit SHA] \
|
||||
[--follow BOTTLE] # one host chain's events in (epoch, seq) order
|
||||
[--json | --table]
|
||||
audit verify [--segment S] # offline chain check; exit non-zero on any break
|
||||
audit rebuild # drop + replay journal → index
|
||||
audit import <segment> # graft an external segment (see above)
|
||||
```
|
||||
|
||||
Type filters accept a `group.*` glob. A read-only local HTTP endpoint
|
||||
mirrors the same filters for the future review console; both are pure reads
|
||||
over the index and can never mutate the journal.
|
||||
|
||||
### Event registry (initial)
|
||||
|
||||
Dotted `type` names, grouped. The registry is a table mapping each type to
|
||||
its required `payload` keys, its `untrusted` keys (if any), a default
|
||||
`sensitivity`, and its correlation behavior — so producers and the verifier
|
||||
agree on shape and "trusted vs claimed" is pinned per type. Initial
|
||||
coverage (the issue's mandated set):
|
||||
|
||||
| Group / type | Producer | Required `payload` (trusted) | `untrusted` | Default sensitivity |
|
||||
|---|---|---|---|---|
|
||||
| **lifecycle.*** — `bottled_agent_start` / `_stop` / `_crash` | host-controller (#468) | `manifest_digest`, `exit` (for stop/crash) | — | normal |
|
||||
| **hostctl.*** — `broker_launch`, `broker_teardown`, `broker_reject` | host-controller (#468) | `op`, `request_digest` | — | security |
|
||||
| **decision.*** — `proposed`, `resolved` | supervise | `tool`, `operator_action`, `justification`, `diff_digest` | `agent_rationale` | security |
|
||||
| **egress.*** — `request`, `decision`, `cutoff`, `anomaly` | egress proxy | `route_id`, `detector` (on match), `bytes` (cutoff) | `reason`, `target_claimed` | security |
|
||||
| **forge.*** — `push_accepted`, `push_rejected`, `pr_opened` | git-gate | `repository`, `ref`, `gitleaks_result` | `title`, `description` | security |
|
||||
| **commit.signed** (#480) | git-gate | `repository`, `commit_sha`, `activation_key_id`, `signature_ref` | `commit_message` | security |
|
||||
| **auth.*** — `token_minted`, `token_rejected`, `authz_denied` | control plane | `role`, `token_id`, `reason_code` | — | security |
|
||||
| **audit.*** — `segment_open`, `verify_failed`, `truncation_detected`, `export_failed` | audit writer/verifier | `segment`, `detail` | — | security |
|
||||
|
||||
Notes:
|
||||
- **`egress.request` vs `egress.decision`** share a `correlation_id`; the
|
||||
`decision`'s `causation_id` points at the `request`, and a later `cutoff`
|
||||
chains onto the `decision` — so a flow is reconstructable.
|
||||
- **`audit.*` self-events** make the audit subsystem audit itself: a failed
|
||||
verification, a detected truncation, or a dropped export is itself a
|
||||
chained, tamper-evident record — you cannot silence the alarm without
|
||||
breaking the chain that carries it.
|
||||
- **Free-text and remote-echoed fields are always `untrusted`** (`reason`,
|
||||
`agent_rationale`, PR `title`/`description`, `target_claimed`), because
|
||||
they originate in the bottle or a remote response; the host-observed
|
||||
counterpart (`resource`, `outcome`, `gitleaks_result`) is the trusted
|
||||
fact.
|
||||
|
||||
**Sensitivity + redaction per type.** Every type's default `sensitivity`
|
||||
is listed above; a producer may raise it (never lower it). `restricted`
|
||||
events keep their `payload` in the journal but the export projection ships
|
||||
only the envelope + a payload digest unless the consumer is authorized —
|
||||
so a `security`/`restricted` record is still counted and correlated
|
||||
downstream without leaking its body. The credential-shape redaction rules
|
||||
(next) apply to **every** type regardless of sensitivity.
|
||||
|
||||
**Schema evolution & backward-compatible readers.** The registry is
|
||||
append-only: **adding** a type, an optional `payload` key, or an
|
||||
`untrusted` key does **not** bump `v`; readers ignore unknown fields
|
||||
(forward-compatible) and treat absent optional fields as `null`.
|
||||
**Removing** or **re-typing** a field, or making an optional field
|
||||
required, bumps `v`. A reader declares the max `v` it understands and
|
||||
refuses to *interpret* a higher-`v` record, but the **verifier is
|
||||
version-agnostic** — the hash covers whatever fields exist, so chain
|
||||
integrity is checkable across versions without understanding semantics.
|
||||
Every `v` bump ships a migration note and updated golden vectors.
|
||||
|
||||
### Redaction rule
|
||||
|
||||
Redaction runs at the envelope boundary, before a record is written, in two
|
||||
layers:
|
||||
|
||||
1. **Key deny-list (structural).** A field whose *key* matches a known
|
||||
credential shape (`token`, `secret`, `password`, `authorization`,
|
||||
`*_key`) is refused — the producer must pass a reference (a token *id*
|
||||
or `sha256` fingerprint), never the raw value. `auth.token_minted`
|
||||
therefore records the token id and role, not the JWT. This is the
|
||||
primary guard: it is cheap, deterministic, and catches the intended
|
||||
mistake (a producer stuffing a credential into a named field).
|
||||
|
||||
2. **Value scan — reuse the egress DLP detectors.** Per review, the value
|
||||
layer reuses the *same* deterministic credential-shape detectors the
|
||||
egress proxy already ships:
|
||||
`bot_bottle/gateway/egress/dlp_detectors.py` —
|
||||
`scan_token_patterns` / `redact_tokens` (and `scan_known_secrets` for
|
||||
host-known secret material). They are pure-Python, mitmproxy-free, and
|
||||
already the project's source of truth for "what a leaked credential
|
||||
looks like," so a single detector set governs both what may leave over
|
||||
the wire and what may land in the journal — they can't drift apart.
|
||||
|
||||
**Scoped deliberately:** only the pattern/known-secret detectors are
|
||||
reused, **not** `scan_entropy`. Entropy scoring is tuned for large
|
||||
streamed request bodies; on the short, high-entropy structured values an
|
||||
audit event legitimately carries (hashes, uuids, base64 ids) it would
|
||||
false-positive and start redacting the very fingerprints the log needs.
|
||||
So the shared layer is the deterministic detectors; entropy stays an
|
||||
egress-only concern. (This is the "evaluate how reasonable that is" from
|
||||
review: reuse the deterministic detectors — yes; share the entropy
|
||||
heuristic — no.)
|
||||
|
||||
On a value-layer match the default is **redact** (scrub to a placeholder
|
||||
and keep the event) rather than drop, so a producer bug can never make an
|
||||
audit event vanish; the key deny-list stays a hard refusal because a
|
||||
credential in a named field is always a producer bug worth surfacing.
|
||||
|
||||
**No raw-payload capture by default.** The envelope carries *decisions and
|
||||
metadata*, not traffic. Prompts, model responses, request/response bodies,
|
||||
and file contents are **not** recorded unless a producer opts a specific,
|
||||
reviewed field in — and such a field is `untrusted` and subject to both
|
||||
redaction layers. This keeps the audit log from becoming a covert copy of
|
||||
the very data the sandbox exists to contain (the issue's "unsafe payload
|
||||
capture" non-goal).
|
||||
|
||||
### Export / interoperability (CloudEvents, OpenTelemetry Logs)
|
||||
|
||||
#487 requires the envelope to map onto the **OpenTelemetry Logs data
|
||||
model** and/or a **CloudEvents JSON** envelope *without losing integrity or
|
||||
attribution semantics*. The flattened shape (one `untrusted` region,
|
||||
everything else trusted at top level) does **not** conflict with either — it
|
||||
maps *more* cleanly than a nested `trusted`/`untrusted` pair would, because
|
||||
both target models expect a flat set of top-level fields plus one payload
|
||||
subtree.
|
||||
|
||||
**CloudEvents.** Context attributes MUST be scalar simple types — a map
|
||||
cannot be a context attribute — so a nested `trusted` block would have had
|
||||
to be flattened for CloudEvents anyway. Our flat top level maps directly:
|
||||
`id`→`id`, `type`→`type`, `producer`+`host`→`source`,
|
||||
`bottled_agent`→`subject`, `ts_event`→`time` (`ts_recorded` as an extension); the integrity/chain fields
|
||||
(`epoch`, `seq`, `prev`, `hash`, `v`) ride as **extension attributes**
|
||||
(scalars — legal). The `untrusted` map goes in `data`. Only mechanical
|
||||
transform needed: extension attribute names must be lowercase-alphanumeric,
|
||||
so `bottled_agent`/`ts_mono`/etc. are renamed at export (e.g. a
|
||||
`botbottle`-prefixed form) — a naming rule, not a schema conflict.
|
||||
|
||||
**OpenTelemetry Logs.** `ts_event`→`Timestamp`, `ts_recorded`→`ObservedTimestamp`; `type`→the `event.name`
|
||||
attribute; the flat trusted fields → `Attributes` under a `botbottle.*`
|
||||
namespace (`botbottle.bottled_agent`, `botbottle.producer`,
|
||||
`botbottle.chain.hash`, …); `untrusted.*` → `Attributes` under
|
||||
`botbottle.untrusted.*` (or `Body`). OTel attributes are a dotted map that
|
||||
happily carries the nested subtree.
|
||||
|
||||
**Attribution is preserved** precisely because the boundary is now
|
||||
structural: on export, top-level fields become trusted context/attributes
|
||||
and the `untrusted` subtree stays a single, clearly-named region — so a
|
||||
downstream consumer still sees exactly which fields an agent claimed.
|
||||
Nothing agent-claimed is promoted to a trusted-looking position.
|
||||
|
||||
**Integrity has one deliberate caveat.** CloudEvents/OTel are
|
||||
representation envelopes with their own (or no) canonicalization; `hash`
|
||||
and `prev` are computed over **our** canonical JSON, not over the exported
|
||||
form. So the chain fields travel *as data* for reference, but
|
||||
tamper-evidence is always verified against the **native journal** (the
|
||||
source of truth) — never re-derived from an exported CloudEvents/OTel
|
||||
record, whose key ordering / number formatting the exporter may change.
|
||||
Export is thus a lossless-for-attribution **projection** that carries the
|
||||
integrity fields along; verification stays on the canonical journal. This
|
||||
satisfies "without losing integrity or attribution semantics": both are
|
||||
carried, neither is *relied upon* in the foreign format.
|
||||
|
||||
The export adapters themselves are follow-up implementation — this PRD
|
||||
fixes the *schema* so that projection is a field re-map, never a reformat.
|
||||
|
||||
#### The #324 delivery contract (payload, cursor, backpressure)
|
||||
|
||||
#324 transports events off-box; it must not invent a second envelope. This
|
||||
PRD fixes the contract it depends on:
|
||||
|
||||
- **Payload.** #324 ships the **native canonical record verbatim** (the
|
||||
exact bytes the hash covers), optionally wrapped in the CloudEvents
|
||||
projection whose `data` *is* that record. Either way the integrity fields
|
||||
travel intact and the receiver can verify against the same bytes.
|
||||
- **Cursor.** Export maintains one cursor per native host chain:
|
||||
`(host, epoch, seq)` (equivalently that chain's last exported `hash`).
|
||||
It advances **only on acknowledgement**, so delivery is at-least-once and
|
||||
gap-free; a crash re-sends from the last acked cursor. Imported foreign
|
||||
chains use independent cursors and never share a bare `(epoch, seq)`
|
||||
namespace with the local chain.
|
||||
- **Idempotency / replay.** Dedup is on `id` (stable across replay), so
|
||||
at-least-once delivery is safe — the receiver collapses re-sends.
|
||||
- **Backpressure.** The outbox is the journal itself plus a cursor; when
|
||||
the endpoint is slow the cursor simply lags — the writer never blocks on
|
||||
export, and audit never applies backpressure to the data plane it
|
||||
records.
|
||||
- **Retention interaction.** Retention/rotation **must not** prune a
|
||||
segment whose records are still behind the export cursor; the reaper
|
||||
honors `min(cursor)` across all configured consumers. (The schedule
|
||||
itself stays the retention follow-up; this is the *ordering* constraint
|
||||
that follow-up must respect.)
|
||||
|
||||
#### #480 signed-commit attribution maps in without weakening it
|
||||
|
||||
#480 binds a commit's bytes to a per-activation signing key. It maps to the
|
||||
`commit.signed` event: `payload` carries `repository`, `commit_sha`,
|
||||
`activation_key_id`, and a `signature_ref` (the detached-signature
|
||||
location or its digest) — **not** the private key and not a re-derived
|
||||
signature. The audit event therefore *references and timestamps* #480's
|
||||
existing byte-to-activation-key proof inside the tamper-evident chain; it
|
||||
does not re-implement or replace it, so #480's guarantee is unweakened —
|
||||
the signature still verifies against the commit bytes independently, and
|
||||
the audit record adds only "this binding was observed at this point in the
|
||||
chain". The trusted `actor`/`activation` fields and the `commit_sha`
|
||||
payload are host-observed at the gate, so attribution cannot be forged by
|
||||
the committing agent.
|
||||
|
||||
## Implementation chunks
|
||||
|
||||
1. **(this PR — PRD only.)** The contract above. No code; scheduled to land
|
||||
right after #468.
|
||||
2. **Envelope + canonical + chain core.** `AuditEvent` dataclass,
|
||||
`canonical()`, chain hashing, and the single-writer journal appender in
|
||||
`bot_bottle/store/` (reusing `sha256_hex`); redaction wired to the
|
||||
existing `gateway/egress/dlp_detectors` (`scan_token_patterns` /
|
||||
`redact_tokens`); embed the git SHA at build so `engine` is populated
|
||||
(only `version = "0.1.0"` exists in `pyproject.toml` today — the build
|
||||
must stamp the SHA); unit tests for determinism, chain-break detection,
|
||||
`epoch`/`seq` continuity across a simulated restart, and redaction of
|
||||
both a deny-listed key and a token-shaped value.
|
||||
3. **SQLite index + `audit` CLI.** New `audit_events` migration (indexable
|
||||
fields above); replay-from-journal; offline chain verifier
|
||||
(`truncated-tail` vs `gap`); `query` / `verify` / `rebuild` / `import`;
|
||||
idempotent `UPSERT` by `id`.
|
||||
4. **Host controller as first producer (#468).** Wire
|
||||
`lifecycle.bottled_agent_*` and `hostctl.*` emission into the host
|
||||
controller; establish the `epoch` bump + chain-head carry + segment
|
||||
rotation on writer restart here (it owns the single writer).
|
||||
5. **Migrate existing producers.** Re-emit supervise `decision.*` (retiring
|
||||
the standalone `supervise_audit_entries` shape behind the index), egress
|
||||
`egress.*`, git-gate `forge.*` + `commit.signed` (#480), control-plane
|
||||
`auth.*`; add the `audit.*` self-events (verify/truncation/export
|
||||
failure).
|
||||
6. **CloudEvents / OTel export adapters + #324 delivery.** Projection layer
|
||||
(field re-map per *Export / interoperability*) plus the outbox cursor,
|
||||
ack-driven advance, and retention-ordering guard the #324 contract
|
||||
specifies.
|
||||
7. **(follow-up.)** Cross-host merge transport; per-writer signing +
|
||||
external anchoring on the chain head; retention/rotation *schedule*.
|
||||
|
||||
## Acceptance-criteria coverage (#487)
|
||||
|
||||
The issue defines the contract; implementation is explicitly split into
|
||||
follow-up PRs. This PRD is the durable decision record; each acceptance box
|
||||
maps to a section:
|
||||
|
||||
| #487 acceptance criterion | Where |
|
||||
|---|---|
|
||||
| Durable PRD defines versioned envelope + initial registry | *The envelope*, *Event registry* |
|
||||
| Canonical JSON + hash-chain rules, unambiguous, with test vectors | *Canonical serialization + hash chain* → *Test vectors* |
|
||||
| Trust provenance explicit for every common + event-specific field | *Trust provenance of every common field*; per-type `untrusted` in *Event registry* |
|
||||
| Redaction prohibits credentials / raw secrets / unsafe capture by default | *Redaction rule*; `untrusted`-only claims; sensitivity classes |
|
||||
| JSONL journal canonical; SQLite index fully rebuildable | *Journal + SQLite index* (`audit rebuild`) |
|
||||
| Minimum local search/query contract | *Journal + SQLite index* → *Local query surface* |
|
||||
| #324 can transport/replay without a second envelope | *The #324 delivery contract* |
|
||||
| #480 maps in without weakening its byte-to-activation-key guarantee | *#480 signed-commit attribution maps in…* |
|
||||
| Schema evolution + backward-compatible readers | *Schema evolution & backward-compatible readers* |
|
||||
| Integrity detects modification / deletion / reorder / bad continuation | *Test vectors* (tamper), *Behavior across rotation…truncation*, `audit verify` |
|
||||
|
||||
Two acceptance items are **specified here, implemented later** by design
|
||||
(the issue permits this): the concrete test-vector *fixtures* and the
|
||||
`audit` CLI land in impl chunks 2–3; the #324 outbox lands in chunk 6.
|
||||
Nothing in the contract is left undefined — only its code is deferred.
|
||||
|
||||
## Resolved in review (#495)
|
||||
|
||||
- **Single writer per host — decided.** The host controller owns the sole
|
||||
appender; per-producer chains are a future scaling path only. (Design →
|
||||
*Single writer; ordering across restarts*.)
|
||||
- **Restarts — decided.** An `epoch` counter (bumped per writer boot) plus
|
||||
carrying the last chain head as the next `prev` gives a total order of
|
||||
`(epoch, seq)` that survives restarts; `ts_mono` orders only within an
|
||||
epoch. (Design → *ordering across restarts*.)
|
||||
- **Flatten to one `untrusted` region — decided.** Everything outside
|
||||
`untrusted` (chain metadata, `producer`/`host`, `bottled_agent`, `ts_*`)
|
||||
is trusted by construction, so the separate `trusted` sub-block is
|
||||
removed; a field is trusted unless deliberately placed under `untrusted`.
|
||||
(Design → *The envelope*.)
|
||||
- **Subject term is `bottled_agent` everywhere** — the top-level field and
|
||||
the `lifecycle.bottled_agent_*` leaf names. (Design → *The envelope* /
|
||||
*Event registry*.)
|
||||
- **Retention head-carry — yes.** When a journal segment is rotated out,
|
||||
the new segment's first record carries the rotated-out head as `prev`, so
|
||||
the verifier still trusts the current head across a rotation. (Folds into the
|
||||
retention follow-up.)
|
||||
- **Redaction reuses the egress detectors — yes, scoped.** Reuse the
|
||||
deterministic `dlp_detectors` (`scan_token_patterns` / `redact_tokens` /
|
||||
`scan_known_secrets`); exclude `scan_entropy` as brittle on the short,
|
||||
high-entropy structured values audit records carry. (Design → *Redaction
|
||||
rule*.)
|
||||
|
||||
## Open questions
|
||||
|
||||
- **Value-scan cost on the hot path.** The single writer runs the reused
|
||||
detectors on every event's `untrusted` block inline. Is that cheap enough
|
||||
at lifecycle-event volume, or should the value scan move to index-build
|
||||
time (journal stays raw, index stores the redacted view)? Leaning inline
|
||||
so the raw journal never contains a leaked value in the first place.
|
||||
- **`epoch` persistence location.** Store the per-writer `epoch` + chain
|
||||
head in the SQLite index (rebuildable, but then the writer needs the DB
|
||||
at boot) or in a tiny sidecar file next to the journal (independent of
|
||||
the index)? Leaning sidecar, so the writer can start and append without
|
||||
the index present.
|
||||
@@ -4,7 +4,7 @@ Spike branch: `spike/rootless-docker-macos` (`a4d8461`)
|
||||
|
||||
**Outcome:** the podman recommendation below shipped as the
|
||||
`nested_containers` bottle flag — see
|
||||
[`docs/prds/prd-new-nested-containers.md`](../prds/prd-new-nested-containers.md).
|
||||
[`docs/prds/0075-nested-containers.md`](../prds/0075-nested-containers.md).
|
||||
The `docker_access` name used throughout the spike text was renamed on the
|
||||
way in; it granted no access to anything on the host.
|
||||
|
||||
|
||||
+13
-4
@@ -94,13 +94,22 @@ fi
|
||||
|
||||
# --- locate the entry point --------------------------------------------------
|
||||
|
||||
# The pip --user scripts directory is platform-specific: ~/.local/bin on Linux,
|
||||
# but ~/Library/Python/<X.Y>/bin on a python.org macOS interpreter. Ask the
|
||||
# interpreter for its own user-scheme scripts dir instead of hardcoding.
|
||||
USER_SCRIPTS="$(python3 - <<'PY'
|
||||
import sysconfig
|
||||
print(sysconfig.get_path("scripts", sysconfig.get_preferred_scheme("user")))
|
||||
PY
|
||||
)"
|
||||
|
||||
if command -v bot-bottle >/dev/null 2>&1; then
|
||||
BOT_BOTTLE_BIN="bot-bottle"
|
||||
elif [ -x "${HOME}/.local/bin/bot-bottle" ]; then
|
||||
BOT_BOTTLE_BIN="${HOME}/.local/bin/bot-bottle"
|
||||
say "note: add ${HOME}/.local/bin to your PATH to run 'bot-bottle' directly"
|
||||
elif [ -n "${USER_SCRIPTS}" ] && [ -x "${USER_SCRIPTS}/bot-bottle" ]; then
|
||||
BOT_BOTTLE_BIN="${USER_SCRIPTS}/bot-bottle"
|
||||
say "note: add ${USER_SCRIPTS} to your PATH to run 'bot-bottle' directly"
|
||||
else
|
||||
die "bot-bottle was installed but is not on PATH; add ~/.local/bin to PATH and re-run"
|
||||
die "bot-bottle was installed but is not on PATH; add ${USER_SCRIPTS:-your user scripts dir} to PATH and re-run"
|
||||
fi
|
||||
|
||||
# --- verify ------------------------------------------------------------------
|
||||
|
||||
@@ -54,18 +54,20 @@ def check_pull_request(event: dict[str, Any], api: GiteaApi) -> list[str]:
|
||||
pull = event["pull_request"]
|
||||
errors: list[str] = []
|
||||
labels = pull.get("labels") or []
|
||||
if labels:
|
||||
errors.append(
|
||||
"PRs must be unlabeled; put tracker metadata on the linked issue "
|
||||
f"(found: {', '.join(label['name'] for label in labels)})."
|
||||
)
|
||||
|
||||
numbers = deliberate_issue_numbers(pull.get("title", ""), pull.get("body", ""))
|
||||
if not numbers:
|
||||
if labels and numbers:
|
||||
errors.append(
|
||||
"PR must reference an issue with Closes/Fixes/Resolves #N, "
|
||||
"Part of #N, Related to #N, Refs #N, or References #N."
|
||||
"PR must use exactly one tracking mode: remove PR labels when "
|
||||
"linking an issue, or remove the issue reference when labels "
|
||||
"belong on the PR."
|
||||
)
|
||||
if not numbers:
|
||||
if not labels:
|
||||
errors.append(
|
||||
"PR must either have a label or reference an issue with "
|
||||
"Closes/Fixes/Resolves #N, Part of #N, Related to #N, "
|
||||
"Refs #N, or References #N."
|
||||
)
|
||||
return errors
|
||||
|
||||
real_issues = 0
|
||||
|
||||
@@ -67,14 +67,25 @@ _DUMMY_HOST_KEY = (
|
||||
)
|
||||
|
||||
|
||||
# Backends whose CI runner is HOST-mode (self-hosted), so the test process
|
||||
# and the backend share a host. The containerized act_runner (docker on
|
||||
# ubuntu-latest) is the one that can't see the host bind mount egress_tls_init
|
||||
# uses and hides sibling-gateway network topology; host-mode runners
|
||||
# (firecracker/KVM, macos-container) don't have those constraints, so the test
|
||||
# runs there. Keep this in sync with the `runs-on` labels in
|
||||
# .gitea/workflows/test.yml.
|
||||
_HOST_MODE_CI_BACKENDS = frozenset({"firecracker", "macos-container"})
|
||||
|
||||
|
||||
@skip_unless_selected_backend_available()
|
||||
@unittest.skipIf(
|
||||
os.environ.get("GITEA_ACTIONS") == "true"
|
||||
and os.environ.get("BOT_BOTTLE_BACKEND") != "firecracker",
|
||||
"skipped under act_runner unless BOT_BOTTLE_BACKEND=firecracker: "
|
||||
and os.environ.get("BOT_BOTTLE_BACKEND") not in _HOST_MODE_CI_BACKENDS,
|
||||
"skipped under the containerized act_runner (docker on ubuntu-latest): "
|
||||
"egress_tls_init uses a host bind mount the runner container can't "
|
||||
"see, and the network topology hides sibling-gateway visibility — "
|
||||
"these constraints don't apply on the self-hosted KVM runner",
|
||||
"these constraints don't apply on the self-hosted host-mode runners "
|
||||
"(firecracker/KVM, macos-container)",
|
||||
)
|
||||
class TestSandboxEscape(unittest.TestCase):
|
||||
"""End-to-end attacks against a real bottle. The bottle stays
|
||||
|
||||
@@ -2,8 +2,12 @@
|
||||
|
||||
`doctor` is a store-free diagnostic — it must run on a fresh install
|
||||
before any DB migration, and its exit code gates only the two hard
|
||||
prerequisites (Python and an available backend). The config-dir check is
|
||||
advisory and never affects the exit code.
|
||||
prerequisites (Python and at least one *ready* backend). The config-dir
|
||||
check is advisory and never affects the exit code.
|
||||
|
||||
Backend readiness is probed with `is_backend_ready()` (a full status()
|
||||
check), not the cheap PATH-only `is_backend_available()` — a host with a
|
||||
stopped daemon or half-configured backend must not report `ok`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -26,26 +30,43 @@ def _run(argv: list[str] | None = None) -> tuple[int, str]:
|
||||
|
||||
|
||||
class TestDoctor(unittest.TestCase):
|
||||
def test_passes_when_python_and_backend_ok(self):
|
||||
def test_passes_when_python_and_backend_ready(self):
|
||||
with patch.object(doctor, "known_backend_names", return_value=("docker",)), \
|
||||
patch.object(doctor, "is_backend_available", return_value=True):
|
||||
patch.object(doctor, "is_backend_ready", return_value=True):
|
||||
code, out = _run()
|
||||
self.assertEqual(0, code)
|
||||
self.assertIn("ok: python", out)
|
||||
self.assertIn("ok: backend: available: docker", out)
|
||||
self.assertIn("ok: backend: docker: ready", out)
|
||||
|
||||
def test_fails_when_no_backend_available(self):
|
||||
def test_fails_when_no_backend_ready(self):
|
||||
# The regression the reviewer flagged: a backend whose binary is on PATH
|
||||
# but whose daemon/pool isn't ready must NOT pass. is_backend_ready is
|
||||
# the full status() check, so returning False here means "not ready".
|
||||
with patch.object(doctor, "known_backend_names", return_value=("docker", "firecracker")), \
|
||||
patch.object(doctor, "is_backend_available", return_value=False):
|
||||
patch.object(doctor, "is_backend_ready", return_value=False):
|
||||
code, out = _run()
|
||||
self.assertEqual(1, code)
|
||||
self.assertIn("fail: backend", out)
|
||||
self.assertIn("warn: backend: docker: not ready", out)
|
||||
|
||||
def test_passes_when_at_least_one_backend_ready(self):
|
||||
# docker not ready, firecracker ready → overall pass, mixed report.
|
||||
def ready(name: str, *, quiet: bool = False) -> bool:
|
||||
del quiet
|
||||
return name == "firecracker"
|
||||
|
||||
with patch.object(doctor, "known_backend_names", return_value=("docker", "firecracker")), \
|
||||
patch.object(doctor, "is_backend_ready", side_effect=ready):
|
||||
code, out = _run()
|
||||
self.assertEqual(0, code)
|
||||
self.assertIn("warn: backend: docker: not ready", out)
|
||||
self.assertIn("ok: backend: firecracker: ready", out)
|
||||
|
||||
def test_fails_when_python_too_old(self):
|
||||
# Force the version gate to fail without touching the interpreter.
|
||||
with patch.object(doctor, "MIN_PYTHON", (99, 0)), \
|
||||
patch.object(doctor, "known_backend_names", return_value=("docker",)), \
|
||||
patch.object(doctor, "is_backend_available", return_value=True):
|
||||
patch.object(doctor, "is_backend_ready", return_value=True):
|
||||
code, out = _run()
|
||||
self.assertEqual(1, code)
|
||||
self.assertIn("fail: python", out)
|
||||
@@ -57,7 +78,7 @@ class TestDoctor(unittest.TestCase):
|
||||
with tempfile.TemporaryDirectory() as tmp, \
|
||||
patch.object(doctor.Path, "home", return_value=Path(tmp)), \
|
||||
patch.object(doctor, "known_backend_names", return_value=("docker",)), \
|
||||
patch.object(doctor, "is_backend_available", return_value=True):
|
||||
patch.object(doctor, "is_backend_ready", return_value=True):
|
||||
code, out = _run()
|
||||
self.assertEqual(0, code)
|
||||
self.assertIn("warn: config", out)
|
||||
@@ -66,7 +87,7 @@ class TestDoctor(unittest.TestCase):
|
||||
with tempfile.TemporaryDirectory() as tmp, \
|
||||
patch.object(doctor.Path, "home", return_value=Path(tmp)), \
|
||||
patch.object(doctor, "known_backend_names", return_value=("docker",)), \
|
||||
patch.object(doctor, "is_backend_available", return_value=True):
|
||||
patch.object(doctor, "is_backend_ready", return_value=True):
|
||||
(Path(tmp) / ".bot-bottle").mkdir()
|
||||
code, out = _run()
|
||||
self.assertEqual(0, code)
|
||||
|
||||
@@ -19,7 +19,9 @@ _ORCH = "bot_bottle.backend.docker.orchestrator"
|
||||
_RUN = f"{_ORCH}.run_docker"
|
||||
_SLEEP = f"{_ORCH}.time.sleep"
|
||||
_MONOTONIC = f"{_ORCH}.time.monotonic"
|
||||
_TOKEN = f"{_ORCH}.host_orchestrator_token"
|
||||
# The signing key is read through the shared provisioning contract (#476); patch
|
||||
# its host-canonical key file read to keep it off the real host file.
|
||||
_TOKEN = "bot_bottle.trust_domain.host_signing_key"
|
||||
# The ABC's is_healthy probes /health via urllib in the lifecycle module.
|
||||
_URLOPEN = "bot_bottle.orchestrator.lifecycle.urllib.request.urlopen"
|
||||
|
||||
|
||||
@@ -44,7 +44,7 @@ class TestEnsureRunning(unittest.TestCase):
|
||||
vm = infra_vm.InfraVm(guest_ip="10.243.255.1", private_key=Path("/k"), vm=MagicMock())
|
||||
with patch.object(infra_vm, "boot_vm", return_value=vm) as boot, \
|
||||
patch.object(infra_vm, "push_secret") as push, \
|
||||
patch(f"{_ORCH}.host_orchestrator_token", return_value="host-key"), \
|
||||
patch("bot_bottle.trust_domain.host_signing_key", return_value="host-key"), \
|
||||
patch.object(orch, "is_running", return_value=False), \
|
||||
patch.object(orch, "_ensure_registry_volume", return_value=Path("/reg")), \
|
||||
patch.object(orch, "_wait_for_health"):
|
||||
|
||||
@@ -9,6 +9,7 @@ create the config tree, install the package, and verify with `doctor`.
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sysconfig
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
@@ -68,6 +69,33 @@ class TestInstallScript(unittest.TestCase):
|
||||
self.assertIn("EXTERNALLY-MANAGED", self.text)
|
||||
self.assertIn("pipx", self.text)
|
||||
|
||||
def test_resolves_user_scripts_dir_not_hardcoded(self):
|
||||
# The pip --user scripts dir differs by platform; the script must ask
|
||||
# the interpreter (sysconfig + the preferred *user* scheme) rather than
|
||||
# hardcoding Linux's ~/.local/bin (which is wrong on macOS python.org).
|
||||
self.assertIn("get_preferred_scheme", self.text)
|
||||
self.assertIn("sysconfig", self.text)
|
||||
# No hardcoded Linux path in executable lines (a comment may mention it).
|
||||
code = "\n".join(
|
||||
ln for ln in self.text.splitlines()
|
||||
if ln.strip() and not ln.lstrip().startswith("#")
|
||||
)
|
||||
self.assertNotIn(".local/bin", code)
|
||||
|
||||
def test_macos_user_scheme_is_not_dot_local_bin(self):
|
||||
# The case the fix exists for: a python.org macOS interpreter uses the
|
||||
# osx_framework_user scheme, whose scripts land under
|
||||
# ~/Library/Python/<X.Y>/bin — NOT ~/.local/bin. Drive the same
|
||||
# sysconfig lookup install.sh uses, with a mac-like userbase, to prove
|
||||
# it resolves a non-~/.local/bin directory.
|
||||
self.assertIn("osx_framework_user", sysconfig.get_scheme_names())
|
||||
scripts = sysconfig.get_path(
|
||||
"scripts", "osx_framework_user",
|
||||
vars={"userbase": "/Users/dev/Library/Python/3.11"},
|
||||
)
|
||||
self.assertEqual("/Users/dev/Library/Python/3.11/bin", scripts)
|
||||
self.assertNotIn("/.local/bin", scripts)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -33,7 +33,7 @@ class TestMacosOrchestratorRun(unittest.TestCase):
|
||||
def _run(self) -> list[str]:
|
||||
run = Mock(return_value=_ok())
|
||||
with patch(f"{_ORCH}.container_mod") as mod, \
|
||||
patch(f"{_ORCH}.host_orchestrator_token", return_value="k"):
|
||||
patch("bot_bottle.trust_domain.host_signing_key", return_value="k"):
|
||||
mod.dns_server.return_value = "1.1.1.1"
|
||||
mod.bind_mount_spec.side_effect = _spec
|
||||
mod.run_container_argv = run
|
||||
@@ -65,7 +65,7 @@ class TestMacosOrchestratorRun(unittest.TestCase):
|
||||
|
||||
def test_start_failure_raises(self) -> None:
|
||||
with patch(f"{_ORCH}.container_mod") as mod, \
|
||||
patch(f"{_ORCH}.host_orchestrator_token", return_value="k"):
|
||||
patch("bot_bottle.trust_domain.host_signing_key", return_value="k"):
|
||||
mod.dns_server.return_value = "1.1.1.1"
|
||||
mod.run_container_argv = Mock(return_value=_fail())
|
||||
with self.assertRaises(OrchestratorStartError):
|
||||
|
||||
@@ -82,5 +82,26 @@ class TestMintVerify(unittest.TestCase):
|
||||
mint(ROLE_GATEWAY, "")
|
||||
|
||||
|
||||
class TestCustomRoleSet(unittest.TestCase):
|
||||
"""The `roles=` seam a trust domain other than the control plane uses (#476):
|
||||
mint/verify are scoped to the passed role set, not the module default."""
|
||||
|
||||
_ROLES = frozenset({"host"})
|
||||
|
||||
def test_round_trips_a_role_in_the_custom_set(self) -> None:
|
||||
tok = mint("host", _KEY, roles=self._ROLES)
|
||||
self.assertEqual("host", verify(tok, _KEY, roles=self._ROLES))
|
||||
|
||||
def test_mint_rejects_a_role_outside_the_custom_set(self) -> None:
|
||||
with self.assertRaises(ValueError):
|
||||
mint(ROLE_CLI, _KEY, roles=self._ROLES)
|
||||
|
||||
def test_verify_rejects_a_role_outside_the_verifiers_set(self) -> None:
|
||||
# A validly signed token whose role isn't in the verifier's set fails —
|
||||
# this is what keeps one domain's key from asserting another's role.
|
||||
tok = mint(ROLE_CLI, _KEY) # a control-plane `cli` token
|
||||
self.assertIsNone(verify(tok, _KEY, roles=self._ROLES))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -20,13 +20,15 @@ _URLOPEN = "bot_bottle.orchestrator.client.urllib.request.urlopen"
|
||||
|
||||
class TestHostAuthToken(unittest.TestCase):
|
||||
def test_mints_a_cli_token_from_the_host_key(self) -> None:
|
||||
with patch("bot_bottle.orchestrator.client.host_orchestrator_token",
|
||||
# The CLI mints its `cli` token from the control-plane trust domain's
|
||||
# host-canonical key (#476) — patch the key file read underneath it.
|
||||
with patch("bot_bottle.trust_domain.host_signing_key",
|
||||
return_value="signing-key"):
|
||||
tok = _host_auth_token()
|
||||
self.assertEqual(ROLE_CLI, verify(tok, "signing-key"))
|
||||
|
||||
def test_returns_empty_when_key_unreadable(self) -> None:
|
||||
with patch("bot_bottle.orchestrator.client.host_orchestrator_token",
|
||||
with patch("bot_bottle.trust_domain.host_signing_key",
|
||||
side_effect=OSError("no host root")):
|
||||
self.assertEqual("", _host_auth_token())
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
"""Unit tests for per-bottle egress secret encryption (PRD prd-new-secret-provider)."""
|
||||
"""Unit tests for per-bottle egress secret encryption (PRD 0080)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
@@ -109,6 +109,15 @@ class TestWheelMode(unittest.TestCase):
|
||||
(root2 / "bot_bottle" / "cli" / "__init__.py").read_text(),
|
||||
)
|
||||
|
||||
def test_failed_stage_cleans_up_temp_dir(self):
|
||||
# A failure mid-stage must not leave a half-written temp dir behind.
|
||||
with self._wheel():
|
||||
base = resources.bot_bottle_root() / "build-root"
|
||||
with patch.object(resources.shutil, "copytree", side_effect=OSError("boom")):
|
||||
with self.assertRaises(OSError):
|
||||
resources.build_root()
|
||||
self.assertEqual([], list(base.glob(".staging-*")))
|
||||
|
||||
def test_rebuilds_when_stage_incomplete(self):
|
||||
# A crash mid-stage can leave a dir without its `.complete` marker; the
|
||||
# next call must rebuild it rather than trust the partial tree.
|
||||
|
||||
@@ -728,7 +728,7 @@ class TestResolvedRoutesPayload(unittest.TestCase):
|
||||
|
||||
|
||||
class TestNonBlockingSupervise(unittest.TestCase):
|
||||
"""PRD prd-new / issue #412: pending responses carry the proposal id, and
|
||||
"""PRD 0072 / issue #412: pending responses carry the proposal id, and
|
||||
`check-proposal` polls a queued proposal without blocking or re-proposing."""
|
||||
|
||||
_ROUTES = "routes:\n - host: example.com\n"
|
||||
|
||||
@@ -27,7 +27,41 @@ class TestCheckPullRequest(unittest.TestCase):
|
||||
event = {"pull_request": {"title": "Change", "body": "Part of #12", "labels": []}}
|
||||
self.assertEqual(check_pull_request(event, api), [])
|
||||
|
||||
def test_rejects_labels_and_pr_reference(self):
|
||||
def test_accepts_labelled_pr_without_issue(self):
|
||||
api = Mock()
|
||||
event = {
|
||||
"pull_request": {
|
||||
"title": "Change",
|
||||
"body": "",
|
||||
"labels": [{"name": "Kind/Documentation"}],
|
||||
}
|
||||
}
|
||||
self.assertEqual(check_pull_request(event, api), [])
|
||||
api.request.assert_not_called()
|
||||
|
||||
def test_rejects_unlabelled_pr_without_issue(self):
|
||||
api = Mock()
|
||||
event = {"pull_request": {"title": "Change", "body": "", "labels": []}}
|
||||
errors = check_pull_request(event, api)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("either have a label or reference an issue", errors[0])
|
||||
api.request.assert_not_called()
|
||||
|
||||
def test_rejects_labelled_pr_linked_to_real_issue(self):
|
||||
api = Mock()
|
||||
api.request.return_value = {"number": 12, "pull_request": None}
|
||||
event = {
|
||||
"pull_request": {
|
||||
"title": "Change",
|
||||
"body": "Closes #12",
|
||||
"labels": [{"name": "Kind/Documentation"}],
|
||||
}
|
||||
}
|
||||
errors = check_pull_request(event, api)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("exactly one tracking mode", errors[0])
|
||||
|
||||
def test_still_validates_issue_reference_when_both_modes_are_used(self):
|
||||
api = Mock()
|
||||
api.request.return_value = {"number": 12, "pull_request": {}}
|
||||
event = {
|
||||
@@ -39,7 +73,7 @@ class TestCheckPullRequest(unittest.TestCase):
|
||||
}
|
||||
errors = check_pull_request(event, api)
|
||||
self.assertEqual(len(errors), 2)
|
||||
self.assertIn("unlabeled", errors[0])
|
||||
self.assertIn("exactly one tracking mode", errors[0])
|
||||
self.assertIn("not an issue", errors[1])
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,105 @@
|
||||
"""Unit: trust domains + the shared control-plane provisioning contract (#476)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
from bot_bottle import orchestrator_auth
|
||||
from bot_bottle.orchestrator_auth import ROLE_CLI, ROLE_GATEWAY
|
||||
from bot_bottle.trust_domain import (
|
||||
CONTROL_PLANE,
|
||||
ControlPlaneProvisioning,
|
||||
ProvisioningError,
|
||||
TrustDomain,
|
||||
)
|
||||
|
||||
# A second, unrelated domain — the shape #468's host controller would take: its
|
||||
# own key file, its own role, its own env vars. Distinct from CONTROL_PLANE.
|
||||
_HOST_CTRL = TrustDomain(
|
||||
name="host-controller",
|
||||
key_filename="host-controller-token",
|
||||
roles=frozenset({"host"}),
|
||||
key_env="BOT_BOTTLE_HOST_CONTROLLER_TOKEN",
|
||||
token_env="BOT_BOTTLE_HOST_CONTROLLER_JWT",
|
||||
)
|
||||
|
||||
|
||||
class TestTrustDomainMintVerify(unittest.TestCase):
|
||||
def test_mint_verify_round_trips_within_a_domain(self) -> None:
|
||||
with patch("bot_bottle.trust_domain.host_signing_key", return_value="k"):
|
||||
tok = CONTROL_PLANE.mint(ROLE_GATEWAY)
|
||||
self.assertEqual(ROLE_GATEWAY, CONTROL_PLANE.verify(tok, "k"))
|
||||
|
||||
def test_mint_rejects_a_role_outside_the_domain(self) -> None:
|
||||
# `host` is a valid role in _HOST_CTRL but not in the control plane —
|
||||
# the control-plane key must refuse to mint it.
|
||||
with patch("bot_bottle.trust_domain.host_signing_key", return_value="k"):
|
||||
with self.assertRaises(ValueError):
|
||||
CONTROL_PLANE.mint("host")
|
||||
|
||||
def test_a_custom_role_set_verifies_its_own_role(self) -> None:
|
||||
with patch("bot_bottle.trust_domain.host_signing_key", return_value="k"):
|
||||
tok = _HOST_CTRL.mint("host")
|
||||
self.assertEqual("host", _HOST_CTRL.verify(tok, "k"))
|
||||
|
||||
def test_key_from_env_reads_the_domains_env_var(self) -> None:
|
||||
self.assertEqual(
|
||||
"secret", CONTROL_PLANE.key_from_env({CONTROL_PLANE.key_env: " secret "}))
|
||||
self.assertEqual("", CONTROL_PLANE.key_from_env({}))
|
||||
|
||||
|
||||
class TestDomainBoundary(unittest.TestCase):
|
||||
"""The #476/#468 invariant: two domains, two keys, two role sets — one
|
||||
domain's key can neither mint nor verify the other's tokens."""
|
||||
|
||||
def test_a_control_plane_token_does_not_verify_under_another_domains_key(
|
||||
self,
|
||||
) -> None:
|
||||
# Even with an IDENTICAL underlying key, a `cli` token minted under the
|
||||
# control-plane domain must not verify as a role in the host-controller
|
||||
# domain — the role isn't in that domain's set.
|
||||
cli_tok = orchestrator_auth.mint(ROLE_CLI, "shared-bytes")
|
||||
self.assertIsNone(_HOST_CTRL.verify(cli_tok, "shared-bytes"))
|
||||
|
||||
def test_distinct_keys_do_not_cross_verify(self) -> None:
|
||||
# The realistic case: distinct host-canonical keys per domain. A token
|
||||
# signed by one key never verifies under the other.
|
||||
host_tok = orchestrator_auth.mint(
|
||||
"host", "host-ctrl-key", roles=_HOST_CTRL.roles)
|
||||
self.assertIsNone(_HOST_CTRL.verify(host_tok, "control-plane-key"))
|
||||
self.assertEqual("host", _HOST_CTRL.verify(host_tok, "host-ctrl-key"))
|
||||
|
||||
|
||||
class TestControlPlaneProvisioning(unittest.TestCase):
|
||||
def test_orchestrator_key_returns_the_canonical_key(self) -> None:
|
||||
prov = ControlPlaneProvisioning()
|
||||
with patch("bot_bottle.trust_domain.host_signing_key", return_value="key"):
|
||||
self.assertEqual("key", prov.orchestrator_key())
|
||||
|
||||
def test_orchestrator_key_fail_closes_when_empty(self) -> None:
|
||||
# Invariant 4: the orchestrator must never start without a key — it would
|
||||
# run OPEN and grant every caller that reaches it full `cli`. There is no
|
||||
# topology opt-out: a separate host does not stop the gateway (or any
|
||||
# other caller) from reaching the control-plane listener.
|
||||
prov = ControlPlaneProvisioning()
|
||||
with patch("bot_bottle.trust_domain.host_signing_key", return_value=""):
|
||||
with self.assertRaises(ProvisioningError):
|
||||
prov.orchestrator_key()
|
||||
|
||||
def test_gateway_token_is_a_verifiable_gateway_role_token(self) -> None:
|
||||
prov = ControlPlaneProvisioning()
|
||||
with patch("bot_bottle.trust_domain.host_signing_key", return_value="k"):
|
||||
tok = prov.gateway_token()
|
||||
self.assertEqual(ROLE_GATEWAY, CONTROL_PLANE.verify(tok, "k"))
|
||||
|
||||
def test_gateway_token_cannot_be_reused_as_cli(self) -> None:
|
||||
# The data plane's token is `gateway`-scoped: it never carries `cli`.
|
||||
prov = ControlPlaneProvisioning()
|
||||
with patch("bot_bottle.trust_domain.host_signing_key", return_value="k"):
|
||||
tok = prov.gateway_token()
|
||||
self.assertNotEqual(ROLE_CLI, CONTROL_PLANE.verify(tok, "k"))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user