From a928d5eec3d0cbfb7b4a54993ee2dd4c29248730 Mon Sep 17 00:00:00 2001 From: Christian Manivong Date: Mon, 28 Sep 2026 19:55:22 +0200 Subject: [PATCH] Initial import of the netOrk deploy script --- .gitea/workflows/ci.yml | 31 +++ .gitignore | 5 + LICENSE | 21 ++ README.md | 112 ++++++++ deploy.env.example | 24 ++ deploy.sh | 329 +++++++++++++++++++++++ pyproject.toml | 6 + tests/test_deploy_script_bootstrap.py | 114 ++++++++ tests/test_deploy_script_fails_loudly.py | 131 +++++++++ 9 files changed, 773 insertions(+) create mode 100644 .gitea/workflows/ci.yml create mode 100644 .gitignore create mode 100644 LICENSE create mode 100644 README.md create mode 100644 deploy.env.example create mode 100644 deploy.sh create mode 100644 pyproject.toml create mode 100644 tests/test_deploy_script_bootstrap.py create mode 100644 tests/test_deploy_script_fails_loudly.py diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml new file mode 100644 index 0000000..d4a6ebb --- /dev/null +++ b/.gitea/workflows/ci.yml @@ -0,0 +1,31 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + +jobs: + check: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + + - name: Install tools + # Pinned: a newer ruff or shellcheck can turn an untouched tree red. + run: pip install pytest==9.1.1 ruff==0.16.9 shellcheck-py==0.11.0.1 + + - name: shellcheck + run: shellcheck deploy.sh + + - name: ruff + run: | + ruff format --check . + ruff check . + + - name: Tests + run: pytest diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..89cdf7d --- /dev/null +++ b/.gitignore @@ -0,0 +1,5 @@ +deploy.env +__pycache__/ +.pytest_cache/ +.ruff_cache/ +.venv/ diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..ac8e1dd --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Christian Manivong + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.md b/README.md new file mode 100644 index 0000000..17173af --- /dev/null +++ b/README.md @@ -0,0 +1,112 @@ +# netOrk deploy + +Deploys [netOrk](https://git.netork.io/NetOrk/netork) to one or more Docker hosts in +parallel, from pre-built images in a container registry. + +The machine running the deploy needs this repository, `ssh`, and a `deploy.env`. It does +not need a netOrk checkout. Each server pulls the images itself. The compose files come +out of the engine image for the tag being deployed, so they always match the images they +start, and database migrations run from that same image. + +## Requirements + +**On the machine running the deploy:** +- bash +- `ssh` with key-based access to every target + +**On every target server:** +- Docker with the compose plugin +- a directory `~/netork/` with netOrk's `.env`, which holds the database and application + settings the compose file reads (`env_file: .env`) + +**Images:** `netork/engine` and `netork/ui` must be published in the registry under the +tag you deploy. The engine image must carry its compose files under `/app/deploy/`. +Tags built before that change are refused with a clear message. + +## Setup + +```bash +cp deploy.env.example deploy.env +$EDITOR deploy.env +``` + +`deploy.env` is gitignored. Point `DEPLOY_ENV_FILE` at another file to keep it +elsewhere. + +| Variable | Meaning | +|---|---| +| `DEPLOY_SERVERS` | Space-separated default targets | +| `UI_SERVER` | Targets that also run `netork-ui` | +| `REGISTRY_HOST` | Registry to pull from; required | +| `NETORK_VERSION` | Tag to deploy (default `latest`) | +| `REGISTRY_USER`, `REGISTRY_PASSWORD` | Registry login used on every server | +| `REGISTRY_USER_`, `REGISTRY_PASSWORD_`, `NETORK_VERSION_` | Per-server overrides. `` has its dots replaced by underscores, e.g. `_10_0_0_2` | + +## Usage + +```bash +./deploy.sh # every server in DEPLOY_SERVERS +./deploy.sh 10.0.0.1 # one server +./deploy.sh 10.0.0.1 10.0.0.2 # several, in parallel +./deploy.sh --version=main-1a2b3c4 10.0.0.1 # one-off tag, this run only +``` + +Only tags netOrk's CI publishes are accepted: +- `latest` +- `latest-dev` +- `main-` +- `feature-` +- `X.Y.Z` + +Anything else is refused before any server is touched. + +If you deploy a branch build, wait until its CI run has published the images. A deploy +started earlier pulls whatever image the registry held before, which is stale. + +## What a deploy does, per server + +1. Logs in to the registry. The password travels over ssh's stdin, never on a command + line. +2. Pulls the engine image and copies `docker-compose.yml` and + `docker-compose.registry.yml` out of it into `~/netork/`. +3. Pulls the netOrk images, then force-recreates the API, the workers, `netork-beat`, + `flower` and, on UI servers, `netork-ui`. +4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its + definition changed. It never touches `postgres` or `redis`. +5. Verifies that the containers run exactly the image that was pulled. If they don't, the + deploy fails. +6. Records `REGISTRY_HOST` and `NETORK_VERSION` in `~/netork/.env`, so that a + hand-typed `docker compose` on the host uses the same images. +7. Runs `alembic upgrade head` inside `netork-api`. +8. Prunes unused images. + +A failing step stops that server's deploy with a non-zero exit, and the other servers +carry on. The script exits non-zero if any server failed and names those servers. + +## Operating notes + +- **Never delete these Docker volumes:** + - `celerybeat_schedule` holds the scheduler state. + - `signal_cli_data` holds netOrk's Signal device link. Losing it means pairing again + by QR code. +- `netork-beat` always rolls out together with the workers. The script does this for + you; keep it that way if you deploy by hand. +- Always deploy with this script. Do not point a local Docker client at a remote host + (`DOCKER_HOST=ssh://…`): it resolves volume paths locally and breaks the remote + containers. + +## Development + +```bash +pip install pytest ruff shellcheck-py +shellcheck deploy.sh +ruff format --check . && ruff check . +pytest +``` + +The tests are static and behavioural checks of `deploy.sh`. None of them reach a real +host. + +## License + +MIT, see [LICENSE](LICENSE). diff --git a/deploy.env.example b/deploy.env.example new file mode 100644 index 0000000..2336256 --- /dev/null +++ b/deploy.env.example @@ -0,0 +1,24 @@ +# Copy to deploy.env (next to deploy.sh) and fill in. deploy.env is gitignored. + +# Space-separated list of servers to deploy to when none are given on the +# command line. Anything ssh accepts: an IP, a hostname, an ssh config alias. +DEPLOY_SERVERS="10.0.0.1 10.0.0.2" + +# Servers that also run the netork-ui container. +UI_SERVER="10.0.0.1" + +# Registry the servers pull netOrk's images from. +REGISTRY_HOST=registry.example.com + +# Image tag to deploy: latest, latest-dev, main-, feature- or X.Y.Z. +NETORK_VERSION=latest + +# Registry credentials used on every server without its own entry below. +REGISTRY_USER= +REGISTRY_PASSWORD= + +# Per-server overrides. The suffix is the server as written in DEPLOY_SERVERS, +# with dots replaced by underscores. +REGISTRY_USER_10_0_0_2= +REGISTRY_PASSWORD_10_0_0_2= +NETORK_VERSION_10_0_0_2=latest-dev diff --git a/deploy.sh b/deploy.sh new file mode 100644 index 0000000..52a0911 --- /dev/null +++ b/deploy.sh @@ -0,0 +1,329 @@ +#!/usr/bin/env bash +# Deploy netOrk to one or more servers in parallel. +# +# Usage: +# ./deploy.sh [--version=] [server1 server2 ...] +# +# --version overrides the per-server NETORK_VERSION from deploy.env for this +# run only — useful to test one commit (--version=main-) without moving +# a server to a different release channel. +# +# Default targets and registry credentials are read from deploy.env next to +# this script (see deploy.env.example). DEPLOY_ENV_FILE points elsewhere. +# +# The servers pull pre-built images from REGISTRY_HOST. Nothing is built or +# copied from a working tree: the compose files come out of the engine image +# for the tag being deployed, and migrations run from that same image. +set -euo pipefail + +HERE="$(cd "$(dirname "$0")" && pwd)" + +ENV_FILE="${DEPLOY_ENV_FILE:-${HERE}/deploy.env}" +if [[ -f "$ENV_FILE" ]]; then + # shellcheck source=/dev/null + source "$ENV_FILE" +fi + +DEPLOY_SERVERS="${DEPLOY_SERVERS:-}" +UI_SERVER="${UI_SERVER:-}" +REGISTRY_HOST="${REGISTRY_HOST:-}" +REGISTRY_USER="${REGISTRY_USER:-}" +REGISTRY_PASSWORD="${REGISTRY_PASSWORD:-}" +NETORK_VERSION="${NETORK_VERSION:-latest}" + +# The compose files the engine image carries under /app/deploy/. +COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml" + +read -ra SERVERS <<< "$DEPLOY_SERVERS" + +# Run one step on a server, and let its failure stop the deploy. +# +# ssh reports the exit status of the *remote* command — and when that command +# ends in a pipe, the status is the last stage's, not the work's: +# +# ssh "$SERVER" "docker compose pull … 2>&1 | grep -E 'Pulled|Already|Error'" +# +# A failed pull prints a line containing `Error`, grep matches it and exits 0, +# and the deploy walks on to swap the containers. `set -o pipefail` cannot help: +# that pipeline ran in the remote shell, and from here the ssh succeeded. +# +# So the remote side stays pipe-free and the filtering happens here, where the +# status is still the one the command returned. +# +# run_remote +run_remote() { + local server=$1 label=$2 pattern=$3 command=$4 + local output status + + output="$(ssh -n "$server" "$command" 2>&1)" && status=0 || status=$? + + if [[ $status -ne 0 ]]; then + printf '%s\n' "$output" | tail -25 | sed "s/^/[${server}] /" >&2 + echo "[${server}] FAILED (exit ${status}): ${label}" >&2 + return "$status" + fi + + if [[ -n "$pattern" ]]; then + printf '%s\n' "$output" | grep -E "$pattern" | sed "s/^/[${server}] /" || true + else + printf '%s\n' "$output" | tail -6 | sed "s/^/[${server}] /" + fi +} + +VERSION_OVERRIDE="" +EXPLICIT_SERVERS=() +i=1 +while [[ $i -le $# ]]; do + arg="${!i}" + case "$arg" in + --version) + i=$(( i + 1 )) + VERSION_OVERRIDE="${!i:-}" + ;; + --version=*) VERSION_OVERRIDE="${arg#--version=}" ;; + -*) + echo "Unknown option: ${arg}" >&2 + exit 2 + ;; + *) EXPLICIT_SERVERS+=("$arg") ;; + esac + i=$(( i + 1 )) +done +if [[ ${#EXPLICIT_SERVERS[@]} -gt 0 ]]; then + SERVERS=("${EXPLICIT_SERVERS[@]}") +fi + +# ── Version validation ─────────────────────────────────────────────────────── +# A deploy only pulls images — it never verifies the pulled tag actually +# corresponds to a real, recently-built image for the commit/branch you think +# you're deploying. A stale or accidentally-reused tag (or a typo'd branch +# name that happens to collide with some ancient leftover tag) is pulled +# silently, with the only symptom surfacing later at `alembic upgrade head`, +# by which point containers are already switched over. Reject anything that +# doesn't match a tag format netOrk's CI actually publishes: latest, +# latest-dev, main-, feature- (feature/* only), or a release +# version (X.Y.Z). See NetOrk/netork#55. +VALID_VERSION_RE='^(latest|latest-dev|main-[0-9a-f]{7,40}|feature-[A-Za-z0-9._-]+|[0-9]+\.[0-9]+\.[0-9]+)$' + +_validate_version() { + local version="$1" + if [[ ! "$version" =~ $VALID_VERSION_RE ]]; then + echo "Refusing to deploy version '${version}': not a tag CI publishes (expected latest," \ + "latest-dev, main-, feature-, or X.Y.Z)." >&2 + return 1 + fi +} + +if [[ -n "$VERSION_OVERRIDE" ]]; then + _validate_version "$VERSION_OVERRIDE" +fi + +if [[ -z "$REGISTRY_HOST" ]]; then + echo "REGISTRY_HOST is not set (deploy.env or environment): nowhere to pull images from." >&2 + exit 1 +fi + +if [[ ${#SERVERS[@]} -eq 0 ]]; then + echo "No servers given: pass them as arguments or set DEPLOY_SERVERS in deploy.env." >&2 + exit 1 +fi + +deploy_server() { + local SERVER="$1" + + # Per-server lookup — falls back to global values + local KEY="${SERVER//./_}" + local USER_VAR="REGISTRY_USER_${KEY}" + local PASS_VAR="REGISTRY_PASSWORD_${KEY}" + local VER_VAR="NETORK_VERSION_${KEY}" + local SERVER_USER="${!USER_VAR:-${REGISTRY_USER}}" + local SERVER_PASS="${!PASS_VAR:-${REGISTRY_PASSWORD}}" + local SERVER_VERSION="${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}" + _validate_version "$SERVER_VERSION" || return 1 + + local ENGINE_IMAGE="${REGISTRY_HOST}/netork/engine:${SERVER_VERSION}" + + # The password goes over ssh's stdin. Inside the ssh argument it would be + # part of the remote shell's command line, readable through `ps` by every + # user on the host for as long as the login runs. + echo "[${SERVER}] Logging in to registry..." + # shellcheck disable=SC2029 # host and user are meant to expand here + printf '%s\n' "$SERVER_PASS" \ + | ssh "$SERVER" "docker login '${REGISTRY_HOST}' -u '${SERVER_USER}' --password-stdin 2>&1" \ + | sed "s/^/[${SERVER}] /" + + # The compose files are versioned with the images they start, so they are + # taken from the engine image of this very tag rather than from whatever + # checkout happens to run the deploy. Both are written to .new first and only + # moved into place once both were read, so a failure leaves the previous pair + # intact. An image built before the files moved into it has no /app/deploy/. + echo "[${SERVER}] Fetching compose files from ${ENGINE_IMAGE}..." + run_remote "$SERVER" "fetching compose files from ${ENGINE_IMAGE}" 'Status|Error' \ + "mkdir -p ~/netork && cd ~/netork && \ + docker pull '${ENGINE_IMAGE}' && \ + for f in ${COMPOSE_FILES}; do \ + docker run --rm --entrypoint cat '${ENGINE_IMAGE}' /app/deploy/\$f > \$f.new \ + || { rm -f \$f.new; echo \"${ENGINE_IMAGE} carries no /app/deploy/\$f — the tag predates compose files in the engine image\"; exit 3; }; \ + done && \ + for f in ${COMPOSE_FILES}; do mv \$f.new \$f; done" + + # Services running netOrk's own images. flower belongs here: it is the engine + # image with a different command, so leaving it out let the monitoring UI run + # nine days behind the workers it monitors, on an image that had since been + # untagged. See NetOrk/netork#95. + local SERVICES="netork-api netork-worker netork-poll-worker netork-ansible-worker netork-beat flower" + if [[ " $UI_SERVER " == *" $SERVER "* ]]; then + SERVICES="${SERVICES} netork-ui" + fi + + # Containers on upstream images. Reconciled, not force-recreated (below). + # postgres and redis are deliberately absent: restarting a database on every + # deploy would be worse than any compose change one could miss. + local INFRA_SERVICES="registry apt-cacher-ng signal-api" + + echo "[${SERVER}] Pulling images..." + run_remote "$SERVER" "pulling images for ${SERVER_VERSION}" 'Pulled|Already|Error' \ + "cd ~/netork && \ + REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ + docker compose -f docker-compose.yml -f docker-compose.registry.yml pull ${SERVICES}" + + # --force-recreate is deliberate: `docker compose up -d` decides what to + # replace from the service definition's config hash, and the image reference + # (`.../ui:latest-dev`) does not change when the tag is moved to a new image. + # A deploy then leaves the old container running while reporting success — + # observed on netork-ui, which kept serving a stale image after its tag had + # already advanced. Recreating unconditionally costs a restart per deploy, + # which a deliberate rollout wants anyway. See NetOrk/netork#90. + echo "[${SERVER}] Starting containers..." + run_remote "$SERVER" "recreating ${SERVICES}" "" \ + "cd ~/netork && \ + REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ + docker compose -f docker-compose.yml -f docker-compose.registry.yml \ + up -d --no-deps --force-recreate ${SERVICES}" + + # Infrastructure containers, deliberately left out of the force-recreate + # above. Plain `up -d`: compose compares each service definition against the + # running container and recreates only what actually changed, so an edit to a + # healthcheck or an option in docker-compose.yml takes effect while a + # container nobody touched is left alone. signal-api is the reason this is + # not --force-recreate: it holds netOrk's Signal device link and recreating + # it every deploy is churn nobody asked for. See NetOrk/netork#95. + # + # `|| true`: infrastructure is reconciled opportunistically and its failure + # is not a reason to abandon a deploy that has already succeeded — but it + # prints as a failure instead of scrolling past. + echo "[${SERVER}] Reconciling infrastructure containers..." + run_remote "$SERVER" "reconciling ${INFRA_SERVICES}" "" \ + "cd ~/netork && \ + REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ + docker compose -f docker-compose.yml -f docker-compose.registry.yml \ + up -d --no-deps ${INFRA_SERVICES}" || true + + # Verify the containers actually run the image we just pulled. `docker + # compose up -d` reports "Running" (not "Started") when it decides nothing + # changed, and `pull` prints "Pulled" even when the tag was already local — + # so a deploy that had no effect at all is indistinguishable from a real one + # in the output above. Both the wrong-tag case (NETORK_VERSION pointing at a + # tag CI never republished) and a silently failed pull land here. + # See NetOrk/netork#88. + echo "[${SERVER}] Verifying containers run the pulled image..." + local VERIFY_PAIRS="netork-api:engine flower:engine" + if [[ " $UI_SERVER " == *" $SERVER "* ]]; then + VERIFY_PAIRS="${VERIFY_PAIRS} netork-ui:ui" + fi + + local MISMATCH + MISMATCH=$(ssh -n "$SERVER" "cd ~/netork && for pair in ${VERIFY_PAIRS}; do \ + svc=\${pair%%:*}; repo=\${pair##*:}; \ + want=\$(docker image inspect --format '{{.Id}}' '${REGISTRY_HOST}/netork/'\${repo}':${SERVER_VERSION}' 2>/dev/null || true); \ + cid=\$(REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ + docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -q \${svc} 2>/dev/null || true); \ + got=\$(docker inspect --format '{{.Image}}' \${cid} 2>/dev/null || true); \ + if [ -z \"\${want}\" ] || [ -z \"\${got}\" ] || [ \"\${want}\" != \"\${got}\" ]; then echo \${svc}; fi; \ + done" || true) + + if [[ -n "$MISMATCH" ]]; then + echo "[${SERVER}] ERROR: still running an older image after deploy: ${MISMATCH//$'\n'/ }" >&2 + echo "[${SERVER}] Wanted tag '${SERVER_VERSION}'. On main pushes CI publishes" >&2 + echo "[${SERVER}] 'latest-dev' and 'main-'; ':latest' only on version tags." >&2 + return 1 + fi + echo "[${SERVER}] Verified: containers run ${SERVER_VERSION}." + + # docker compose substitutes REGISTRY_HOST / NETORK_VERSION from the project + # .env when they are not in the environment. This script passes both inline, + # but a hand-typed `docker compose up -d flower` on the host needs them there + # or dies on "invalid reference format". Values are rewritten in place, so + # the file does not grow a new line per deploy. + # + # Only now. This used to run before the pull, so a deploy that failed on a tag + # the registry does not have still left the host recording that tag — and the + # next `docker compose up` a human ran there reached for an image that does + # not exist. The file is a record of what this server runs, so it is written + # once that is true: after the swap and after the image-id check agreed. + echo "[${SERVER}] Recording registry settings in ~/netork/.env..." + # shellcheck disable=SC2087 # registry and version expand here; \$ escapes run remotely + ssh "$SERVER" "bash -s" <> .env +for kv in "REGISTRY_HOST=${REGISTRY_HOST}" "NETORK_VERSION=${SERVER_VERSION}"; do + key=\${kv%%=*} + if grep -q "^\${key}=" .env; then + sed -i "s|^\${key}=.*|\${kv}|" .env + else + printf '%s\n' "\${kv}" >> .env + fi +done +REMOTE_ENV + + # netOrk migrates *after* the swap — its API tolerates the gap. What it does + # not tolerate is the migration failing unnoticed: the new code is already + # serving, so a missing column is a 500 on every request that touches it + # until someone looks. Loud, and non-zero. The revisions are the image's own. + echo "[${SERVER}] Running migrations..." + run_remote "$SERVER" "alembic upgrade head (containers are ALREADY swapped)" "" \ + "cd ~/netork && docker compose exec -T netork-api alembic upgrade head" + + echo "[${SERVER}] Pruning unused images..." + ssh -n "$SERVER" "docker image prune -af 2>&1 | tail -1" | sed "s/^/[${SERVER}] /" + + echo "[${SERVER}] Done." +} + +# ── Dispatch ────────────────────────────────────────────────────────────────── + +echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}" + +export -f deploy_server _validate_version run_remote +export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \ + VALID_VERSION_RE COMPOSE_FILES +# Export all per-server credential vars so subshells can resolve them +while IFS='=' read -r key _; do + if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then + export "${key?}" + fi +done < <(compgen -v) + +PIDS=() +for SERVER in "${SERVERS[@]}"; do + deploy_server "$SERVER" & + PIDS+=($!) +done + +FAILED=() +for i in "${!PIDS[@]}"; do + if ! wait "${PIDS[$i]}"; then + FAILED+=("${SERVERS[$i]}") + fi +done + +if [[ ${#FAILED[@]} -gt 0 ]]; then + echo "FAILED on: ${FAILED[*]}" >&2 + exit 1 +fi + +echo "All servers deployed successfully." diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..607d94d --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,6 @@ +[tool.ruff] +line-length = 100 +target-version = "py312" + +[tool.pytest.ini_options] +testpaths = ["tests"] diff --git a/tests/test_deploy_script_bootstrap.py b/tests/test_deploy_script_bootstrap.py new file mode 100644 index 0000000..53f39e2 --- /dev/null +++ b/tests/test_deploy_script_bootstrap.py @@ -0,0 +1,114 @@ +"""What a deploy needs from the host running it: this script, ssh, and deploy.env. + +No checkout of the application repository. The compose files come out of the +engine image for the tag being deployed, and migrations run from that same image, +so nothing is rsynced from a working tree any more. +""" + +from __future__ import annotations + +import os +import subprocess +from pathlib import Path + +import pytest + +DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh" + + +def _logical_lines() -> list[str]: + joined = DEPLOY.read_text().replace("\\\n", " ") + return [line for line in joined.splitlines() if not line.lstrip().startswith("#")] + + +def _position(needle: str) -> int: + text = DEPLOY.read_text() + assert text.count(needle) == 1, f"{needle!r} appears {text.count(needle)} times" + return text.index(needle) + + +@pytest.fixture +def run(tmp_path: Path): + """Run the script with no deploy.env and an ssh that records being reached.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + marker = tmp_path / "ssh-was-called" + fake = bin_dir / "ssh" + fake.write_text(f"#!/bin/sh\ntouch {marker}\nexit 255\n") + fake.chmod(0o755) + + def _run(*args: str, **env: str) -> subprocess.CompletedProcess[str]: + base = { + "PATH": f"{bin_dir}:{os.environ['PATH']}", + "HOME": str(tmp_path), + "DEPLOY_ENV_FILE": "/dev/null", + } + result = subprocess.run( + ["bash", str(DEPLOY), *args], + env={**base, **env}, + capture_output=True, + text=True, + timeout=30, + check=False, + ) + assert not marker.exists(), "the script reached ssh before refusing" + return result + + return _run + + +class TestNoCheckoutNeeded: + def test_nothing_is_rsynced(self) -> None: + offenders = [line.strip() for line in _logical_lines() if "rsync" in line] + assert not offenders, f"deploy still copies files from a working tree: {offenders}" + + def test_compose_files_come_from_the_engine_image(self) -> None: + assert "/app/deploy/" in DEPLOY.read_text() + + def test_compose_files_are_fetched_before_the_pull(self) -> None: + assert _position("Fetching compose files") < _position("Pulling images...") + + +class TestTheRegistryPasswordStaysOffCommandLines: + """Anything inside the ssh argument becomes the remote shell's command line, + readable by every user on the host through `ps`. The password travels over + ssh's stdin instead.""" + + def test_no_ssh_argument_carries_the_password(self) -> None: + offenders = [ + line.strip() + for line in _logical_lines() + if "ssh " in line and "PASS" in line.split("ssh ", 1)[1] + ] + assert not offenders, f"password on a remote command line: {offenders}" + + def test_login_reads_the_password_from_stdin(self) -> None: + assert any("--password-stdin" in line and "| ssh " in line for line in _logical_lines()), ( + "docker login no longer gets the password piped through ssh" + ) + + +class TestRefusesBeforeTouchingAHost: + """Every one of these must stop before the first ssh — the fixture's fake ssh + fails the test if it is reached.""" + + def test_an_unpublished_version_is_refused(self, run) -> None: + r = run("--version=bogus", "host", REGISTRY_HOST="registry.example") + assert r.returncode != 0 + assert "Refusing to deploy version 'bogus'" in r.stderr + + def test_no_registry_is_refused(self, run) -> None: + r = run("host") + assert r.returncode != 0 + assert "REGISTRY_HOST" in r.stderr + + def test_no_servers_is_refused(self, run) -> None: + r = run(REGISTRY_HOST="registry.example") + assert r.returncode != 0 + assert "No servers" in r.stderr + + @pytest.mark.parametrize("flag", ["--no-cache", "--bogus"]) + def test_an_unknown_flag_is_refused(self, run, flag: str) -> None: + r = run(flag, "host", REGISTRY_HOST="registry.example") + assert r.returncode != 0 + assert f"Unknown option: {flag}" in r.stderr diff --git a/tests/test_deploy_script_fails_loudly.py b/tests/test_deploy_script_fails_loudly.py new file mode 100644 index 0000000..d77c5ea --- /dev/null +++ b/tests/test_deploy_script_fails_loudly.py @@ -0,0 +1,131 @@ +"""A load-bearing deploy step must not report success when it failed. + +`deploy.sh` runs its real work over ssh. The exit status ssh hands back +is the status of the *remote* command — and when that command ends in a pipe, +the status belongs to the last stage of the pipe, not to the work: + + ssh "$SERVER" "docker compose pull … 2>&1 | grep -E 'Pulled|Already|Error'" + +A failed pull prints a line containing `Error`, `grep` matches it and exits 0, and +the deploy walks on to swap the containers. `set -o pipefail` cannot see this: the +remote pipeline ran inside the remote shell and from here the ssh succeeded. + +A sibling service hit the same shape on 2026-09-15 with +`alembic upgrade head 2>&1 | tail -5`. The migration raised a TypeError, the +script printed the traceback and swapped the containers anyway, and the service +answered 500 on every request touching the changed table for eleven minutes. + +So: tail and grep locally, on captured output, where `pipefail` and `set -e` can +still act. The rule below is deliberately narrow — it covers only the steps whose +failure must stop a deploy. `docker image prune | tail -1` is cosmetic and stays +exempt, because a rule that forbids every pipe would be worked around rather than +followed. +""" + +from __future__ import annotations + +import re +from pathlib import Path + +DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh" + +#: The steps a deploy must not walk past. Anything else may pipe as it likes. +LOAD_BEARING = re.compile( + r"docker pull|compose[^\n]*\bpull\b|compose[^\n]*\bup -d\b|alembic upgrade|/app/deploy/" +) + +#: Filters that replace the exit status of everything before them. +SWALLOWS_STATUS = re.compile(r"\|\s*(tail|head|grep)\b") + +#: Every ssh line ends `| sed "s/^/[${SERVER}] /"`, which runs *here*. Local +#: pipes are covered by `set -o pipefail`; only what runs on the far side is +#: invisible, so the local suffix is cut off before the check. +LOCAL_SUFFIX = '| sed "s/^/' + + +def _logical_lines() -> list[str]: + """The script with backslash-continuations joined, so one command is one line. + + Comments are dropped. The first version of this check flagged the comment + inside `run_remote` that quotes the offending line in order to explain it — + prose about a statement is not the statement, and a checker that cannot tell + them apart makes the explanation unwritable. + """ + joined = DEPLOY.read_text().replace("\\\n", " ") + return [line for line in joined.splitlines() if not line.lstrip().startswith("#")] + + +def _remote_part(line: str) -> str: + head, _, _ = line.partition(LOCAL_SUFFIX) + return head + + +def _offenders() -> list[str]: + return [ + " ".join(line.split())[:120] + for line in _logical_lines() + if "ssh " in line + and LOAD_BEARING.search(line) + and SWALLOWS_STATUS.search(_remote_part(line)) + ] + + +def test_the_detector_sees_the_shape_it_exists_for() -> None: + """Guards the guard, against the exact line that caused the outage.""" + broken = 'ssh -n "$SERVER" "cd ~/x && docker compose pull api 2>&1 | grep -E \'Pulled\'"' + + assert LOAD_BEARING.search(broken) + assert SWALLOWS_STATUS.search(_remote_part(broken)) + + +def test_a_local_tail_is_not_an_offence() -> None: + """`pipefail` already covers this side of the ssh; the rule is about the far + side. A check that flagged both would push the fix in the wrong direction.""" + fine = 'ssh -n "$SERVER" "docker compose pull api" | sed "s/^/[x] /" | tail -3' + + assert not SWALLOWS_STATUS.search(_remote_part(fine)) + + +def test_the_script_is_there_and_has_ssh_steps() -> None: + """An unreadable or renamed script would make the test below vacuously green.""" + assert [line for line in _logical_lines() if "ssh " in line] + + +def test_no_load_bearing_step_hides_its_exit_status() -> None: + offenders = _offenders() + + assert not offenders, "remote pipes swallow the status of:\n " + "\n ".join(offenders) + + +class TestTheHostRecordsOnlyWhatItRuns: + """`~/netork/.env` on the server is a record, not an intention. + + It used to be written before the pull. A deploy that then failed — a tag the + registry does not have, say — left the host recording a version it had never + run, and the next `docker compose up` a human typed there reached for an + image that does not exist. Observed on a test host while verifying the + failure path above: the deploy stopped exactly where it should, and still + left `NETORK_VERSION=main-0000000` behind. + + So it is written after the swap and after the image-id check agreed, and this + holds that order. Positional rather than behavioural, which is the honest + limit of a static check — but the order is the whole property. + """ + + @staticmethod + def _position(needle: str) -> int: + text = DEPLOY.read_text() + assert text.count(needle) == 1, f"{needle!r} appears {text.count(needle)} times" + return text.index(needle) + + def test_env_is_recorded_after_the_image_check(self) -> None: + verified = self._position("Verified: containers run") + recorded = self._position("Recording registry settings") + + assert recorded > verified, ( + "the host records NETORK_VERSION before the deploy is verified; " + "a failed deploy then leaves it pointing at a version never run" + ) + + def test_env_is_recorded_after_the_pull(self) -> None: + assert self._position("Recording registry settings") > self._position("Pulling images...")