Files
2026-09-28 19:55:22 +02:00

330 lines
14 KiB
Bash

#!/usr/bin/env bash
# Deploy netOrk to one or more servers in parallel.
#
# Usage:
# ./deploy.sh [--version=<tag>] [server1 server2 ...]
#
# --version overrides the per-server NETORK_VERSION from deploy.env for this
# run only — useful to test one commit (--version=main-<sha>) without moving
# a server to a different release channel.
#
# Default targets and registry credentials are read from deploy.env next to
# this script (see deploy.env.example). DEPLOY_ENV_FILE points elsewhere.
#
# The servers pull pre-built images from REGISTRY_HOST. Nothing is built or
# copied from a working tree: the compose files come out of the engine image
# for the tag being deployed, and migrations run from that same image.
set -euo pipefail
HERE="$(cd "$(dirname "$0")" && pwd)"
ENV_FILE="${DEPLOY_ENV_FILE:-${HERE}/deploy.env}"
if [[ -f "$ENV_FILE" ]]; then
# shellcheck source=/dev/null
source "$ENV_FILE"
fi
DEPLOY_SERVERS="${DEPLOY_SERVERS:-}"
UI_SERVER="${UI_SERVER:-}"
REGISTRY_HOST="${REGISTRY_HOST:-}"
REGISTRY_USER="${REGISTRY_USER:-}"
REGISTRY_PASSWORD="${REGISTRY_PASSWORD:-}"
NETORK_VERSION="${NETORK_VERSION:-latest}"
# The compose files the engine image carries under /app/deploy/.
COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml"
read -ra SERVERS <<< "$DEPLOY_SERVERS"
# Run one step on a server, and let its failure stop the deploy.
#
# ssh reports the exit status of the *remote* command — and when that command
# ends in a pipe, the status is the last stage's, not the work's:
#
# ssh "$SERVER" "docker compose pull … 2>&1 | grep -E 'Pulled|Already|Error'"
#
# A failed pull prints a line containing `Error`, grep matches it and exits 0,
# and the deploy walks on to swap the containers. `set -o pipefail` cannot help:
# that pipeline ran in the remote shell, and from here the ssh succeeded.
#
# So the remote side stays pipe-free and the filtering happens here, where the
# status is still the one the command returned.
#
# run_remote <server> <what it is> <grep -E pattern, or "" for a tail> <command>
run_remote() {
local server=$1 label=$2 pattern=$3 command=$4
local output status
output="$(ssh -n "$server" "$command" 2>&1)" && status=0 || status=$?
if [[ $status -ne 0 ]]; then
printf '%s\n' "$output" | tail -25 | sed "s/^/[${server}] /" >&2
echo "[${server}] FAILED (exit ${status}): ${label}" >&2
return "$status"
fi
if [[ -n "$pattern" ]]; then
printf '%s\n' "$output" | grep -E "$pattern" | sed "s/^/[${server}] /" || true
else
printf '%s\n' "$output" | tail -6 | sed "s/^/[${server}] /"
fi
}
VERSION_OVERRIDE=""
EXPLICIT_SERVERS=()
i=1
while [[ $i -le $# ]]; do
arg="${!i}"
case "$arg" in
--version)
i=$(( i + 1 ))
VERSION_OVERRIDE="${!i:-}"
;;
--version=*) VERSION_OVERRIDE="${arg#--version=}" ;;
-*)
echo "Unknown option: ${arg}" >&2
exit 2
;;
*) EXPLICIT_SERVERS+=("$arg") ;;
esac
i=$(( i + 1 ))
done
if [[ ${#EXPLICIT_SERVERS[@]} -gt 0 ]]; then
SERVERS=("${EXPLICIT_SERVERS[@]}")
fi
# ── Version validation ───────────────────────────────────────────────────────
# A deploy only pulls images — it never verifies the pulled tag actually
# corresponds to a real, recently-built image for the commit/branch you think
# you're deploying. A stale or accidentally-reused tag (or a typo'd branch
# name that happens to collide with some ancient leftover tag) is pulled
# silently, with the only symptom surfacing later at `alembic upgrade head`,
# by which point containers are already switched over. Reject anything that
# doesn't match a tag format netOrk's CI actually publishes: latest,
# latest-dev, main-<sha>, feature-<branch> (feature/* only), or a release
# version (X.Y.Z). See NetOrk/netork#55.
VALID_VERSION_RE='^(latest|latest-dev|main-[0-9a-f]{7,40}|feature-[A-Za-z0-9._-]+|[0-9]+\.[0-9]+\.[0-9]+)$'
_validate_version() {
local version="$1"
if [[ ! "$version" =~ $VALID_VERSION_RE ]]; then
echo "Refusing to deploy version '${version}': not a tag CI publishes (expected latest," \
"latest-dev, main-<sha>, feature-<branch>, or X.Y.Z)." >&2
return 1
fi
}
if [[ -n "$VERSION_OVERRIDE" ]]; then
_validate_version "$VERSION_OVERRIDE"
fi
if [[ -z "$REGISTRY_HOST" ]]; then
echo "REGISTRY_HOST is not set (deploy.env or environment): nowhere to pull images from." >&2
exit 1
fi
if [[ ${#SERVERS[@]} -eq 0 ]]; then
echo "No servers given: pass them as arguments or set DEPLOY_SERVERS in deploy.env." >&2
exit 1
fi
deploy_server() {
local SERVER="$1"
# Per-server lookup — falls back to global values
local KEY="${SERVER//./_}"
local USER_VAR="REGISTRY_USER_${KEY}"
local PASS_VAR="REGISTRY_PASSWORD_${KEY}"
local VER_VAR="NETORK_VERSION_${KEY}"
local SERVER_USER="${!USER_VAR:-${REGISTRY_USER}}"
local SERVER_PASS="${!PASS_VAR:-${REGISTRY_PASSWORD}}"
local SERVER_VERSION="${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}"
_validate_version "$SERVER_VERSION" || return 1
local ENGINE_IMAGE="${REGISTRY_HOST}/netork/engine:${SERVER_VERSION}"
# The password goes over ssh's stdin. Inside the ssh argument it would be
# part of the remote shell's command line, readable through `ps` by every
# user on the host for as long as the login runs.
echo "[${SERVER}] Logging in to registry..."
# shellcheck disable=SC2029 # host and user are meant to expand here
printf '%s\n' "$SERVER_PASS" \
| ssh "$SERVER" "docker login '${REGISTRY_HOST}' -u '${SERVER_USER}' --password-stdin 2>&1" \
| sed "s/^/[${SERVER}] /"
# The compose files are versioned with the images they start, so they are
# taken from the engine image of this very tag rather than from whatever
# checkout happens to run the deploy. Both are written to .new first and only
# moved into place once both were read, so a failure leaves the previous pair
# intact. An image built before the files moved into it has no /app/deploy/.
echo "[${SERVER}] Fetching compose files from ${ENGINE_IMAGE}..."
run_remote "$SERVER" "fetching compose files from ${ENGINE_IMAGE}" 'Status|Error' \
"mkdir -p ~/netork && cd ~/netork && \
docker pull '${ENGINE_IMAGE}' && \
for f in ${COMPOSE_FILES}; do \
docker run --rm --entrypoint cat '${ENGINE_IMAGE}' /app/deploy/\$f > \$f.new \
|| { rm -f \$f.new; echo \"${ENGINE_IMAGE} carries no /app/deploy/\$f — the tag predates compose files in the engine image\"; exit 3; }; \
done && \
for f in ${COMPOSE_FILES}; do mv \$f.new \$f; done"
# Services running netOrk's own images. flower belongs here: it is the engine
# image with a different command, so leaving it out let the monitoring UI run
# nine days behind the workers it monitors, on an image that had since been
# untagged. See NetOrk/netork#95.
local SERVICES="netork-api netork-worker netork-poll-worker netork-ansible-worker netork-beat flower"
if [[ " $UI_SERVER " == *" $SERVER "* ]]; then
SERVICES="${SERVICES} netork-ui"
fi
# Containers on upstream images. Reconciled, not force-recreated (below).
# postgres and redis are deliberately absent: restarting a database on every
# deploy would be worse than any compose change one could miss.
local INFRA_SERVICES="registry apt-cacher-ng signal-api"
echo "[${SERVER}] Pulling images..."
run_remote "$SERVER" "pulling images for ${SERVER_VERSION}" 'Pulled|Already|Error' \
"cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml pull ${SERVICES}"
# --force-recreate is deliberate: `docker compose up -d` decides what to
# replace from the service definition's config hash, and the image reference
# (`.../ui:latest-dev`) does not change when the tag is moved to a new image.
# A deploy then leaves the old container running while reporting success —
# observed on netork-ui, which kept serving a stale image after its tag had
# already advanced. Recreating unconditionally costs a restart per deploy,
# which a deliberate rollout wants anyway. See NetOrk/netork#90.
echo "[${SERVER}] Starting containers..."
run_remote "$SERVER" "recreating ${SERVICES}" "" \
"cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps --force-recreate ${SERVICES}"
# Infrastructure containers, deliberately left out of the force-recreate
# above. Plain `up -d`: compose compares each service definition against the
# running container and recreates only what actually changed, so an edit to a
# healthcheck or an option in docker-compose.yml takes effect while a
# container nobody touched is left alone. signal-api is the reason this is
# not --force-recreate: it holds netOrk's Signal device link and recreating
# it every deploy is churn nobody asked for. See NetOrk/netork#95.
#
# `|| true`: infrastructure is reconciled opportunistically and its failure
# is not a reason to abandon a deploy that has already succeeded — but it
# prints as a failure instead of scrolling past.
echo "[${SERVER}] Reconciling infrastructure containers..."
run_remote "$SERVER" "reconciling ${INFRA_SERVICES}" "" \
"cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps ${INFRA_SERVICES}" || true
# Verify the containers actually run the image we just pulled. `docker
# compose up -d` reports "Running" (not "Started") when it decides nothing
# changed, and `pull` prints "Pulled" even when the tag was already local —
# so a deploy that had no effect at all is indistinguishable from a real one
# in the output above. Both the wrong-tag case (NETORK_VERSION pointing at a
# tag CI never republished) and a silently failed pull land here.
# See NetOrk/netork#88.
echo "[${SERVER}] Verifying containers run the pulled image..."
local VERIFY_PAIRS="netork-api:engine flower:engine"
if [[ " $UI_SERVER " == *" $SERVER "* ]]; then
VERIFY_PAIRS="${VERIFY_PAIRS} netork-ui:ui"
fi
local MISMATCH
MISMATCH=$(ssh -n "$SERVER" "cd ~/netork && for pair in ${VERIFY_PAIRS}; do \
svc=\${pair%%:*}; repo=\${pair##*:}; \
want=\$(docker image inspect --format '{{.Id}}' '${REGISTRY_HOST}/netork/'\${repo}':${SERVER_VERSION}' 2>/dev/null || true); \
cid=\$(REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -q \${svc} 2>/dev/null || true); \
got=\$(docker inspect --format '{{.Image}}' \${cid} 2>/dev/null || true); \
if [ -z \"\${want}\" ] || [ -z \"\${got}\" ] || [ \"\${want}\" != \"\${got}\" ]; then echo \${svc}; fi; \
done" || true)
if [[ -n "$MISMATCH" ]]; then
echo "[${SERVER}] ERROR: still running an older image after deploy: ${MISMATCH//$'\n'/ }" >&2
echo "[${SERVER}] Wanted tag '${SERVER_VERSION}'. On main pushes CI publishes" >&2
echo "[${SERVER}] 'latest-dev' and 'main-<sha>'; ':latest' only on version tags." >&2
return 1
fi
echo "[${SERVER}] Verified: containers run ${SERVER_VERSION}."
# docker compose substitutes REGISTRY_HOST / NETORK_VERSION from the project
# .env when they are not in the environment. This script passes both inline,
# but a hand-typed `docker compose up -d flower` on the host needs them there
# or dies on "invalid reference format". Values are rewritten in place, so
# the file does not grow a new line per deploy.
#
# Only now. This used to run before the pull, so a deploy that failed on a tag
# the registry does not have still left the host recording that tag — and the
# next `docker compose up` a human ran there reached for an image that does
# not exist. The file is a record of what this server runs, so it is written
# once that is true: after the swap and after the image-id check agreed.
echo "[${SERVER}] Recording registry settings in ~/netork/.env..."
# shellcheck disable=SC2087 # registry and version expand here; \$ escapes run remotely
ssh "$SERVER" "bash -s" <<REMOTE_ENV
set -eu
cd ~/netork
touch .env
# Append to a file that does not end in a newline, and the last line grows a
# suffix instead of gaining a neighbour.
[ ! -s .env ] || [ -z "\$(tail -c1 .env)" ] || printf '\n' >> .env
for kv in "REGISTRY_HOST=${REGISTRY_HOST}" "NETORK_VERSION=${SERVER_VERSION}"; do
key=\${kv%%=*}
if grep -q "^\${key}=" .env; then
sed -i "s|^\${key}=.*|\${kv}|" .env
else
printf '%s\n' "\${kv}" >> .env
fi
done
REMOTE_ENV
# netOrk migrates *after* the swap — its API tolerates the gap. What it does
# not tolerate is the migration failing unnoticed: the new code is already
# serving, so a missing column is a 500 on every request that touches it
# until someone looks. Loud, and non-zero. The revisions are the image's own.
echo "[${SERVER}] Running migrations..."
run_remote "$SERVER" "alembic upgrade head (containers are ALREADY swapped)" "" \
"cd ~/netork && docker compose exec -T netork-api alembic upgrade head"
echo "[${SERVER}] Pruning unused images..."
ssh -n "$SERVER" "docker image prune -af 2>&1 | tail -1" | sed "s/^/[${SERVER}] /"
echo "[${SERVER}] Done."
}
# ── Dispatch ──────────────────────────────────────────────────────────────────
echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}"
export -f deploy_server _validate_version run_remote
export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \
VALID_VERSION_RE COMPOSE_FILES
# Export all per-server credential vars so subshells can resolve them
while IFS='=' read -r key _; do
if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then
export "${key?}"
fi
done < <(compgen -v)
PIDS=()
for SERVER in "${SERVERS[@]}"; do
deploy_server "$SERVER" &
PIDS+=($!)
done
FAILED=()
for i in "${!PIDS[@]}"; do
if ! wait "${PIDS[$i]}"; then
FAILED+=("${SERVERS[$i]}")
fi
done
if [[ ${#FAILED[@]} -gt 0 ]]; then
echo "FAILED on: ${FAILED[*]}" >&2
exit 1
fi
echo "All servers deployed successfully."