Files
deploy/deploy.sh
Christian Manivong 923ec4ce65
CI / check (pull_request) Successful in 13s
fix: a failed container recreate is tried once more before the deploy gives up
On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers":
docker compose up -d --force-recreate recreates the services in parallel,
lost a container it had just renamed ("No such container: 02586df7…") and
stopped with every engine container Created and none running. The old API
was already gone, so netOrk was down for about two minutes, until the same
deploy was run again and went through cleanly.

The script now does that second run itself, after DEPLOY_RECREATE_RETRY_DELAY
seconds (5 by default). If the second attempt fails too, it prints the
container states (docker compose ps -a) and fails as before.

Tests run the script against a fake ssh that fails the recreate zero, one
or two times. README step 3 says what happens.

Refs NetOrk/netork#586
2026-10-05 21:58:36 +02:00

349 lines
15 KiB
Bash

#!/usr/bin/env bash
# Deploy netOrk to one or more servers in parallel.
#
# Usage:
# ./deploy.sh [--version=<tag>] [server1 server2 ...]
#
# --version overrides the per-server NETORK_VERSION from deploy.env for this
# run only — useful to test one commit (--version=main-<sha>) without moving
# a server to a different release channel.
#
# Default targets and registry credentials are read from deploy.env next to
# this script (see deploy.env.example). DEPLOY_ENV_FILE points elsewhere.
#
# The servers pull pre-built images from REGISTRY_HOST. Nothing is built or
# copied from a working tree: the compose files come out of the engine image
# for the tag being deployed, and migrations run from that same image.
set -euo pipefail
HERE="$(cd "$(dirname "$0")" && pwd)"
ENV_FILE="${DEPLOY_ENV_FILE:-${HERE}/deploy.env}"
if [[ -f "$ENV_FILE" ]]; then
# shellcheck source=/dev/null
source "$ENV_FILE"
fi
DEPLOY_SERVERS="${DEPLOY_SERVERS:-}"
UI_SERVER="${UI_SERVER:-}"
REGISTRY_HOST="${REGISTRY_HOST:-}"
REGISTRY_USER="${REGISTRY_USER:-}"
REGISTRY_PASSWORD="${REGISTRY_PASSWORD:-}"
NETORK_VERSION="${NETORK_VERSION:-latest}"
# The compose files the engine image carries under /app/deploy/.
COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml"
read -ra SERVERS <<< "$DEPLOY_SERVERS"
# Run one step on a server, and let its failure stop the deploy.
#
# ssh reports the exit status of the *remote* command — and when that command
# ends in a pipe, the status is the last stage's, not the work's:
#
# ssh "$SERVER" "docker compose pull … 2>&1 | grep -E 'Pulled|Already|Error'"
#
# A failed pull prints a line containing `Error`, grep matches it and exits 0,
# and the deploy walks on to swap the containers. `set -o pipefail` cannot help:
# that pipeline ran in the remote shell, and from here the ssh succeeded.
#
# So the remote side stays pipe-free and the filtering happens here, where the
# status is still the one the command returned.
#
# run_remote <server> <what it is> <grep -E pattern, or "" for a tail> <command>
run_remote() {
local server=$1 label=$2 pattern=$3 command=$4
local output status
output="$(ssh -n "$server" "$command" 2>&1)" && status=0 || status=$?
if [[ $status -ne 0 ]]; then
printf '%s\n' "$output" | tail -25 | sed "s/^/[${server}] /" >&2
echo "[${server}] FAILED (exit ${status}): ${label}" >&2
return "$status"
fi
if [[ -n "$pattern" ]]; then
printf '%s\n' "$output" | grep -E "$pattern" | sed "s/^/[${server}] /" || true
else
printf '%s\n' "$output" | tail -6 | sed "s/^/[${server}] /"
fi
}
VERSION_OVERRIDE=""
EXPLICIT_SERVERS=()
i=1
while [[ $i -le $# ]]; do
arg="${!i}"
case "$arg" in
--version)
i=$(( i + 1 ))
VERSION_OVERRIDE="${!i:-}"
;;
--version=*) VERSION_OVERRIDE="${arg#--version=}" ;;
-*)
echo "Unknown option: ${arg}" >&2
exit 2
;;
*) EXPLICIT_SERVERS+=("$arg") ;;
esac
i=$(( i + 1 ))
done
if [[ ${#EXPLICIT_SERVERS[@]} -gt 0 ]]; then
SERVERS=("${EXPLICIT_SERVERS[@]}")
fi
# ── Version validation ───────────────────────────────────────────────────────
# A deploy only pulls images — it never verifies the pulled tag actually
# corresponds to a real, recently-built image for the commit/branch you think
# you're deploying. A stale or accidentally-reused tag (or a typo'd branch
# name that happens to collide with some ancient leftover tag) is pulled
# silently, with the only symptom surfacing later at `alembic upgrade head`,
# by which point containers are already switched over. Reject anything that
# doesn't match a tag format netOrk's CI actually publishes: latest,
# latest-dev, main-<sha>, feature-<branch> (feature/* only), or a release
# version (X.Y.Z). See NetOrk/netork#55.
VALID_VERSION_RE='^(latest|latest-dev|main-[0-9a-f]{7,40}|feature-[A-Za-z0-9._-]+|[0-9]+\.[0-9]+\.[0-9]+)$'
_validate_version() {
local version="$1"
if [[ ! "$version" =~ $VALID_VERSION_RE ]]; then
echo "Refusing to deploy version '${version}': not a tag CI publishes (expected latest," \
"latest-dev, main-<sha>, feature-<branch>, or X.Y.Z)." >&2
return 1
fi
}
if [[ -n "$VERSION_OVERRIDE" ]]; then
_validate_version "$VERSION_OVERRIDE"
fi
if [[ -z "$REGISTRY_HOST" ]]; then
echo "REGISTRY_HOST is not set (deploy.env or environment): nowhere to pull images from." >&2
exit 1
fi
if [[ ${#SERVERS[@]} -eq 0 ]]; then
echo "No servers given: pass them as arguments or set DEPLOY_SERVERS in deploy.env." >&2
exit 1
fi
deploy_server() {
local SERVER="$1"
# Per-server lookup — falls back to global values
local KEY="${SERVER//./_}"
local USER_VAR="REGISTRY_USER_${KEY}"
local PASS_VAR="REGISTRY_PASSWORD_${KEY}"
local VER_VAR="NETORK_VERSION_${KEY}"
local SERVER_USER="${!USER_VAR:-${REGISTRY_USER}}"
local SERVER_PASS="${!PASS_VAR:-${REGISTRY_PASSWORD}}"
local SERVER_VERSION="${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}"
_validate_version "$SERVER_VERSION" || return 1
local ENGINE_IMAGE="${REGISTRY_HOST}/netork/engine:${SERVER_VERSION}"
# The password goes over ssh's stdin. Inside the ssh argument it would be
# part of the remote shell's command line, readable through `ps` by every
# user on the host for as long as the login runs.
echo "[${SERVER}] Logging in to registry..."
# shellcheck disable=SC2029 # host and user are meant to expand here
printf '%s\n' "$SERVER_PASS" \
| ssh "$SERVER" "docker login '${REGISTRY_HOST}' -u '${SERVER_USER}' --password-stdin 2>&1" \
| sed "s/^/[${SERVER}] /"
# The compose files are versioned with the images they start, so they are
# taken from the engine image of this very tag rather than from whatever
# checkout happens to run the deploy. Both are written to .new first and only
# moved into place once both were read, so a failure leaves the previous pair
# intact. An image built before the files moved into it has no /app/deploy/.
echo "[${SERVER}] Fetching compose files from ${ENGINE_IMAGE}..."
run_remote "$SERVER" "fetching compose files from ${ENGINE_IMAGE}" 'Status|Error' \
"mkdir -p ~/netork && cd ~/netork && \
docker pull '${ENGINE_IMAGE}' && \
for f in ${COMPOSE_FILES}; do \
docker run --rm --entrypoint cat '${ENGINE_IMAGE}' /app/deploy/\$f > \$f.new \
|| { rm -f \$f.new; echo \"${ENGINE_IMAGE} carries no /app/deploy/\$f — the tag predates compose files in the engine image\"; exit 3; }; \
done && \
for f in ${COMPOSE_FILES}; do mv \$f.new \$f; done"
# Services running netOrk's own images. flower belongs here: it is the engine
# image with a different command, so leaving it out let the monitoring UI run
# nine days behind the workers it monitors, on an image that had since been
# untagged. See NetOrk/netork#95.
local SERVICES="netork-api netork-worker netork-poll-worker netork-ansible-worker netork-beat flower"
if [[ " $UI_SERVER " == *" $SERVER "* ]]; then
SERVICES="${SERVICES} netork-ui"
fi
# Containers on upstream images. Reconciled, not force-recreated (below).
# postgres and redis are deliberately absent: restarting a database on every
# deploy would be worse than any compose change one could miss.
local INFRA_SERVICES="registry apt-cacher-ng signal-api"
echo "[${SERVER}] Pulling images..."
run_remote "$SERVER" "pulling images for ${SERVER_VERSION}" 'Pulled|Already|Error' \
"cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml pull ${SERVICES}"
# --force-recreate is deliberate: `docker compose up -d` decides what to
# replace from the service definition's config hash, and the image reference
# (`.../ui:latest-dev`) does not change when the tag is moved to a new image.
# A deploy then leaves the old container running while reporting success —
# observed on netork-ui, which kept serving a stale image after its tag had
# already advanced. Recreating unconditionally costs a restart per deploy,
# which a deliberate rollout wants anyway. See NetOrk/netork#90.
#
# Tried twice. compose recreates the services in parallel, and on 2026-10-05 it
# lost a container it had just renamed ("No such container") and stopped with
# every engine container Created and none running: the old API was gone, so
# netOrk was down until someone ran the deploy again, which then went through
# (NetOrk/netork#586). The second attempt is that rerun. Should it fail as well,
# the container states go to the log and the deploy fails as before.
echo "[${SERVER}] Starting containers..."
local RECREATE="cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps --force-recreate ${SERVICES}"
if ! run_remote "$SERVER" "recreating ${SERVICES}" "" "$RECREATE"; then
local delay="${DEPLOY_RECREATE_RETRY_DELAY:-5}"
echo "[${SERVER}] Recreating failed; trying once more in ${delay}s..." >&2
sleep "$delay"
if ! run_remote "$SERVER" "recreating ${SERVICES} (second attempt)" "" "$RECREATE"; then
echo "[${SERVER}] Containers after two failed attempts:" >&2
ssh -n "$SERVER" "cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -a" 2>&1 \
| sed "s/^/[${SERVER}] /" >&2 || true
return 1
fi
fi
# Infrastructure containers, deliberately left out of the force-recreate
# above. Plain `up -d`: compose compares each service definition against the
# running container and recreates only what actually changed, so an edit to a
# healthcheck or an option in docker-compose.yml takes effect while a
# container nobody touched is left alone. signal-api is the reason this is
# not --force-recreate: it holds netOrk's Signal device link and recreating
# it every deploy is churn nobody asked for. See NetOrk/netork#95.
#
# `|| true`: infrastructure is reconciled opportunistically and its failure
# is not a reason to abandon a deploy that has already succeeded — but it
# prints as a failure instead of scrolling past.
echo "[${SERVER}] Reconciling infrastructure containers..."
run_remote "$SERVER" "reconciling ${INFRA_SERVICES}" "" \
"cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps ${INFRA_SERVICES}" || true
# Verify the containers actually run the image we just pulled. `docker
# compose up -d` reports "Running" (not "Started") when it decides nothing
# changed, and `pull` prints "Pulled" even when the tag was already local —
# so a deploy that had no effect at all is indistinguishable from a real one
# in the output above. Both the wrong-tag case (NETORK_VERSION pointing at a
# tag CI never republished) and a silently failed pull land here.
# See NetOrk/netork#88.
echo "[${SERVER}] Verifying containers run the pulled image..."
local VERIFY_PAIRS="netork-api:engine flower:engine"
if [[ " $UI_SERVER " == *" $SERVER "* ]]; then
VERIFY_PAIRS="${VERIFY_PAIRS} netork-ui:ui"
fi
local MISMATCH
MISMATCH=$(ssh -n "$SERVER" "cd ~/netork && for pair in ${VERIFY_PAIRS}; do \
svc=\${pair%%:*}; repo=\${pair##*:}; \
want=\$(docker image inspect --format '{{.Id}}' '${REGISTRY_HOST}/netork/'\${repo}':${SERVER_VERSION}' 2>/dev/null || true); \
cid=\$(REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -q \${svc} 2>/dev/null || true); \
got=\$(docker inspect --format '{{.Image}}' \${cid} 2>/dev/null || true); \
if [ -z \"\${want}\" ] || [ -z \"\${got}\" ] || [ \"\${want}\" != \"\${got}\" ]; then echo \${svc}; fi; \
done" || true)
if [[ -n "$MISMATCH" ]]; then
echo "[${SERVER}] ERROR: still running an older image after deploy: ${MISMATCH//$'\n'/ }" >&2
echo "[${SERVER}] Wanted tag '${SERVER_VERSION}'. On main pushes CI publishes" >&2
echo "[${SERVER}] 'latest-dev' and 'main-<sha>'; ':latest' only on version tags." >&2
return 1
fi
echo "[${SERVER}] Verified: containers run ${SERVER_VERSION}."
# docker compose substitutes REGISTRY_HOST / NETORK_VERSION from the project
# .env when they are not in the environment. This script passes both inline,
# but a hand-typed `docker compose up -d flower` on the host needs them there
# or dies on "invalid reference format". Values are rewritten in place, so
# the file does not grow a new line per deploy.
#
# Only now. This used to run before the pull, so a deploy that failed on a tag
# the registry does not have still left the host recording that tag — and the
# next `docker compose up` a human ran there reached for an image that does
# not exist. The file is a record of what this server runs, so it is written
# once that is true: after the swap and after the image-id check agreed.
echo "[${SERVER}] Recording registry settings in ~/netork/.env..."
# shellcheck disable=SC2087 # registry and version expand here; \$ escapes run remotely
ssh "$SERVER" "bash -s" <<REMOTE_ENV
set -eu
cd ~/netork
touch .env
# Append to a file that does not end in a newline, and the last line grows a
# suffix instead of gaining a neighbour.
[ ! -s .env ] || [ -z "\$(tail -c1 .env)" ] || printf '\n' >> .env
for kv in "REGISTRY_HOST=${REGISTRY_HOST}" "NETORK_VERSION=${SERVER_VERSION}"; do
key=\${kv%%=*}
if grep -q "^\${key}=" .env; then
sed -i "s|^\${key}=.*|\${kv}|" .env
else
printf '%s\n' "\${kv}" >> .env
fi
done
REMOTE_ENV
# netOrk migrates *after* the swap — its API tolerates the gap. What it does
# not tolerate is the migration failing unnoticed: the new code is already
# serving, so a missing column is a 500 on every request that touches it
# until someone looks. Loud, and non-zero. The revisions are the image's own.
echo "[${SERVER}] Running migrations..."
run_remote "$SERVER" "alembic upgrade head (containers are ALREADY swapped)" "" \
"cd ~/netork && docker compose exec -T netork-api alembic upgrade head"
echo "[${SERVER}] Pruning unused images..."
ssh -n "$SERVER" "docker image prune -af 2>&1 | tail -1" | sed "s/^/[${SERVER}] /"
echo "[${SERVER}] Done."
}
# ── Dispatch ──────────────────────────────────────────────────────────────────
echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}"
export -f deploy_server _validate_version run_remote
export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \
VALID_VERSION_RE COMPOSE_FILES
# Export all per-server credential vars so subshells can resolve them
while IFS='=' read -r key _; do
if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then
export "${key?}"
fi
done < <(compgen -v)
PIDS=()
for SERVER in "${SERVERS[@]}"; do
deploy_server "$SERVER" &
PIDS+=($!)
done
FAILED=()
for i in "${!PIDS[@]}"; do
if ! wait "${PIDS[$i]}"; then
FAILED+=("${SERVERS[$i]}")
fi
done
if [[ ${#FAILED[@]} -gt 0 ]]; then
echo "FAILED on: ${FAILED[*]}" >&2
exit 1
fi
echo "All servers deployed successfully."