Files
deploy/deploy.sh
T
Christian Manivong de46ab8a0f
CI / check (pull_request) Successful in 16s
fix: remove the bundled registry from every host, and stop starting it
netOrk's docker-compose.yml shipped a registry:2 for the satellite image,
and this script started it on every deploy as infrastructure. Satellites
pull from registry.netork.io; on 172.22.8.50 the registry held one old
image, nothing had pulled from it in 30 days, and it accepted anonymous
pushes on port 5000 of every instance (NetOrk/netork#763).

- registry leaves INFRA_SERVICES.
- A new step removes services netOrk no longer ships: the container
  netork-registry-1, then the volume netork_registry_data. `docker
  compose up` never removes a container whose service left the compose
  file, so without this each host would keep it until someone removed it
  by hand. The step is idempotent and works with the old compose file as
  well as the new one, so it can go out before netOrk drops the service.
2026-10-07 16:17:23 +02:00

440 lines
20 KiB
Bash

#!/usr/bin/env bash
# Deploy netOrk to one or more servers in parallel.
#
# Usage:
# ./deploy.sh [--version=<tag>] [server1 server2 ...]
#
# --version overrides the per-server NETORK_VERSION from deploy.env for this
# run only — useful to test one commit (--version=main-<sha>) without moving
# a server to a different release channel.
#
# Default targets and registry credentials are read from deploy.env next to
# this script (see deploy.env.example). DEPLOY_ENV_FILE points elsewhere.
#
# The servers pull pre-built images from REGISTRY_HOST. Nothing is built or
# copied from a working tree: the compose files come out of the engine image
# for the tag being deployed, and migrations run from that same image.
set -euo pipefail
HERE="$(cd "$(dirname "$0")" && pwd)"
ENV_FILE="${DEPLOY_ENV_FILE:-${HERE}/deploy.env}"
if [[ -f "$ENV_FILE" ]]; then
# shellcheck source=/dev/null
source "$ENV_FILE"
fi
DEPLOY_SERVERS="${DEPLOY_SERVERS:-}"
UI_SERVER="${UI_SERVER:-}"
REGISTRY_HOST="${REGISTRY_HOST:-}"
REGISTRY_USER="${REGISTRY_USER:-}"
REGISTRY_PASSWORD="${REGISTRY_PASSWORD:-}"
NETORK_VERSION="${NETORK_VERSION:-latest}"
# The compose files the engine image carries under /app/deploy/.
COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml"
# How long a deploy waits for another one on the same host before giving up.
DEPLOY_LOCK_WAIT="${DEPLOY_LOCK_WAIT:-900}"
read -ra SERVERS <<< "$DEPLOY_SERVERS"
# Run one step on a server, and let its failure stop the deploy.
#
# ssh reports the exit status of the *remote* command — and when that command
# ends in a pipe, the status is the last stage's, not the work's:
#
# ssh "$SERVER" "docker compose pull … 2>&1 | grep -E 'Pulled|Already|Error'"
#
# A failed pull prints a line containing `Error`, grep matches it and exits 0,
# and the deploy walks on to swap the containers. `set -o pipefail` cannot help:
# that pipeline ran in the remote shell, and from here the ssh succeeded.
#
# So the remote side stays pipe-free and the filtering happens here, where the
# status is still the one the command returned.
#
# run_remote <server> <what it is> <grep -E pattern, or "" for a tail> <command>
run_remote() {
local server=$1 label=$2 pattern=$3 command=$4
local output status
output="$(ssh -n "$server" "$command" 2>&1)" && status=0 || status=$?
if [[ $status -ne 0 ]]; then
printf '%s\n' "$output" | tail -25 | sed "s/^/[${server}] /" >&2
echo "[${server}] FAILED (exit ${status}): ${label}" >&2
return "$status"
fi
if [[ -n "$pattern" ]]; then
printf '%s\n' "$output" | grep -E "$pattern" | sed "s/^/[${server}] /" || true
else
printf '%s\n' "$output" | tail -6 | sed "s/^/[${server}] /"
fi
}
# ── One deploy per host at a time ────────────────────────────────────────────
# Several sessions deploy to the same test server, and two runs used to overlap:
# on 2026-10-05 two `up -d --force-recreate` recreated each other's containers,
# on 2026-10-06 two runs renamed each other's *.new compose files, and with two
# tags one tag's compose files could start the other's images (NetOrk/deploy#1,
# #3). So each deploy first takes flock on ~/netork/.deploy.lock on the host.
#
# An ssh session holds it: the remote side opens the lock file on fd 9, takes
# the lock, says LOCKED, and then waits on its stdin. Closing that stdin ends
# the session and frees the lock, and so does anything else that ends it: the
# deploy failing, being interrupted, or losing the network. No stale lock file
# can outlive the deploy that took it.
#
# A host without flock (util-linux) is deployed without the lock, with a
# warning, rather than becoming undeployable.
#
# hold_host_lock <server> <who and what, for the next deploy to read>
hold_host_lock() {
local server=$1 info=$2 line
# shellcheck disable=SC2029 # the wait and the info are meant to expand here
coproc HOST_LOCK {
ssh "$server" "mkdir -p ~/netork && cd ~/netork && exec 9>.deploy.lock && \
if ! command -v flock >/dev/null 2>&1; then echo NOLOCK; exec cat >/dev/null; fi; \
if ! flock -n 9; then \
echo \"BUSY \$(cat .deploy.lock.info 2>/dev/null)\"; \
flock -w ${DEPLOY_LOCK_WAIT} 9 || { echo TIMEOUT; exit 1; }; \
fi; \
echo '${info}' > .deploy.lock.info; echo LOCKED; exec cat >/dev/null" 2>&1
}
while IFS= read -r -t "$(( DEPLOY_LOCK_WAIT + 60 ))" line <&"${HOST_LOCK[0]}"; do
case "$line" in
LOCKED) return 0 ;;
NOLOCK)
echo "[${server}] WARNING: no flock on ${server}; deploying without the host lock." >&2
return 0
;;
BUSY*)
echo "[${server}] another deploy holds ${server}: ${line#BUSY } — waiting up to ${DEPLOY_LOCK_WAIT}s..." >&2
;;
TIMEOUT)
echo "[${server}] the deploy lock on ${server} is still held after ${DEPLOY_LOCK_WAIT}s; giving up." >&2
return 1
;;
*) echo "[${server}] ${line}" >&2 ;;
esac
done
echo "[${server}] could not take the deploy lock on ${server} (the ssh session ended)." >&2
return 1
}
# Free the lock: closing the session's stdin ends the remote side.
release_host_lock() {
local fd="${HOST_LOCK[1]}" pid="${HOST_LOCK_PID}"
exec {fd}>&-
wait "$pid" 2>/dev/null || true
}
VERSION_OVERRIDE=""
EXPLICIT_SERVERS=()
i=1
while [[ $i -le $# ]]; do
arg="${!i}"
case "$arg" in
--version)
i=$(( i + 1 ))
VERSION_OVERRIDE="${!i:-}"
;;
--version=*) VERSION_OVERRIDE="${arg#--version=}" ;;
-*)
echo "Unknown option: ${arg}" >&2
exit 2
;;
*) EXPLICIT_SERVERS+=("$arg") ;;
esac
i=$(( i + 1 ))
done
if [[ ${#EXPLICIT_SERVERS[@]} -gt 0 ]]; then
SERVERS=("${EXPLICIT_SERVERS[@]}")
fi
# ── Version validation ───────────────────────────────────────────────────────
# A deploy only pulls images — it never verifies the pulled tag actually
# corresponds to a real, recently-built image for the commit/branch you think
# you're deploying. A stale or accidentally-reused tag (or a typo'd branch
# name that happens to collide with some ancient leftover tag) is pulled
# silently, with the only symptom surfacing later at `alembic upgrade head`,
# by which point containers are already switched over. Reject anything that
# doesn't match a tag format netOrk's CI actually publishes: latest,
# latest-dev, main-<sha>, feature-<branch> (feature/* only), or a release
# version (X.Y.Z). See NetOrk/netork#55.
VALID_VERSION_RE='^(latest|latest-dev|main-[0-9a-f]{7,40}|feature-[A-Za-z0-9._-]+|[0-9]+\.[0-9]+\.[0-9]+)$'
_validate_version() {
local version="$1"
if [[ ! "$version" =~ $VALID_VERSION_RE ]]; then
echo "Refusing to deploy version '${version}': not a tag CI publishes (expected latest," \
"latest-dev, main-<sha>, feature-<branch>, or X.Y.Z)." >&2
return 1
fi
}
if [[ -n "$VERSION_OVERRIDE" ]]; then
_validate_version "$VERSION_OVERRIDE"
fi
if [[ -z "$REGISTRY_HOST" ]]; then
echo "REGISTRY_HOST is not set (deploy.env or environment): nowhere to pull images from." >&2
exit 1
fi
if [[ ${#SERVERS[@]} -eq 0 ]]; then
echo "No servers given: pass them as arguments or set DEPLOY_SERVERS in deploy.env." >&2
exit 1
fi
# One server: under its host lock, the steps below.
#
# A step that fails stops the subshell (set -e) before release_host_lock runs.
# That is fine: the subshell's end closes the lock session's stdin all the same.
deploy_server() {
local SERVER="$1" KEY="${1//./_}" VER_VAR
VER_VAR="NETORK_VERSION_${KEY}"
hold_host_lock "$SERVER" \
"since $(date -u +%Y-%m-%dT%H:%M:%SZ), by ${USER:-?}@$(hostname -s), deploying ${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}"
deploy_steps "$SERVER"
release_host_lock
}
deploy_steps() {
local SERVER="$1"
# Per-server lookup — falls back to global values
local KEY="${SERVER//./_}"
local USER_VAR="REGISTRY_USER_${KEY}"
local PASS_VAR="REGISTRY_PASSWORD_${KEY}"
local VER_VAR="NETORK_VERSION_${KEY}"
local SERVER_USER="${!USER_VAR:-${REGISTRY_USER}}"
local SERVER_PASS="${!PASS_VAR:-${REGISTRY_PASSWORD}}"
local SERVER_VERSION="${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}"
_validate_version "$SERVER_VERSION" || return 1
local ENGINE_IMAGE="${REGISTRY_HOST}/netork/engine:${SERVER_VERSION}"
# The password goes over ssh's stdin. Inside the ssh argument it would be
# part of the remote shell's command line, readable through `ps` by every
# user on the host for as long as the login runs.
echo "[${SERVER}] Logging in to registry..."
# shellcheck disable=SC2029 # host and user are meant to expand here
printf '%s\n' "$SERVER_PASS" \
| ssh "$SERVER" "docker login '${REGISTRY_HOST}' -u '${SERVER_USER}' --password-stdin 2>&1" \
| sed "s/^/[${SERVER}] /"
# The compose files are versioned with the images they start, so they are
# taken from the engine image of this very tag rather than from whatever
# checkout happens to run the deploy. Both are written to .new first and only
# moved into place once both were read, so a failure leaves the previous pair
# intact. An image built before the files moved into it has no /app/deploy/.
echo "[${SERVER}] Fetching compose files from ${ENGINE_IMAGE}..."
run_remote "$SERVER" "fetching compose files from ${ENGINE_IMAGE}" 'Status|Error' \
"mkdir -p ~/netork && cd ~/netork && \
docker pull '${ENGINE_IMAGE}' && \
for f in ${COMPOSE_FILES}; do \
docker run --rm --entrypoint cat '${ENGINE_IMAGE}' /app/deploy/\$f > \$f.new \
|| { rm -f \$f.new; echo \"${ENGINE_IMAGE} carries no /app/deploy/\$f — the tag predates compose files in the engine image\"; exit 3; }; \
done && \
for f in ${COMPOSE_FILES}; do mv \$f.new \$f; done"
# Services running netOrk's own images. flower belongs here: it is the engine
# image with a different command, so leaving it out let the monitoring UI run
# nine days behind the workers it monitors, on an image that had since been
# untagged. See NetOrk/netork#95.
local SERVICES="netork-api netork-worker netork-poll-worker netork-ansible-worker netork-beat flower"
if [[ " $UI_SERVER " == *" $SERVER "* ]]; then
SERVICES="${SERVICES} netork-ui"
fi
# Containers on upstream images. Reconciled, not force-recreated (below).
# postgres and redis are deliberately absent: restarting a database on every
# deploy would be worse than any compose change one could miss.
local INFRA_SERVICES="apt-cacher-ng signal-api"
echo "[${SERVER}] Pulling images..."
run_remote "$SERVER" "pulling images for ${SERVER_VERSION}" 'Pulled|Already|Error' \
"cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml pull ${SERVICES}"
# --force-recreate is deliberate: `docker compose up -d` decides what to
# replace from the service definition's config hash, and the image reference
# (`.../ui:latest-dev`) does not change when the tag is moved to a new image.
# A deploy then leaves the old container running while reporting success —
# observed on netork-ui, which kept serving a stale image after its tag had
# already advanced. Recreating unconditionally costs a restart per deploy,
# which a deliberate rollout wants anyway. See NetOrk/netork#90.
#
# Tried twice. compose recreates the services in parallel, and on 2026-10-05 it
# lost a container it had just renamed ("No such container") and stopped with
# every engine container Created and none running: the old API was gone, so
# netOrk was down until someone ran the deploy again, which then went through
# (NetOrk/netork#586). The second attempt is that rerun. Should it fail as well,
# the container states go to the log and the deploy fails as before.
echo "[${SERVER}] Starting containers..."
local RECREATE="cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps --force-recreate ${SERVICES}"
if ! run_remote "$SERVER" "recreating ${SERVICES}" "" "$RECREATE"; then
local delay="${DEPLOY_RECREATE_RETRY_DELAY:-5}"
echo "[${SERVER}] Recreating failed; trying once more in ${delay}s..." >&2
sleep "$delay"
if ! run_remote "$SERVER" "recreating ${SERVICES} (second attempt)" "" "$RECREATE"; then
echo "[${SERVER}] Containers after two failed attempts:" >&2
ssh -n "$SERVER" "cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -a" 2>&1 \
| sed "s/^/[${SERVER}] /" >&2 || true
return 1
fi
fi
# Infrastructure containers, deliberately left out of the force-recreate
# above. Plain `up -d`: compose compares each service definition against the
# running container and recreates only what actually changed, so an edit to a
# healthcheck or an option in docker-compose.yml takes effect while a
# container nobody touched is left alone. signal-api is the reason this is
# not --force-recreate: it holds netOrk's Signal device link and recreating
# it every deploy is churn nobody asked for. See NetOrk/netork#95.
#
# `|| true`: infrastructure is reconciled opportunistically and its failure
# is not a reason to abandon a deploy that has already succeeded — but it
# prints as a failure instead of scrolling past.
echo "[${SERVER}] Reconciling infrastructure containers..."
run_remote "$SERVER" "reconciling ${INFRA_SERVICES}" "" \
"cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps ${INFRA_SERVICES}" || true
# Services netOrk no longer ships, with the data only they used. `docker
# compose up` never removes a container whose service has left the compose
# file, so each would keep running on every host until someone removed it by
# hand. Idempotent: a host that never had one, or was cleaned already, passes
# untouched. Containers first; a volume still in use cannot be removed.
# registry (netork-registry-1, netork_registry_data): the bundled registry:2
# for satellite images. Satellites pull from registry.netork.io; it held one
# old image, nothing had pulled from it in 30 days, and it accepted
# anonymous pushes on port 5000 (NetOrk/netork#763).
local RETIRED_CONTAINERS="netork-registry-1"
local RETIRED_VOLUMES="netork_registry_data"
echo "[${SERVER}] Removing retired services..."
run_remote "$SERVER" "removing ${RETIRED_CONTAINERS} ${RETIRED_VOLUMES}" "" \
"for c in ${RETIRED_CONTAINERS}; do \
docker rm -f \$c >/dev/null 2>&1 && echo \"removed container \$c\"; done; \
for v in ${RETIRED_VOLUMES}; do \
docker volume rm \$v >/dev/null 2>&1 && echo \"removed volume \$v\"; done; true" || true
# Verify the containers actually run the image we just pulled. `docker
# compose up -d` reports "Running" (not "Started") when it decides nothing
# changed, and `pull` prints "Pulled" even when the tag was already local —
# so a deploy that had no effect at all is indistinguishable from a real one
# in the output above. Both the wrong-tag case (NETORK_VERSION pointing at a
# tag CI never republished) and a silently failed pull land here.
# See NetOrk/netork#88.
echo "[${SERVER}] Verifying containers run the pulled image..."
local VERIFY_PAIRS="netork-api:engine flower:engine"
if [[ " $UI_SERVER " == *" $SERVER "* ]]; then
VERIFY_PAIRS="${VERIFY_PAIRS} netork-ui:ui"
fi
local MISMATCH
MISMATCH=$(ssh -n "$SERVER" "cd ~/netork && for pair in ${VERIFY_PAIRS}; do \
svc=\${pair%%:*}; repo=\${pair##*:}; \
want=\$(docker image inspect --format '{{.Id}}' '${REGISTRY_HOST}/netork/'\${repo}':${SERVER_VERSION}' 2>/dev/null || true); \
cid=\$(REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -q \${svc} 2>/dev/null || true); \
got=\$(docker inspect --format '{{.Image}}' \${cid} 2>/dev/null || true); \
if [ -z \"\${want}\" ] || [ -z \"\${got}\" ] || [ \"\${want}\" != \"\${got}\" ]; then echo \${svc}; fi; \
done" || true)
if [[ -n "$MISMATCH" ]]; then
echo "[${SERVER}] ERROR: still running an older image after deploy: ${MISMATCH//$'\n'/ }" >&2
echo "[${SERVER}] Wanted tag '${SERVER_VERSION}'. On main pushes CI publishes" >&2
echo "[${SERVER}] 'latest-dev' and 'main-<sha>'; ':latest' only on version tags." >&2
return 1
fi
echo "[${SERVER}] Verified: containers run ${SERVER_VERSION}."
# docker compose substitutes REGISTRY_HOST / NETORK_VERSION from the project
# .env when they are not in the environment. This script passes both inline,
# but a hand-typed `docker compose up -d flower` on the host needs them there
# or dies on "invalid reference format". Values are rewritten in place, so
# the file does not grow a new line per deploy.
#
# Only now. This used to run before the pull, so a deploy that failed on a tag
# the registry does not have still left the host recording that tag — and the
# next `docker compose up` a human ran there reached for an image that does
# not exist. The file is a record of what this server runs, so it is written
# once that is true: after the swap and after the image-id check agreed.
echo "[${SERVER}] Recording registry settings in ~/netork/.env..."
# shellcheck disable=SC2087 # registry and version expand here; \$ escapes run remotely
ssh "$SERVER" "bash -s" <<REMOTE_ENV
set -eu
cd ~/netork
touch .env
# Append to a file that does not end in a newline, and the last line grows a
# suffix instead of gaining a neighbour.
[ ! -s .env ] || [ -z "\$(tail -c1 .env)" ] || printf '\n' >> .env
for kv in "REGISTRY_HOST=${REGISTRY_HOST}" "NETORK_VERSION=${SERVER_VERSION}"; do
key=\${kv%%=*}
if grep -q "^\${key}=" .env; then
sed -i "s|^\${key}=.*|\${kv}|" .env
else
printf '%s\n' "\${kv}" >> .env
fi
done
REMOTE_ENV
# netOrk migrates *after* the swap — its API tolerates the gap. What it does
# not tolerate is the migration failing unnoticed: the new code is already
# serving, so a missing column is a 500 on every request that touches it
# until someone looks. Loud, and non-zero. The revisions are the image's own.
echo "[${SERVER}] Running migrations..."
run_remote "$SERVER" "alembic upgrade head (containers are ALREADY swapped)" "" \
"cd ~/netork && docker compose exec -T netork-api alembic upgrade head"
echo "[${SERVER}] Pruning unused images..."
ssh -n "$SERVER" "docker image prune -af 2>&1 | tail -1" | sed "s/^/[${SERVER}] /"
echo "[${SERVER}] Done."
}
# ── Dispatch ──────────────────────────────────────────────────────────────────
echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}"
export -f deploy_server deploy_steps hold_host_lock release_host_lock _validate_version run_remote
export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \
VALID_VERSION_RE COMPOSE_FILES DEPLOY_LOCK_WAIT
# Export all per-server credential vars so subshells can resolve them
while IFS='=' read -r key _; do
if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then
export "${key?}"
fi
done < <(compgen -v)
PIDS=()
for SERVER in "${SERVERS[@]}"; do
deploy_server "$SERVER" &
PIDS+=($!)
done
FAILED=()
for i in "${!PIDS[@]}"; do
if ! wait "${PIDS[$i]}"; then
FAILED+=("${SERVERS[$i]}")
fi
done
if [[ ${#FAILED[@]} -gt 0 ]]; then
echo "FAILED on: ${FAILED[*]}" >&2
exit 1
fi
echo "All servers deployed successfully."