This commit is contained in:
@@ -0,0 +1,329 @@
|
||||
#!/usr/bin/env bash
|
||||
# Deploy netOrk to one or more servers in parallel.
|
||||
#
|
||||
# Usage:
|
||||
# ./deploy.sh [--version=<tag>] [server1 server2 ...]
|
||||
#
|
||||
# --version overrides the per-server NETORK_VERSION from deploy.env for this
|
||||
# run only — useful to test one commit (--version=main-<sha>) without moving
|
||||
# a server to a different release channel.
|
||||
#
|
||||
# Default targets and registry credentials are read from deploy.env next to
|
||||
# this script (see deploy.env.example). DEPLOY_ENV_FILE points elsewhere.
|
||||
#
|
||||
# The servers pull pre-built images from REGISTRY_HOST. Nothing is built or
|
||||
# copied from a working tree: the compose files come out of the engine image
|
||||
# for the tag being deployed, and migrations run from that same image.
|
||||
set -euo pipefail
|
||||
|
||||
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||
|
||||
ENV_FILE="${DEPLOY_ENV_FILE:-${HERE}/deploy.env}"
|
||||
if [[ -f "$ENV_FILE" ]]; then
|
||||
# shellcheck source=/dev/null
|
||||
source "$ENV_FILE"
|
||||
fi
|
||||
|
||||
DEPLOY_SERVERS="${DEPLOY_SERVERS:-}"
|
||||
UI_SERVER="${UI_SERVER:-}"
|
||||
REGISTRY_HOST="${REGISTRY_HOST:-}"
|
||||
REGISTRY_USER="${REGISTRY_USER:-}"
|
||||
REGISTRY_PASSWORD="${REGISTRY_PASSWORD:-}"
|
||||
NETORK_VERSION="${NETORK_VERSION:-latest}"
|
||||
|
||||
# The compose files the engine image carries under /app/deploy/.
|
||||
COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml"
|
||||
|
||||
read -ra SERVERS <<< "$DEPLOY_SERVERS"
|
||||
|
||||
# Run one step on a server, and let its failure stop the deploy.
|
||||
#
|
||||
# ssh reports the exit status of the *remote* command — and when that command
|
||||
# ends in a pipe, the status is the last stage's, not the work's:
|
||||
#
|
||||
# ssh "$SERVER" "docker compose pull … 2>&1 | grep -E 'Pulled|Already|Error'"
|
||||
#
|
||||
# A failed pull prints a line containing `Error`, grep matches it and exits 0,
|
||||
# and the deploy walks on to swap the containers. `set -o pipefail` cannot help:
|
||||
# that pipeline ran in the remote shell, and from here the ssh succeeded.
|
||||
#
|
||||
# So the remote side stays pipe-free and the filtering happens here, where the
|
||||
# status is still the one the command returned.
|
||||
#
|
||||
# run_remote <server> <what it is> <grep -E pattern, or "" for a tail> <command>
|
||||
run_remote() {
|
||||
local server=$1 label=$2 pattern=$3 command=$4
|
||||
local output status
|
||||
|
||||
output="$(ssh -n "$server" "$command" 2>&1)" && status=0 || status=$?
|
||||
|
||||
if [[ $status -ne 0 ]]; then
|
||||
printf '%s\n' "$output" | tail -25 | sed "s/^/[${server}] /" >&2
|
||||
echo "[${server}] FAILED (exit ${status}): ${label}" >&2
|
||||
return "$status"
|
||||
fi
|
||||
|
||||
if [[ -n "$pattern" ]]; then
|
||||
printf '%s\n' "$output" | grep -E "$pattern" | sed "s/^/[${server}] /" || true
|
||||
else
|
||||
printf '%s\n' "$output" | tail -6 | sed "s/^/[${server}] /"
|
||||
fi
|
||||
}
|
||||
|
||||
VERSION_OVERRIDE=""
|
||||
EXPLICIT_SERVERS=()
|
||||
i=1
|
||||
while [[ $i -le $# ]]; do
|
||||
arg="${!i}"
|
||||
case "$arg" in
|
||||
--version)
|
||||
i=$(( i + 1 ))
|
||||
VERSION_OVERRIDE="${!i:-}"
|
||||
;;
|
||||
--version=*) VERSION_OVERRIDE="${arg#--version=}" ;;
|
||||
-*)
|
||||
echo "Unknown option: ${arg}" >&2
|
||||
exit 2
|
||||
;;
|
||||
*) EXPLICIT_SERVERS+=("$arg") ;;
|
||||
esac
|
||||
i=$(( i + 1 ))
|
||||
done
|
||||
if [[ ${#EXPLICIT_SERVERS[@]} -gt 0 ]]; then
|
||||
SERVERS=("${EXPLICIT_SERVERS[@]}")
|
||||
fi
|
||||
|
||||
# ── Version validation ───────────────────────────────────────────────────────
|
||||
# A deploy only pulls images — it never verifies the pulled tag actually
|
||||
# corresponds to a real, recently-built image for the commit/branch you think
|
||||
# you're deploying. A stale or accidentally-reused tag (or a typo'd branch
|
||||
# name that happens to collide with some ancient leftover tag) is pulled
|
||||
# silently, with the only symptom surfacing later at `alembic upgrade head`,
|
||||
# by which point containers are already switched over. Reject anything that
|
||||
# doesn't match a tag format netOrk's CI actually publishes: latest,
|
||||
# latest-dev, main-<sha>, feature-<branch> (feature/* only), or a release
|
||||
# version (X.Y.Z). See NetOrk/netork#55.
|
||||
VALID_VERSION_RE='^(latest|latest-dev|main-[0-9a-f]{7,40}|feature-[A-Za-z0-9._-]+|[0-9]+\.[0-9]+\.[0-9]+)$'
|
||||
|
||||
_validate_version() {
|
||||
local version="$1"
|
||||
if [[ ! "$version" =~ $VALID_VERSION_RE ]]; then
|
||||
echo "Refusing to deploy version '${version}': not a tag CI publishes (expected latest," \
|
||||
"latest-dev, main-<sha>, feature-<branch>, or X.Y.Z)." >&2
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
if [[ -n "$VERSION_OVERRIDE" ]]; then
|
||||
_validate_version "$VERSION_OVERRIDE"
|
||||
fi
|
||||
|
||||
if [[ -z "$REGISTRY_HOST" ]]; then
|
||||
echo "REGISTRY_HOST is not set (deploy.env or environment): nowhere to pull images from." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ ${#SERVERS[@]} -eq 0 ]]; then
|
||||
echo "No servers given: pass them as arguments or set DEPLOY_SERVERS in deploy.env." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
deploy_server() {
|
||||
local SERVER="$1"
|
||||
|
||||
# Per-server lookup — falls back to global values
|
||||
local KEY="${SERVER//./_}"
|
||||
local USER_VAR="REGISTRY_USER_${KEY}"
|
||||
local PASS_VAR="REGISTRY_PASSWORD_${KEY}"
|
||||
local VER_VAR="NETORK_VERSION_${KEY}"
|
||||
local SERVER_USER="${!USER_VAR:-${REGISTRY_USER}}"
|
||||
local SERVER_PASS="${!PASS_VAR:-${REGISTRY_PASSWORD}}"
|
||||
local SERVER_VERSION="${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}"
|
||||
_validate_version "$SERVER_VERSION" || return 1
|
||||
|
||||
local ENGINE_IMAGE="${REGISTRY_HOST}/netork/engine:${SERVER_VERSION}"
|
||||
|
||||
# The password goes over ssh's stdin. Inside the ssh argument it would be
|
||||
# part of the remote shell's command line, readable through `ps` by every
|
||||
# user on the host for as long as the login runs.
|
||||
echo "[${SERVER}] Logging in to registry..."
|
||||
# shellcheck disable=SC2029 # host and user are meant to expand here
|
||||
printf '%s\n' "$SERVER_PASS" \
|
||||
| ssh "$SERVER" "docker login '${REGISTRY_HOST}' -u '${SERVER_USER}' --password-stdin 2>&1" \
|
||||
| sed "s/^/[${SERVER}] /"
|
||||
|
||||
# The compose files are versioned with the images they start, so they are
|
||||
# taken from the engine image of this very tag rather than from whatever
|
||||
# checkout happens to run the deploy. Both are written to .new first and only
|
||||
# moved into place once both were read, so a failure leaves the previous pair
|
||||
# intact. An image built before the files moved into it has no /app/deploy/.
|
||||
echo "[${SERVER}] Fetching compose files from ${ENGINE_IMAGE}..."
|
||||
run_remote "$SERVER" "fetching compose files from ${ENGINE_IMAGE}" 'Status|Error' \
|
||||
"mkdir -p ~/netork && cd ~/netork && \
|
||||
docker pull '${ENGINE_IMAGE}' && \
|
||||
for f in ${COMPOSE_FILES}; do \
|
||||
docker run --rm --entrypoint cat '${ENGINE_IMAGE}' /app/deploy/\$f > \$f.new \
|
||||
|| { rm -f \$f.new; echo \"${ENGINE_IMAGE} carries no /app/deploy/\$f — the tag predates compose files in the engine image\"; exit 3; }; \
|
||||
done && \
|
||||
for f in ${COMPOSE_FILES}; do mv \$f.new \$f; done"
|
||||
|
||||
# Services running netOrk's own images. flower belongs here: it is the engine
|
||||
# image with a different command, so leaving it out let the monitoring UI run
|
||||
# nine days behind the workers it monitors, on an image that had since been
|
||||
# untagged. See NetOrk/netork#95.
|
||||
local SERVICES="netork-api netork-worker netork-poll-worker netork-ansible-worker netork-beat flower"
|
||||
if [[ " $UI_SERVER " == *" $SERVER "* ]]; then
|
||||
SERVICES="${SERVICES} netork-ui"
|
||||
fi
|
||||
|
||||
# Containers on upstream images. Reconciled, not force-recreated (below).
|
||||
# postgres and redis are deliberately absent: restarting a database on every
|
||||
# deploy would be worse than any compose change one could miss.
|
||||
local INFRA_SERVICES="registry apt-cacher-ng signal-api"
|
||||
|
||||
echo "[${SERVER}] Pulling images..."
|
||||
run_remote "$SERVER" "pulling images for ${SERVER_VERSION}" 'Pulled|Already|Error' \
|
||||
"cd ~/netork && \
|
||||
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
|
||||
docker compose -f docker-compose.yml -f docker-compose.registry.yml pull ${SERVICES}"
|
||||
|
||||
# --force-recreate is deliberate: `docker compose up -d` decides what to
|
||||
# replace from the service definition's config hash, and the image reference
|
||||
# (`.../ui:latest-dev`) does not change when the tag is moved to a new image.
|
||||
# A deploy then leaves the old container running while reporting success —
|
||||
# observed on netork-ui, which kept serving a stale image after its tag had
|
||||
# already advanced. Recreating unconditionally costs a restart per deploy,
|
||||
# which a deliberate rollout wants anyway. See NetOrk/netork#90.
|
||||
echo "[${SERVER}] Starting containers..."
|
||||
run_remote "$SERVER" "recreating ${SERVICES}" "" \
|
||||
"cd ~/netork && \
|
||||
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
|
||||
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
|
||||
up -d --no-deps --force-recreate ${SERVICES}"
|
||||
|
||||
# Infrastructure containers, deliberately left out of the force-recreate
|
||||
# above. Plain `up -d`: compose compares each service definition against the
|
||||
# running container and recreates only what actually changed, so an edit to a
|
||||
# healthcheck or an option in docker-compose.yml takes effect while a
|
||||
# container nobody touched is left alone. signal-api is the reason this is
|
||||
# not --force-recreate: it holds netOrk's Signal device link and recreating
|
||||
# it every deploy is churn nobody asked for. See NetOrk/netork#95.
|
||||
#
|
||||
# `|| true`: infrastructure is reconciled opportunistically and its failure
|
||||
# is not a reason to abandon a deploy that has already succeeded — but it
|
||||
# prints as a failure instead of scrolling past.
|
||||
echo "[${SERVER}] Reconciling infrastructure containers..."
|
||||
run_remote "$SERVER" "reconciling ${INFRA_SERVICES}" "" \
|
||||
"cd ~/netork && \
|
||||
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
|
||||
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
|
||||
up -d --no-deps ${INFRA_SERVICES}" || true
|
||||
|
||||
# Verify the containers actually run the image we just pulled. `docker
|
||||
# compose up -d` reports "Running" (not "Started") when it decides nothing
|
||||
# changed, and `pull` prints "Pulled" even when the tag was already local —
|
||||
# so a deploy that had no effect at all is indistinguishable from a real one
|
||||
# in the output above. Both the wrong-tag case (NETORK_VERSION pointing at a
|
||||
# tag CI never republished) and a silently failed pull land here.
|
||||
# See NetOrk/netork#88.
|
||||
echo "[${SERVER}] Verifying containers run the pulled image..."
|
||||
local VERIFY_PAIRS="netork-api:engine flower:engine"
|
||||
if [[ " $UI_SERVER " == *" $SERVER "* ]]; then
|
||||
VERIFY_PAIRS="${VERIFY_PAIRS} netork-ui:ui"
|
||||
fi
|
||||
|
||||
local MISMATCH
|
||||
MISMATCH=$(ssh -n "$SERVER" "cd ~/netork && for pair in ${VERIFY_PAIRS}; do \
|
||||
svc=\${pair%%:*}; repo=\${pair##*:}; \
|
||||
want=\$(docker image inspect --format '{{.Id}}' '${REGISTRY_HOST}/netork/'\${repo}':${SERVER_VERSION}' 2>/dev/null || true); \
|
||||
cid=\$(REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
|
||||
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -q \${svc} 2>/dev/null || true); \
|
||||
got=\$(docker inspect --format '{{.Image}}' \${cid} 2>/dev/null || true); \
|
||||
if [ -z \"\${want}\" ] || [ -z \"\${got}\" ] || [ \"\${want}\" != \"\${got}\" ]; then echo \${svc}; fi; \
|
||||
done" || true)
|
||||
|
||||
if [[ -n "$MISMATCH" ]]; then
|
||||
echo "[${SERVER}] ERROR: still running an older image after deploy: ${MISMATCH//$'\n'/ }" >&2
|
||||
echo "[${SERVER}] Wanted tag '${SERVER_VERSION}'. On main pushes CI publishes" >&2
|
||||
echo "[${SERVER}] 'latest-dev' and 'main-<sha>'; ':latest' only on version tags." >&2
|
||||
return 1
|
||||
fi
|
||||
echo "[${SERVER}] Verified: containers run ${SERVER_VERSION}."
|
||||
|
||||
# docker compose substitutes REGISTRY_HOST / NETORK_VERSION from the project
|
||||
# .env when they are not in the environment. This script passes both inline,
|
||||
# but a hand-typed `docker compose up -d flower` on the host needs them there
|
||||
# or dies on "invalid reference format". Values are rewritten in place, so
|
||||
# the file does not grow a new line per deploy.
|
||||
#
|
||||
# Only now. This used to run before the pull, so a deploy that failed on a tag
|
||||
# the registry does not have still left the host recording that tag — and the
|
||||
# next `docker compose up` a human ran there reached for an image that does
|
||||
# not exist. The file is a record of what this server runs, so it is written
|
||||
# once that is true: after the swap and after the image-id check agreed.
|
||||
echo "[${SERVER}] Recording registry settings in ~/netork/.env..."
|
||||
# shellcheck disable=SC2087 # registry and version expand here; \$ escapes run remotely
|
||||
ssh "$SERVER" "bash -s" <<REMOTE_ENV
|
||||
set -eu
|
||||
cd ~/netork
|
||||
touch .env
|
||||
# Append to a file that does not end in a newline, and the last line grows a
|
||||
# suffix instead of gaining a neighbour.
|
||||
[ ! -s .env ] || [ -z "\$(tail -c1 .env)" ] || printf '\n' >> .env
|
||||
for kv in "REGISTRY_HOST=${REGISTRY_HOST}" "NETORK_VERSION=${SERVER_VERSION}"; do
|
||||
key=\${kv%%=*}
|
||||
if grep -q "^\${key}=" .env; then
|
||||
sed -i "s|^\${key}=.*|\${kv}|" .env
|
||||
else
|
||||
printf '%s\n' "\${kv}" >> .env
|
||||
fi
|
||||
done
|
||||
REMOTE_ENV
|
||||
|
||||
# netOrk migrates *after* the swap — its API tolerates the gap. What it does
|
||||
# not tolerate is the migration failing unnoticed: the new code is already
|
||||
# serving, so a missing column is a 500 on every request that touches it
|
||||
# until someone looks. Loud, and non-zero. The revisions are the image's own.
|
||||
echo "[${SERVER}] Running migrations..."
|
||||
run_remote "$SERVER" "alembic upgrade head (containers are ALREADY swapped)" "" \
|
||||
"cd ~/netork && docker compose exec -T netork-api alembic upgrade head"
|
||||
|
||||
echo "[${SERVER}] Pruning unused images..."
|
||||
ssh -n "$SERVER" "docker image prune -af 2>&1 | tail -1" | sed "s/^/[${SERVER}] /"
|
||||
|
||||
echo "[${SERVER}] Done."
|
||||
}
|
||||
|
||||
# ── Dispatch ──────────────────────────────────────────────────────────────────
|
||||
|
||||
echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}"
|
||||
|
||||
export -f deploy_server _validate_version run_remote
|
||||
export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \
|
||||
VALID_VERSION_RE COMPOSE_FILES
|
||||
# Export all per-server credential vars so subshells can resolve them
|
||||
while IFS='=' read -r key _; do
|
||||
if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then
|
||||
export "${key?}"
|
||||
fi
|
||||
done < <(compgen -v)
|
||||
|
||||
PIDS=()
|
||||
for SERVER in "${SERVERS[@]}"; do
|
||||
deploy_server "$SERVER" &
|
||||
PIDS+=($!)
|
||||
done
|
||||
|
||||
FAILED=()
|
||||
for i in "${!PIDS[@]}"; do
|
||||
if ! wait "${PIDS[$i]}"; then
|
||||
FAILED+=("${SERVERS[$i]}")
|
||||
fi
|
||||
done
|
||||
|
||||
if [[ ${#FAILED[@]} -gt 0 ]]; then
|
||||
echo "FAILED on: ${FAILED[*]}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "All servers deployed successfully."
|
||||
Reference in New Issue
Block a user