#!/usr/bin/env bash # Deploy netOrk to one or more servers in parallel. # # Usage: # ./deploy.sh [--version=] [server1 server2 ...] # # --version overrides the per-server NETORK_VERSION from deploy.env for this # run only — useful to test one commit (--version=main-) without moving # a server to a different release channel. # # Default targets and registry credentials are read from deploy.env next to # this script (see deploy.env.example). DEPLOY_ENV_FILE points elsewhere. # # The servers pull pre-built images from REGISTRY_HOST. Nothing is built or # copied from a working tree: the compose files come out of the engine image # for the tag being deployed, and migrations run from that same image. set -euo pipefail HERE="$(cd "$(dirname "$0")" && pwd)" ENV_FILE="${DEPLOY_ENV_FILE:-${HERE}/deploy.env}" if [[ -f "$ENV_FILE" ]]; then # shellcheck source=/dev/null source "$ENV_FILE" fi DEPLOY_SERVERS="${DEPLOY_SERVERS:-}" UI_SERVER="${UI_SERVER:-}" REGISTRY_HOST="${REGISTRY_HOST:-}" REGISTRY_USER="${REGISTRY_USER:-}" REGISTRY_PASSWORD="${REGISTRY_PASSWORD:-}" NETORK_VERSION="${NETORK_VERSION:-latest}" # The compose files the engine image carries under /app/deploy/. COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml" read -ra SERVERS <<< "$DEPLOY_SERVERS" # Run one step on a server, and let its failure stop the deploy. # # ssh reports the exit status of the *remote* command — and when that command # ends in a pipe, the status is the last stage's, not the work's: # # ssh "$SERVER" "docker compose pull … 2>&1 | grep -E 'Pulled|Already|Error'" # # A failed pull prints a line containing `Error`, grep matches it and exits 0, # and the deploy walks on to swap the containers. `set -o pipefail` cannot help: # that pipeline ran in the remote shell, and from here the ssh succeeded. # # So the remote side stays pipe-free and the filtering happens here, where the # status is still the one the command returned. # # run_remote run_remote() { local server=$1 label=$2 pattern=$3 command=$4 local output status output="$(ssh -n "$server" "$command" 2>&1)" && status=0 || status=$? if [[ $status -ne 0 ]]; then printf '%s\n' "$output" | tail -25 | sed "s/^/[${server}] /" >&2 echo "[${server}] FAILED (exit ${status}): ${label}" >&2 return "$status" fi if [[ -n "$pattern" ]]; then printf '%s\n' "$output" | grep -E "$pattern" | sed "s/^/[${server}] /" || true else printf '%s\n' "$output" | tail -6 | sed "s/^/[${server}] /" fi } VERSION_OVERRIDE="" EXPLICIT_SERVERS=() i=1 while [[ $i -le $# ]]; do arg="${!i}" case "$arg" in --version) i=$(( i + 1 )) VERSION_OVERRIDE="${!i:-}" ;; --version=*) VERSION_OVERRIDE="${arg#--version=}" ;; -*) echo "Unknown option: ${arg}" >&2 exit 2 ;; *) EXPLICIT_SERVERS+=("$arg") ;; esac i=$(( i + 1 )) done if [[ ${#EXPLICIT_SERVERS[@]} -gt 0 ]]; then SERVERS=("${EXPLICIT_SERVERS[@]}") fi # ── Version validation ─────────────────────────────────────────────────────── # A deploy only pulls images — it never verifies the pulled tag actually # corresponds to a real, recently-built image for the commit/branch you think # you're deploying. A stale or accidentally-reused tag (or a typo'd branch # name that happens to collide with some ancient leftover tag) is pulled # silently, with the only symptom surfacing later at `alembic upgrade head`, # by which point containers are already switched over. Reject anything that # doesn't match a tag format netOrk's CI actually publishes: latest, # latest-dev, main-, feature- (feature/* only), or a release # version (X.Y.Z). See NetOrk/netork#55. VALID_VERSION_RE='^(latest|latest-dev|main-[0-9a-f]{7,40}|feature-[A-Za-z0-9._-]+|[0-9]+\.[0-9]+\.[0-9]+)$' _validate_version() { local version="$1" if [[ ! "$version" =~ $VALID_VERSION_RE ]]; then echo "Refusing to deploy version '${version}': not a tag CI publishes (expected latest," \ "latest-dev, main-, feature-, or X.Y.Z)." >&2 return 1 fi } if [[ -n "$VERSION_OVERRIDE" ]]; then _validate_version "$VERSION_OVERRIDE" fi if [[ -z "$REGISTRY_HOST" ]]; then echo "REGISTRY_HOST is not set (deploy.env or environment): nowhere to pull images from." >&2 exit 1 fi if [[ ${#SERVERS[@]} -eq 0 ]]; then echo "No servers given: pass them as arguments or set DEPLOY_SERVERS in deploy.env." >&2 exit 1 fi deploy_server() { local SERVER="$1" # Per-server lookup — falls back to global values local KEY="${SERVER//./_}" local USER_VAR="REGISTRY_USER_${KEY}" local PASS_VAR="REGISTRY_PASSWORD_${KEY}" local VER_VAR="NETORK_VERSION_${KEY}" local SERVER_USER="${!USER_VAR:-${REGISTRY_USER}}" local SERVER_PASS="${!PASS_VAR:-${REGISTRY_PASSWORD}}" local SERVER_VERSION="${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}" _validate_version "$SERVER_VERSION" || return 1 local ENGINE_IMAGE="${REGISTRY_HOST}/netork/engine:${SERVER_VERSION}" # The password goes over ssh's stdin. Inside the ssh argument it would be # part of the remote shell's command line, readable through `ps` by every # user on the host for as long as the login runs. echo "[${SERVER}] Logging in to registry..." # shellcheck disable=SC2029 # host and user are meant to expand here printf '%s\n' "$SERVER_PASS" \ | ssh "$SERVER" "docker login '${REGISTRY_HOST}' -u '${SERVER_USER}' --password-stdin 2>&1" \ | sed "s/^/[${SERVER}] /" # The compose files are versioned with the images they start, so they are # taken from the engine image of this very tag rather than from whatever # checkout happens to run the deploy. Both are written to .new first and only # moved into place once both were read, so a failure leaves the previous pair # intact. An image built before the files moved into it has no /app/deploy/. echo "[${SERVER}] Fetching compose files from ${ENGINE_IMAGE}..." run_remote "$SERVER" "fetching compose files from ${ENGINE_IMAGE}" 'Status|Error' \ "mkdir -p ~/netork && cd ~/netork && \ docker pull '${ENGINE_IMAGE}' && \ for f in ${COMPOSE_FILES}; do \ docker run --rm --entrypoint cat '${ENGINE_IMAGE}' /app/deploy/\$f > \$f.new \ || { rm -f \$f.new; echo \"${ENGINE_IMAGE} carries no /app/deploy/\$f — the tag predates compose files in the engine image\"; exit 3; }; \ done && \ for f in ${COMPOSE_FILES}; do mv \$f.new \$f; done" # Services running netOrk's own images. flower belongs here: it is the engine # image with a different command, so leaving it out let the monitoring UI run # nine days behind the workers it monitors, on an image that had since been # untagged. See NetOrk/netork#95. local SERVICES="netork-api netork-worker netork-poll-worker netork-ansible-worker netork-beat flower" if [[ " $UI_SERVER " == *" $SERVER "* ]]; then SERVICES="${SERVICES} netork-ui" fi # Containers on upstream images. Reconciled, not force-recreated (below). # postgres and redis are deliberately absent: restarting a database on every # deploy would be worse than any compose change one could miss. local INFRA_SERVICES="registry apt-cacher-ng signal-api" echo "[${SERVER}] Pulling images..." run_remote "$SERVER" "pulling images for ${SERVER_VERSION}" 'Pulled|Already|Error' \ "cd ~/netork && \ REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ docker compose -f docker-compose.yml -f docker-compose.registry.yml pull ${SERVICES}" # --force-recreate is deliberate: `docker compose up -d` decides what to # replace from the service definition's config hash, and the image reference # (`.../ui:latest-dev`) does not change when the tag is moved to a new image. # A deploy then leaves the old container running while reporting success — # observed on netork-ui, which kept serving a stale image after its tag had # already advanced. Recreating unconditionally costs a restart per deploy, # which a deliberate rollout wants anyway. See NetOrk/netork#90. echo "[${SERVER}] Starting containers..." run_remote "$SERVER" "recreating ${SERVICES}" "" \ "cd ~/netork && \ REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ docker compose -f docker-compose.yml -f docker-compose.registry.yml \ up -d --no-deps --force-recreate ${SERVICES}" # Infrastructure containers, deliberately left out of the force-recreate # above. Plain `up -d`: compose compares each service definition against the # running container and recreates only what actually changed, so an edit to a # healthcheck or an option in docker-compose.yml takes effect while a # container nobody touched is left alone. signal-api is the reason this is # not --force-recreate: it holds netOrk's Signal device link and recreating # it every deploy is churn nobody asked for. See NetOrk/netork#95. # # `|| true`: infrastructure is reconciled opportunistically and its failure # is not a reason to abandon a deploy that has already succeeded — but it # prints as a failure instead of scrolling past. echo "[${SERVER}] Reconciling infrastructure containers..." run_remote "$SERVER" "reconciling ${INFRA_SERVICES}" "" \ "cd ~/netork && \ REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ docker compose -f docker-compose.yml -f docker-compose.registry.yml \ up -d --no-deps ${INFRA_SERVICES}" || true # Verify the containers actually run the image we just pulled. `docker # compose up -d` reports "Running" (not "Started") when it decides nothing # changed, and `pull` prints "Pulled" even when the tag was already local — # so a deploy that had no effect at all is indistinguishable from a real one # in the output above. Both the wrong-tag case (NETORK_VERSION pointing at a # tag CI never republished) and a silently failed pull land here. # See NetOrk/netork#88. echo "[${SERVER}] Verifying containers run the pulled image..." local VERIFY_PAIRS="netork-api:engine flower:engine" if [[ " $UI_SERVER " == *" $SERVER "* ]]; then VERIFY_PAIRS="${VERIFY_PAIRS} netork-ui:ui" fi local MISMATCH MISMATCH=$(ssh -n "$SERVER" "cd ~/netork && for pair in ${VERIFY_PAIRS}; do \ svc=\${pair%%:*}; repo=\${pair##*:}; \ want=\$(docker image inspect --format '{{.Id}}' '${REGISTRY_HOST}/netork/'\${repo}':${SERVER_VERSION}' 2>/dev/null || true); \ cid=\$(REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -q \${svc} 2>/dev/null || true); \ got=\$(docker inspect --format '{{.Image}}' \${cid} 2>/dev/null || true); \ if [ -z \"\${want}\" ] || [ -z \"\${got}\" ] || [ \"\${want}\" != \"\${got}\" ]; then echo \${svc}; fi; \ done" || true) if [[ -n "$MISMATCH" ]]; then echo "[${SERVER}] ERROR: still running an older image after deploy: ${MISMATCH//$'\n'/ }" >&2 echo "[${SERVER}] Wanted tag '${SERVER_VERSION}'. On main pushes CI publishes" >&2 echo "[${SERVER}] 'latest-dev' and 'main-'; ':latest' only on version tags." >&2 return 1 fi echo "[${SERVER}] Verified: containers run ${SERVER_VERSION}." # docker compose substitutes REGISTRY_HOST / NETORK_VERSION from the project # .env when they are not in the environment. This script passes both inline, # but a hand-typed `docker compose up -d flower` on the host needs them there # or dies on "invalid reference format". Values are rewritten in place, so # the file does not grow a new line per deploy. # # Only now. This used to run before the pull, so a deploy that failed on a tag # the registry does not have still left the host recording that tag — and the # next `docker compose up` a human ran there reached for an image that does # not exist. The file is a record of what this server runs, so it is written # once that is true: after the swap and after the image-id check agreed. echo "[${SERVER}] Recording registry settings in ~/netork/.env..." # shellcheck disable=SC2087 # registry and version expand here; \$ escapes run remotely ssh "$SERVER" "bash -s" <> .env for kv in "REGISTRY_HOST=${REGISTRY_HOST}" "NETORK_VERSION=${SERVER_VERSION}"; do key=\${kv%%=*} if grep -q "^\${key}=" .env; then sed -i "s|^\${key}=.*|\${kv}|" .env else printf '%s\n' "\${kv}" >> .env fi done REMOTE_ENV # netOrk migrates *after* the swap — its API tolerates the gap. What it does # not tolerate is the migration failing unnoticed: the new code is already # serving, so a missing column is a 500 on every request that touches it # until someone looks. Loud, and non-zero. The revisions are the image's own. echo "[${SERVER}] Running migrations..." run_remote "$SERVER" "alembic upgrade head (containers are ALREADY swapped)" "" \ "cd ~/netork && docker compose exec -T netork-api alembic upgrade head" echo "[${SERVER}] Pruning unused images..." ssh -n "$SERVER" "docker image prune -af 2>&1 | tail -1" | sed "s/^/[${SERVER}] /" echo "[${SERVER}] Done." } # ── Dispatch ────────────────────────────────────────────────────────────────── echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}" export -f deploy_server _validate_version run_remote export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \ VALID_VERSION_RE COMPOSE_FILES # Export all per-server credential vars so subshells can resolve them while IFS='=' read -r key _; do if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then export "${key?}" fi done < <(compgen -v) PIDS=() for SERVER in "${SERVERS[@]}"; do deploy_server "$SERVER" & PIDS+=($!) done FAILED=() for i in "${!PIDS[@]}"; do if ! wait "${PIDS[$i]}"; then FAILED+=("${SERVERS[$i]}") fi done if [[ ${#FAILED[@]} -gt 0 ]]; then echo "FAILED on: ${FAILED[*]}" >&2 exit 1 fi echo "All servers deployed successfully."