fix: one deploy per host at a time, under a lock the host frees on its own
CI / check (pull_request) Successful in 17s

Several sessions deploy to the same test server, and two runs used to
overlap. On 2026-10-05 two `up -d --force-recreate` runs recreated each
other's containers and the API was down for a minute (#1). On 2026-10-06
two runs renamed each other's *.new compose files and one broke off; with
two different tags, one tag's compose files could have started the
other's images (#3).

- Each deploy first takes flock on ~/netork/.deploy.lock on the host. An
  ssh session holds it: the remote side takes the lock on fd 9, reports
  LOCKED and waits on its stdin, so ending the session frees it, whether
  the deploy finished, failed, was interrupted or lost its connection.
  Checked over real ssh on .50, including a client killed with -9.
- A second deploy prints who holds the lock, since when and with which
  tag, and waits up to DEPLOY_LOCK_WAIT seconds (default 900); then it
  gives up without touching the host.
- A host without flock is deployed without the lock, with a warning.
- deploy_server is now the lock around deploy_steps, the old body.

Closes #1
Closes #3
This commit is contained in:
Christian Manivong
2026-10-07 06:42:13 +02:00
parent 6444aa97f3
commit ec93619f8f
4 changed files with 270 additions and 3 deletions
+75 -2
View File
@@ -34,6 +34,9 @@ NETORK_VERSION="${NETORK_VERSION:-latest}"
# The compose files the engine image carries under /app/deploy/.
COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml"
# How long a deploy waits for another one on the same host before giving up.
DEPLOY_LOCK_WAIT="${DEPLOY_LOCK_WAIT:-900}"
read -ra SERVERS <<< "$DEPLOY_SERVERS"
# Run one step on a server, and let its failure stop the deploy.
@@ -70,6 +73,63 @@ run_remote() {
fi
}
# ── One deploy per host at a time ────────────────────────────────────────────
# Several sessions deploy to the same test server, and two runs used to overlap:
# on 2026-10-05 two `up -d --force-recreate` recreated each other's containers,
# on 2026-10-06 two runs renamed each other's *.new compose files, and with two
# tags one tag's compose files could start the other's images (NetOrk/deploy#1,
# #3). So each deploy first takes flock on ~/netork/.deploy.lock on the host.
#
# An ssh session holds it: the remote side opens the lock file on fd 9, takes
# the lock, says LOCKED, and then waits on its stdin. Closing that stdin ends
# the session and frees the lock, and so does anything else that ends it: the
# deploy failing, being interrupted, or losing the network. No stale lock file
# can outlive the deploy that took it.
#
# A host without flock (util-linux) is deployed without the lock, with a
# warning, rather than becoming undeployable.
#
# hold_host_lock <server> <who and what, for the next deploy to read>
hold_host_lock() {
local server=$1 info=$2 line
# shellcheck disable=SC2029 # the wait and the info are meant to expand here
coproc HOST_LOCK {
ssh "$server" "mkdir -p ~/netork && cd ~/netork && exec 9>.deploy.lock && \
if ! command -v flock >/dev/null 2>&1; then echo NOLOCK; exec cat >/dev/null; fi; \
if ! flock -n 9; then \
echo \"BUSY \$(cat .deploy.lock.info 2>/dev/null)\"; \
flock -w ${DEPLOY_LOCK_WAIT} 9 || { echo TIMEOUT; exit 1; }; \
fi; \
echo '${info}' > .deploy.lock.info; echo LOCKED; exec cat >/dev/null" 2>&1
}
while IFS= read -r -t "$(( DEPLOY_LOCK_WAIT + 60 ))" line <&"${HOST_LOCK[0]}"; do
case "$line" in
LOCKED) return 0 ;;
NOLOCK)
echo "[${server}] WARNING: no flock on ${server}; deploying without the host lock." >&2
return 0
;;
BUSY*)
echo "[${server}] another deploy holds ${server}: ${line#BUSY } — waiting up to ${DEPLOY_LOCK_WAIT}s..." >&2
;;
TIMEOUT)
echo "[${server}] the deploy lock on ${server} is still held after ${DEPLOY_LOCK_WAIT}s; giving up." >&2
return 1
;;
*) echo "[${server}] ${line}" >&2 ;;
esac
done
echo "[${server}] could not take the deploy lock on ${server} (the ssh session ended)." >&2
return 1
}
# Free the lock: closing the session's stdin ends the remote side.
release_host_lock() {
local fd="${HOST_LOCK[1]}" pid="${HOST_LOCK_PID}"
exec {fd}>&-
wait "$pid" 2>/dev/null || true
}
VERSION_OVERRIDE=""
EXPLICIT_SERVERS=()
i=1
@@ -128,7 +188,20 @@ if [[ ${#SERVERS[@]} -eq 0 ]]; then
exit 1
fi
# One server: under its host lock, the steps below.
#
# A step that fails stops the subshell (set -e) before release_host_lock runs.
# That is fine: the subshell's end closes the lock session's stdin all the same.
deploy_server() {
local SERVER="$1" KEY="${1//./_}" VER_VAR
VER_VAR="NETORK_VERSION_${KEY}"
hold_host_lock "$SERVER" \
"since $(date -u +%Y-%m-%dT%H:%M:%SZ), by ${USER:-?}@$(hostname -s), deploying ${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}"
deploy_steps "$SERVER"
release_host_lock
}
deploy_steps() {
local SERVER="$1"
# Per-server lookup — falls back to global values
@@ -317,9 +390,9 @@ REMOTE_ENV
echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}"
export -f deploy_server _validate_version run_remote
export -f deploy_server deploy_steps hold_host_lock release_host_lock _validate_version run_remote
export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \
VALID_VERSION_RE COMPOSE_FILES
VALID_VERSION_RE COMPOSE_FILES DEPLOY_LOCK_WAIT
# Export all per-server credential vars so subshells can resolve them
while IFS='=' read -r key _; do
if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then