fix: one deploy per host at a time, under a lock the host frees on its own #4

Merged
christianmanivong merged 1 commits from fix/host-deploy-lock into main 2026-10-07 04:43:24 +00:00
4 changed files with 270 additions and 3 deletions
+7
View File
@@ -41,6 +41,7 @@ elsewhere.
| `NETORK_VERSION` | Tag to deploy (default `latest`) |
| `REGISTRY_USER`, `REGISTRY_PASSWORD` | Registry login used on every server |
| `REGISTRY_USER_<server>`, `REGISTRY_PASSWORD_<server>`, `NETORK_VERSION_<server>` | Per-server overrides. `<server>` has its dots replaced by underscores, e.g. `_10_0_0_2` |
| `DEPLOY_LOCK_WAIT` | Seconds to wait for another deploy to the same host (default 900) |
## Usage
@@ -65,6 +66,12 @@ started earlier pulls whatever image the registry held before, which is stale.
## What a deploy does, per server
0. Takes the host's deploy lock, `flock` on `~/netork/.deploy.lock`, and holds it until
the deploy ends. A second deploy to the same host waits for it and says who holds
it, since when, and which tag they are deploying. It gives up after 15 minutes
(`DEPLOY_LOCK_WAIT`, in seconds) without touching the host. The lock belongs to an
ssh session, so a deploy that fails, is interrupted or loses its connection frees it
on its own. A host without `flock` is deployed without the lock, with a warning.
1. Logs in to the registry. The password travels over ssh's stdin, never on a command
line.
2. Pulls the engine image and copies `docker-compose.yml` and
+75 -2
View File
@@ -34,6 +34,9 @@ NETORK_VERSION="${NETORK_VERSION:-latest}"
# The compose files the engine image carries under /app/deploy/.
COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml"
# How long a deploy waits for another one on the same host before giving up.
DEPLOY_LOCK_WAIT="${DEPLOY_LOCK_WAIT:-900}"
read -ra SERVERS <<< "$DEPLOY_SERVERS"
# Run one step on a server, and let its failure stop the deploy.
@@ -70,6 +73,63 @@ run_remote() {
fi
}
# ── One deploy per host at a time ────────────────────────────────────────────
# Several sessions deploy to the same test server, and two runs used to overlap:
# on 2026-10-05 two `up -d --force-recreate` recreated each other's containers,
# on 2026-10-06 two runs renamed each other's *.new compose files, and with two
# tags one tag's compose files could start the other's images (NetOrk/deploy#1,
# #3). So each deploy first takes flock on ~/netork/.deploy.lock on the host.
#
# An ssh session holds it: the remote side opens the lock file on fd 9, takes
# the lock, says LOCKED, and then waits on its stdin. Closing that stdin ends
# the session and frees the lock, and so does anything else that ends it: the
# deploy failing, being interrupted, or losing the network. No stale lock file
# can outlive the deploy that took it.
#
# A host without flock (util-linux) is deployed without the lock, with a
# warning, rather than becoming undeployable.
#
# hold_host_lock <server> <who and what, for the next deploy to read>
hold_host_lock() {
local server=$1 info=$2 line
# shellcheck disable=SC2029 # the wait and the info are meant to expand here
coproc HOST_LOCK {
ssh "$server" "mkdir -p ~/netork && cd ~/netork && exec 9>.deploy.lock && \
if ! command -v flock >/dev/null 2>&1; then echo NOLOCK; exec cat >/dev/null; fi; \
if ! flock -n 9; then \
echo \"BUSY \$(cat .deploy.lock.info 2>/dev/null)\"; \
flock -w ${DEPLOY_LOCK_WAIT} 9 || { echo TIMEOUT; exit 1; }; \
fi; \
echo '${info}' > .deploy.lock.info; echo LOCKED; exec cat >/dev/null" 2>&1
}
while IFS= read -r -t "$(( DEPLOY_LOCK_WAIT + 60 ))" line <&"${HOST_LOCK[0]}"; do
case "$line" in
LOCKED) return 0 ;;
NOLOCK)
echo "[${server}] WARNING: no flock on ${server}; deploying without the host lock." >&2
return 0
;;
BUSY*)
echo "[${server}] another deploy holds ${server}: ${line#BUSY } — waiting up to ${DEPLOY_LOCK_WAIT}s..." >&2
;;
TIMEOUT)
echo "[${server}] the deploy lock on ${server} is still held after ${DEPLOY_LOCK_WAIT}s; giving up." >&2
return 1
;;
*) echo "[${server}] ${line}" >&2 ;;
esac
done
echo "[${server}] could not take the deploy lock on ${server} (the ssh session ended)." >&2
return 1
}
# Free the lock: closing the session's stdin ends the remote side.
release_host_lock() {
local fd="${HOST_LOCK[1]}" pid="${HOST_LOCK_PID}"
exec {fd}>&-
wait "$pid" 2>/dev/null || true
}
VERSION_OVERRIDE=""
EXPLICIT_SERVERS=()
i=1
@@ -128,7 +188,20 @@ if [[ ${#SERVERS[@]} -eq 0 ]]; then
exit 1
fi
# One server: under its host lock, the steps below.
#
# A step that fails stops the subshell (set -e) before release_host_lock runs.
# That is fine: the subshell's end closes the lock session's stdin all the same.
deploy_server() {
local SERVER="$1" KEY="${1//./_}" VER_VAR
VER_VAR="NETORK_VERSION_${KEY}"
hold_host_lock "$SERVER" \
"since $(date -u +%Y-%m-%dT%H:%M:%SZ), by ${USER:-?}@$(hostname -s), deploying ${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}"
deploy_steps "$SERVER"
release_host_lock
}
deploy_steps() {
local SERVER="$1"
# Per-server lookup — falls back to global values
@@ -317,9 +390,9 @@ REMOTE_ENV
echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}"
export -f deploy_server _validate_version run_remote
export -f deploy_server deploy_steps hold_host_lock release_host_lock _validate_version run_remote
export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \
VALID_VERSION_RE COMPOSE_FILES
VALID_VERSION_RE COMPOSE_FILES DEPLOY_LOCK_WAIT
# Export all per-server credential vars so subshells can resolve them
while IFS='=' read -r key _; do
if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then
+185
View File
@@ -0,0 +1,185 @@
"""One deploy per host at a time (NetOrk/deploy#1, #3).
Several sessions deploy to the same test server, and nothing stopped two runs
from overlapping:
- 2026-10-05: two `up -d --force-recreate` runs recreated each other's
containers. The API was down for a minute, with leftover `<id>_netork-…`
containers.
- 2026-10-06: two runs renamed each other's `*.new` compose files, and one broke
off at `mv: cannot stat 'docker-compose.yml.new'`. With two different tags,
one tag's compose files could have started the other tag's images.
Now each deploy first takes `flock` on `~/netork/.deploy.lock` on the host,
through an ssh session that holds it until the deploy ends. A second deploy
says who holds it, since when and with which tag, and waits. The lock lives as
long as that session does, so a deploy that dies, or a laptop that loses its
network, frees it on its own.
The "host" here is this machine with `HOME` in a temporary directory: the fake
`ssh` runs the lock holder for real, so the lock is a real `flock`.
"""
from __future__ import annotations
import os
import subprocess
import time
from pathlib import Path
import pytest
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
_FAKE_SSH = """#!/bin/sh
for arg in "$@"; do cmd="$arg"; done
case "$cmd" in
*.deploy.lock*)
# The lock holder runs for real, on this "host".
if [ -n "$FAKE_NO_FLOCK" ]; then exec env PATH="$FAKE_NO_FLOCK" sh -c "$cmd"; fi
exec sh -c "$cmd"
;;
esac
cat > /dev/null
case "$cmd" in
*--force-recreate*)
echo "start $RUN_ID" >> "$FAKE_LOG"
sleep "${FAKE_RECREATE_SLEEP:-0}"
echo "end $RUN_ID" >> "$FAKE_LOG"
if [ -n "$FAKE_RECREATE_FAILS" ]; then echo "Error response from daemon"; exit 1; fi
;;
esac
exit 0
"""
@pytest.fixture
def host(tmp_path: Path) -> dict[str, Path]:
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
fake = bin_dir / "ssh"
fake.write_text(_FAKE_SSH)
fake.chmod(0o755)
home = tmp_path / "home"
home.mkdir()
return {"bin": bin_dir, "home": home, "log": tmp_path / "log"}
def _start(host: dict[str, Path], run_id: str, **env: str) -> subprocess.Popen[str]:
return subprocess.Popen(
["bash", str(DEPLOY), "testhost"],
env={
"PATH": f"{host['bin']}:{os.environ['PATH']}",
"HOME": str(host["home"]),
"USER": "tester",
"DEPLOY_ENV_FILE": "/dev/null",
"REGISTRY_HOST": "registry.example",
"NETORK_VERSION": "latest-dev",
"DEPLOY_RECREATE_RETRY_DELAY": "0",
"RUN_ID": run_id,
"FAKE_LOG": str(host["log"]),
**env,
},
stdin=subprocess.DEVNULL,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
)
def _run(host: dict[str, Path], run_id: str, **env: str) -> subprocess.CompletedProcess[str]:
proc = _start(host, run_id, **env)
out, err = proc.communicate(timeout=60)
return subprocess.CompletedProcess(proc.args, proc.returncode, out, err)
def _lock_file(host: dict[str, Path]) -> Path:
return host["home"] / "netork" / ".deploy.lock"
def _lock_is_free(host: dict[str, Path], seconds: float = 5.0) -> bool:
deadline = time.monotonic() + seconds
while time.monotonic() < deadline:
probe = subprocess.run(["flock", "-n", str(_lock_file(host)), "true"], check=False)
if probe.returncode == 0:
return True
time.sleep(0.05)
return False
def _events(host: dict[str, Path]) -> list[str]:
return host["log"].read_text().split("\n")[:-1] if host["log"].exists() else []
def _wait_for(host: dict[str, Path], event: str, seconds: float = 20.0) -> None:
deadline = time.monotonic() + seconds
while time.monotonic() < deadline:
if event in _events(host):
return
time.sleep(0.05)
raise AssertionError(f"{event!r} never happened: {_events(host)}")
def test_two_deploys_to_one_host_take_turns(host) -> None:
first = _start(host, "A", FAKE_RECREATE_SLEEP="1.5")
try:
_wait_for(host, "start A")
second = _run(host, "B")
finally:
_out, err = first.communicate(timeout=60)
assert first.returncode == 0, err
assert second.returncode == 0, second.stderr
assert _events(host) == ["start A", "end A", "start B", "end B"]
assert "another deploy holds testhost" in second.stderr
assert "deploying latest-dev" in second.stderr
assert "by tester@" in second.stderr
def test_the_lock_is_free_once_a_deploy_is_done(host) -> None:
result = _run(host, "A")
assert result.returncode == 0, result.stderr
assert _lock_file(host).exists()
assert _lock_is_free(host)
def test_a_failed_deploy_frees_the_lock_too(host) -> None:
result = _run(host, "A", FAKE_RECREATE_FAILS="1")
assert result.returncode != 0
assert _lock_is_free(host)
def test_a_deploy_that_waits_too_long_gives_up_without_touching_the_host(host) -> None:
lock = _lock_file(host)
lock.parent.mkdir(parents=True)
holder = subprocess.Popen(["flock", str(lock), "sleep", "30"])
try:
time.sleep(0.2)
result = _run(host, "B", DEPLOY_LOCK_WAIT="1")
finally:
holder.kill()
holder.wait()
assert result.returncode != 0
assert "still held after 1s" in result.stderr
assert "FAILED on: testhost" in result.stderr
assert _events(host) == []
def test_a_host_without_flock_is_deployed_with_a_warning(host, tmp_path) -> None:
"""flock is util-linux; a host without it should not become undeployable."""
no_flock = tmp_path / "no-flock"
no_flock.mkdir()
for tool in ("sh", "mkdir", "cat", "date"):
found = subprocess.run(
["sh", "-c", f"command -v {tool}"], capture_output=True, text=True, check=True
)
(no_flock / tool).symlink_to(found.stdout.strip())
result = _run(host, "A", FAKE_NO_FLOCK=str(no_flock))
assert result.returncode == 0, result.stderr
assert "no flock on testhost" in result.stderr
assert _events(host) == ["start A", "end A"]
+3 -1
View File
@@ -25,9 +25,11 @@ import pytest
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
_FAKE_SSH = """#!/bin/sh
for arg in "$@"; do cmd="$arg"; done
# The host lock's holder runs for real, here (test_deploy_host_lock.py).
case "$cmd" in *.deploy.lock*) exec sh -c "$cmd" ;; esac
# Swallow stdin (docker login, the .env heredoc), then answer by command.
cat > /dev/null
for arg in "$@"; do cmd="$arg"; done
case "$cmd" in
*--force-recreate*)
n=$(cat "{state}/recreates" 2>/dev/null || echo 0)