diff --git a/README.md b/README.md index 15bc601..9eb2b88 100644 --- a/README.md +++ b/README.md @@ -41,6 +41,7 @@ elsewhere. | `NETORK_VERSION` | Tag to deploy (default `latest`) | | `REGISTRY_USER`, `REGISTRY_PASSWORD` | Registry login used on every server | | `REGISTRY_USER_`, `REGISTRY_PASSWORD_`, `NETORK_VERSION_` | Per-server overrides. `` has its dots replaced by underscores, e.g. `_10_0_0_2` | +| `DEPLOY_LOCK_WAIT` | Seconds to wait for another deploy to the same host (default 900) | ## Usage @@ -65,6 +66,12 @@ started earlier pulls whatever image the registry held before, which is stale. ## What a deploy does, per server +0. Takes the host's deploy lock, `flock` on `~/netork/.deploy.lock`, and holds it until + the deploy ends. A second deploy to the same host waits for it and says who holds + it, since when, and which tag they are deploying. It gives up after 15 minutes + (`DEPLOY_LOCK_WAIT`, in seconds) without touching the host. The lock belongs to an + ssh session, so a deploy that fails, is interrupted or loses its connection frees it + on its own. A host without `flock` is deployed without the lock, with a warning. 1. Logs in to the registry. The password travels over ssh's stdin, never on a command line. 2. Pulls the engine image and copies `docker-compose.yml` and diff --git a/deploy.sh b/deploy.sh index dddd34e..2c72f73 100644 --- a/deploy.sh +++ b/deploy.sh @@ -34,6 +34,9 @@ NETORK_VERSION="${NETORK_VERSION:-latest}" # The compose files the engine image carries under /app/deploy/. COMPOSE_FILES="docker-compose.yml docker-compose.registry.yml" +# How long a deploy waits for another one on the same host before giving up. +DEPLOY_LOCK_WAIT="${DEPLOY_LOCK_WAIT:-900}" + read -ra SERVERS <<< "$DEPLOY_SERVERS" # Run one step on a server, and let its failure stop the deploy. @@ -70,6 +73,63 @@ run_remote() { fi } +# ── One deploy per host at a time ──────────────────────────────────────────── +# Several sessions deploy to the same test server, and two runs used to overlap: +# on 2026-10-05 two `up -d --force-recreate` recreated each other's containers, +# on 2026-10-06 two runs renamed each other's *.new compose files, and with two +# tags one tag's compose files could start the other's images (NetOrk/deploy#1, +# #3). So each deploy first takes flock on ~/netork/.deploy.lock on the host. +# +# An ssh session holds it: the remote side opens the lock file on fd 9, takes +# the lock, says LOCKED, and then waits on its stdin. Closing that stdin ends +# the session and frees the lock, and so does anything else that ends it: the +# deploy failing, being interrupted, or losing the network. No stale lock file +# can outlive the deploy that took it. +# +# A host without flock (util-linux) is deployed without the lock, with a +# warning, rather than becoming undeployable. +# +# hold_host_lock +hold_host_lock() { + local server=$1 info=$2 line + # shellcheck disable=SC2029 # the wait and the info are meant to expand here + coproc HOST_LOCK { + ssh "$server" "mkdir -p ~/netork && cd ~/netork && exec 9>.deploy.lock && \ + if ! command -v flock >/dev/null 2>&1; then echo NOLOCK; exec cat >/dev/null; fi; \ + if ! flock -n 9; then \ + echo \"BUSY \$(cat .deploy.lock.info 2>/dev/null)\"; \ + flock -w ${DEPLOY_LOCK_WAIT} 9 || { echo TIMEOUT; exit 1; }; \ + fi; \ + echo '${info}' > .deploy.lock.info; echo LOCKED; exec cat >/dev/null" 2>&1 + } + while IFS= read -r -t "$(( DEPLOY_LOCK_WAIT + 60 ))" line <&"${HOST_LOCK[0]}"; do + case "$line" in + LOCKED) return 0 ;; + NOLOCK) + echo "[${server}] WARNING: no flock on ${server}; deploying without the host lock." >&2 + return 0 + ;; + BUSY*) + echo "[${server}] another deploy holds ${server}: ${line#BUSY } — waiting up to ${DEPLOY_LOCK_WAIT}s..." >&2 + ;; + TIMEOUT) + echo "[${server}] the deploy lock on ${server} is still held after ${DEPLOY_LOCK_WAIT}s; giving up." >&2 + return 1 + ;; + *) echo "[${server}] ${line}" >&2 ;; + esac + done + echo "[${server}] could not take the deploy lock on ${server} (the ssh session ended)." >&2 + return 1 +} + +# Free the lock: closing the session's stdin ends the remote side. +release_host_lock() { + local fd="${HOST_LOCK[1]}" pid="${HOST_LOCK_PID}" + exec {fd}>&- + wait "$pid" 2>/dev/null || true +} + VERSION_OVERRIDE="" EXPLICIT_SERVERS=() i=1 @@ -128,7 +188,20 @@ if [[ ${#SERVERS[@]} -eq 0 ]]; then exit 1 fi +# One server: under its host lock, the steps below. +# +# A step that fails stops the subshell (set -e) before release_host_lock runs. +# That is fine: the subshell's end closes the lock session's stdin all the same. deploy_server() { + local SERVER="$1" KEY="${1//./_}" VER_VAR + VER_VAR="NETORK_VERSION_${KEY}" + hold_host_lock "$SERVER" \ + "since $(date -u +%Y-%m-%dT%H:%M:%SZ), by ${USER:-?}@$(hostname -s), deploying ${VERSION_OVERRIDE:-${!VER_VAR:-${NETORK_VERSION}}}" + deploy_steps "$SERVER" + release_host_lock +} + +deploy_steps() { local SERVER="$1" # Per-server lookup — falls back to global values @@ -317,9 +390,9 @@ REMOTE_ENV echo "Deploying from ${REGISTRY_HOST} to: ${SERVERS[*]}" -export -f deploy_server _validate_version run_remote +export -f deploy_server deploy_steps hold_host_lock release_host_lock _validate_version run_remote export UI_SERVER REGISTRY_HOST REGISTRY_USER REGISTRY_PASSWORD NETORK_VERSION VERSION_OVERRIDE \ - VALID_VERSION_RE COMPOSE_FILES + VALID_VERSION_RE COMPOSE_FILES DEPLOY_LOCK_WAIT # Export all per-server credential vars so subshells can resolve them while IFS='=' read -r key _; do if [[ "$key" =~ ^(REGISTRY_(USER|PASSWORD)|NETORK_VERSION)_ ]]; then diff --git a/tests/test_deploy_host_lock.py b/tests/test_deploy_host_lock.py new file mode 100644 index 0000000..9be87b9 --- /dev/null +++ b/tests/test_deploy_host_lock.py @@ -0,0 +1,185 @@ +"""One deploy per host at a time (NetOrk/deploy#1, #3). + +Several sessions deploy to the same test server, and nothing stopped two runs +from overlapping: + +- 2026-10-05: two `up -d --force-recreate` runs recreated each other's + containers. The API was down for a minute, with leftover `_netork-…` + containers. +- 2026-10-06: two runs renamed each other's `*.new` compose files, and one broke + off at `mv: cannot stat 'docker-compose.yml.new'`. With two different tags, + one tag's compose files could have started the other tag's images. + +Now each deploy first takes `flock` on `~/netork/.deploy.lock` on the host, +through an ssh session that holds it until the deploy ends. A second deploy +says who holds it, since when and with which tag, and waits. The lock lives as +long as that session does, so a deploy that dies, or a laptop that loses its +network, frees it on its own. + +The "host" here is this machine with `HOME` in a temporary directory: the fake +`ssh` runs the lock holder for real, so the lock is a real `flock`. +""" + +from __future__ import annotations + +import os +import subprocess +import time +from pathlib import Path + +import pytest + +DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh" + +_FAKE_SSH = """#!/bin/sh +for arg in "$@"; do cmd="$arg"; done +case "$cmd" in + *.deploy.lock*) + # The lock holder runs for real, on this "host". + if [ -n "$FAKE_NO_FLOCK" ]; then exec env PATH="$FAKE_NO_FLOCK" sh -c "$cmd"; fi + exec sh -c "$cmd" + ;; +esac +cat > /dev/null +case "$cmd" in + *--force-recreate*) + echo "start $RUN_ID" >> "$FAKE_LOG" + sleep "${FAKE_RECREATE_SLEEP:-0}" + echo "end $RUN_ID" >> "$FAKE_LOG" + if [ -n "$FAKE_RECREATE_FAILS" ]; then echo "Error response from daemon"; exit 1; fi + ;; +esac +exit 0 +""" + + +@pytest.fixture +def host(tmp_path: Path) -> dict[str, Path]: + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + fake = bin_dir / "ssh" + fake.write_text(_FAKE_SSH) + fake.chmod(0o755) + home = tmp_path / "home" + home.mkdir() + return {"bin": bin_dir, "home": home, "log": tmp_path / "log"} + + +def _start(host: dict[str, Path], run_id: str, **env: str) -> subprocess.Popen[str]: + return subprocess.Popen( + ["bash", str(DEPLOY), "testhost"], + env={ + "PATH": f"{host['bin']}:{os.environ['PATH']}", + "HOME": str(host["home"]), + "USER": "tester", + "DEPLOY_ENV_FILE": "/dev/null", + "REGISTRY_HOST": "registry.example", + "NETORK_VERSION": "latest-dev", + "DEPLOY_RECREATE_RETRY_DELAY": "0", + "RUN_ID": run_id, + "FAKE_LOG": str(host["log"]), + **env, + }, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + + +def _run(host: dict[str, Path], run_id: str, **env: str) -> subprocess.CompletedProcess[str]: + proc = _start(host, run_id, **env) + out, err = proc.communicate(timeout=60) + return subprocess.CompletedProcess(proc.args, proc.returncode, out, err) + + +def _lock_file(host: dict[str, Path]) -> Path: + return host["home"] / "netork" / ".deploy.lock" + + +def _lock_is_free(host: dict[str, Path], seconds: float = 5.0) -> bool: + deadline = time.monotonic() + seconds + while time.monotonic() < deadline: + probe = subprocess.run(["flock", "-n", str(_lock_file(host)), "true"], check=False) + if probe.returncode == 0: + return True + time.sleep(0.05) + return False + + +def _events(host: dict[str, Path]) -> list[str]: + return host["log"].read_text().split("\n")[:-1] if host["log"].exists() else [] + + +def _wait_for(host: dict[str, Path], event: str, seconds: float = 20.0) -> None: + deadline = time.monotonic() + seconds + while time.monotonic() < deadline: + if event in _events(host): + return + time.sleep(0.05) + raise AssertionError(f"{event!r} never happened: {_events(host)}") + + +def test_two_deploys_to_one_host_take_turns(host) -> None: + first = _start(host, "A", FAKE_RECREATE_SLEEP="1.5") + try: + _wait_for(host, "start A") + second = _run(host, "B") + finally: + _out, err = first.communicate(timeout=60) + + assert first.returncode == 0, err + assert second.returncode == 0, second.stderr + assert _events(host) == ["start A", "end A", "start B", "end B"] + assert "another deploy holds testhost" in second.stderr + assert "deploying latest-dev" in second.stderr + assert "by tester@" in second.stderr + + +def test_the_lock_is_free_once_a_deploy_is_done(host) -> None: + result = _run(host, "A") + + assert result.returncode == 0, result.stderr + assert _lock_file(host).exists() + assert _lock_is_free(host) + + +def test_a_failed_deploy_frees_the_lock_too(host) -> None: + result = _run(host, "A", FAKE_RECREATE_FAILS="1") + + assert result.returncode != 0 + assert _lock_is_free(host) + + +def test_a_deploy_that_waits_too_long_gives_up_without_touching_the_host(host) -> None: + lock = _lock_file(host) + lock.parent.mkdir(parents=True) + holder = subprocess.Popen(["flock", str(lock), "sleep", "30"]) + try: + time.sleep(0.2) + result = _run(host, "B", DEPLOY_LOCK_WAIT="1") + finally: + holder.kill() + holder.wait() + + assert result.returncode != 0 + assert "still held after 1s" in result.stderr + assert "FAILED on: testhost" in result.stderr + assert _events(host) == [] + + +def test_a_host_without_flock_is_deployed_with_a_warning(host, tmp_path) -> None: + """flock is util-linux; a host without it should not become undeployable.""" + no_flock = tmp_path / "no-flock" + no_flock.mkdir() + for tool in ("sh", "mkdir", "cat", "date"): + found = subprocess.run( + ["sh", "-c", f"command -v {tool}"], capture_output=True, text=True, check=True + ) + (no_flock / tool).symlink_to(found.stdout.strip()) + + result = _run(host, "A", FAKE_NO_FLOCK=str(no_flock)) + + assert result.returncode == 0, result.stderr + assert "no flock on testhost" in result.stderr + assert _events(host) == ["start A", "end A"] diff --git a/tests/test_deploy_recreate_retry.py b/tests/test_deploy_recreate_retry.py index b2cc68d..c4c8409 100644 --- a/tests/test_deploy_recreate_retry.py +++ b/tests/test_deploy_recreate_retry.py @@ -25,9 +25,11 @@ import pytest DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh" _FAKE_SSH = """#!/bin/sh +for arg in "$@"; do cmd="$arg"; done +# The host lock's holder runs for real, here (test_deploy_host_lock.py). +case "$cmd" in *.deploy.lock*) exec sh -c "$cmd" ;; esac # Swallow stdin (docker login, the .env heredoc), then answer by command. cat > /dev/null -for arg in "$@"; do cmd="$arg"; done case "$cmd" in *--force-recreate*) n=$(cat "{state}/recreates" 2>/dev/null || echo 0)