fix: one deploy per host at a time, under a lock the host frees on its own
CI / check (pull_request) Successful in 17s

Several sessions deploy to the same test server, and two runs used to
overlap. On 2026-10-05 two `up -d --force-recreate` runs recreated each
other's containers and the API was down for a minute (#1). On 2026-10-06
two runs renamed each other's *.new compose files and one broke off; with
two different tags, one tag's compose files could have started the
other's images (#3).

- Each deploy first takes flock on ~/netork/.deploy.lock on the host. An
  ssh session holds it: the remote side takes the lock on fd 9, reports
  LOCKED and waits on its stdin, so ending the session frees it, whether
  the deploy finished, failed, was interrupted or lost its connection.
  Checked over real ssh on .50, including a client killed with -9.
- A second deploy prints who holds the lock, since when and with which
  tag, and waits up to DEPLOY_LOCK_WAIT seconds (default 900); then it
  gives up without touching the host.
- A host without flock is deployed without the lock, with a warning.
- deploy_server is now the lock around deploy_steps, the old body.

Closes #1
Closes #3
This commit is contained in:
Christian Manivong
2026-10-07 06:42:13 +02:00
parent 6444aa97f3
commit ec93619f8f
4 changed files with 270 additions and 3 deletions
+185
View File
@@ -0,0 +1,185 @@
"""One deploy per host at a time (NetOrk/deploy#1, #3).
Several sessions deploy to the same test server, and nothing stopped two runs
from overlapping:
- 2026-10-05: two `up -d --force-recreate` runs recreated each other's
containers. The API was down for a minute, with leftover `<id>_netork-…`
containers.
- 2026-10-06: two runs renamed each other's `*.new` compose files, and one broke
off at `mv: cannot stat 'docker-compose.yml.new'`. With two different tags,
one tag's compose files could have started the other tag's images.
Now each deploy first takes `flock` on `~/netork/.deploy.lock` on the host,
through an ssh session that holds it until the deploy ends. A second deploy
says who holds it, since when and with which tag, and waits. The lock lives as
long as that session does, so a deploy that dies, or a laptop that loses its
network, frees it on its own.
The "host" here is this machine with `HOME` in a temporary directory: the fake
`ssh` runs the lock holder for real, so the lock is a real `flock`.
"""
from __future__ import annotations
import os
import subprocess
import time
from pathlib import Path
import pytest
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
_FAKE_SSH = """#!/bin/sh
for arg in "$@"; do cmd="$arg"; done
case "$cmd" in
*.deploy.lock*)
# The lock holder runs for real, on this "host".
if [ -n "$FAKE_NO_FLOCK" ]; then exec env PATH="$FAKE_NO_FLOCK" sh -c "$cmd"; fi
exec sh -c "$cmd"
;;
esac
cat > /dev/null
case "$cmd" in
*--force-recreate*)
echo "start $RUN_ID" >> "$FAKE_LOG"
sleep "${FAKE_RECREATE_SLEEP:-0}"
echo "end $RUN_ID" >> "$FAKE_LOG"
if [ -n "$FAKE_RECREATE_FAILS" ]; then echo "Error response from daemon"; exit 1; fi
;;
esac
exit 0
"""
@pytest.fixture
def host(tmp_path: Path) -> dict[str, Path]:
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
fake = bin_dir / "ssh"
fake.write_text(_FAKE_SSH)
fake.chmod(0o755)
home = tmp_path / "home"
home.mkdir()
return {"bin": bin_dir, "home": home, "log": tmp_path / "log"}
def _start(host: dict[str, Path], run_id: str, **env: str) -> subprocess.Popen[str]:
return subprocess.Popen(
["bash", str(DEPLOY), "testhost"],
env={
"PATH": f"{host['bin']}:{os.environ['PATH']}",
"HOME": str(host["home"]),
"USER": "tester",
"DEPLOY_ENV_FILE": "/dev/null",
"REGISTRY_HOST": "registry.example",
"NETORK_VERSION": "latest-dev",
"DEPLOY_RECREATE_RETRY_DELAY": "0",
"RUN_ID": run_id,
"FAKE_LOG": str(host["log"]),
**env,
},
stdin=subprocess.DEVNULL,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
text=True,
)
def _run(host: dict[str, Path], run_id: str, **env: str) -> subprocess.CompletedProcess[str]:
proc = _start(host, run_id, **env)
out, err = proc.communicate(timeout=60)
return subprocess.CompletedProcess(proc.args, proc.returncode, out, err)
def _lock_file(host: dict[str, Path]) -> Path:
return host["home"] / "netork" / ".deploy.lock"
def _lock_is_free(host: dict[str, Path], seconds: float = 5.0) -> bool:
deadline = time.monotonic() + seconds
while time.monotonic() < deadline:
probe = subprocess.run(["flock", "-n", str(_lock_file(host)), "true"], check=False)
if probe.returncode == 0:
return True
time.sleep(0.05)
return False
def _events(host: dict[str, Path]) -> list[str]:
return host["log"].read_text().split("\n")[:-1] if host["log"].exists() else []
def _wait_for(host: dict[str, Path], event: str, seconds: float = 20.0) -> None:
deadline = time.monotonic() + seconds
while time.monotonic() < deadline:
if event in _events(host):
return
time.sleep(0.05)
raise AssertionError(f"{event!r} never happened: {_events(host)}")
def test_two_deploys_to_one_host_take_turns(host) -> None:
first = _start(host, "A", FAKE_RECREATE_SLEEP="1.5")
try:
_wait_for(host, "start A")
second = _run(host, "B")
finally:
_out, err = first.communicate(timeout=60)
assert first.returncode == 0, err
assert second.returncode == 0, second.stderr
assert _events(host) == ["start A", "end A", "start B", "end B"]
assert "another deploy holds testhost" in second.stderr
assert "deploying latest-dev" in second.stderr
assert "by tester@" in second.stderr
def test_the_lock_is_free_once_a_deploy_is_done(host) -> None:
result = _run(host, "A")
assert result.returncode == 0, result.stderr
assert _lock_file(host).exists()
assert _lock_is_free(host)
def test_a_failed_deploy_frees_the_lock_too(host) -> None:
result = _run(host, "A", FAKE_RECREATE_FAILS="1")
assert result.returncode != 0
assert _lock_is_free(host)
def test_a_deploy_that_waits_too_long_gives_up_without_touching_the_host(host) -> None:
lock = _lock_file(host)
lock.parent.mkdir(parents=True)
holder = subprocess.Popen(["flock", str(lock), "sleep", "30"])
try:
time.sleep(0.2)
result = _run(host, "B", DEPLOY_LOCK_WAIT="1")
finally:
holder.kill()
holder.wait()
assert result.returncode != 0
assert "still held after 1s" in result.stderr
assert "FAILED on: testhost" in result.stderr
assert _events(host) == []
def test_a_host_without_flock_is_deployed_with_a_warning(host, tmp_path) -> None:
"""flock is util-linux; a host without it should not become undeployable."""
no_flock = tmp_path / "no-flock"
no_flock.mkdir()
for tool in ("sh", "mkdir", "cat", "date"):
found = subprocess.run(
["sh", "-c", f"command -v {tool}"], capture_output=True, text=True, check=True
)
(no_flock / tool).symlink_to(found.stdout.strip())
result = _run(host, "A", FAKE_NO_FLOCK=str(no_flock))
assert result.returncode == 0, result.stderr
assert "no flock on testhost" in result.stderr
assert _events(host) == ["start A", "end A"]
+3 -1
View File
@@ -25,9 +25,11 @@ import pytest
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
_FAKE_SSH = """#!/bin/sh
for arg in "$@"; do cmd="$arg"; done
# The host lock's holder runs for real, here (test_deploy_host_lock.py).
case "$cmd" in *.deploy.lock*) exec sh -c "$cmd" ;; esac
# Swallow stdin (docker login, the .env heredoc), then answer by command.
cat > /dev/null
for arg in "$@"; do cmd="$arg"; done
case "$cmd" in
*--force-recreate*)
n=$(cat "{state}/recreates" 2>/dev/null || echo 0)