diff --git a/README.md b/README.md index 9eb2b88..20e4e1c 100644 --- a/README.md +++ b/README.md @@ -81,8 +81,10 @@ started earlier pulls whatever image the registry held before, which is stale. more after 5 s (`DEPLOY_RECREATE_RETRY_DELAY`). Compose's parallel recreate can lose a container it just renamed and leave the rest stopped. If the second attempt fails too, the container states are printed and the deploy fails. -4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its +4. Reconciles `apt-cacher-ng` and `signal-api`: each is recreated only if its definition changed. It never touches `postgres` or `redis`. + It then removes services netOrk no longer ships, container and volume, where a host + still has them: today the bundled `registry` (NetOrk/netork#763). 5. Verifies that the containers run exactly the image that was pulled. If they don't, the deploy fails. 6. Records `REGISTRY_HOST` and `NETORK_VERSION` in `~/netork/.env`, so that a diff --git a/deploy.sh b/deploy.sh index 2c72f73..229c40a 100644 --- a/deploy.sh +++ b/deploy.sh @@ -252,7 +252,7 @@ deploy_steps() { # Containers on upstream images. Reconciled, not force-recreated (below). # postgres and redis are deliberately absent: restarting a database on every # deploy would be worse than any compose change one could miss. - local INFRA_SERVICES="registry apt-cacher-ng signal-api" + local INFRA_SERVICES="apt-cacher-ng signal-api" echo "[${SERVER}] Pulling images..." run_remote "$SERVER" "pulling images for ${SERVER_VERSION}" 'Pulled|Already|Error' \ @@ -311,6 +311,24 @@ deploy_steps() { docker compose -f docker-compose.yml -f docker-compose.registry.yml \ up -d --no-deps ${INFRA_SERVICES}" || true + # Services netOrk no longer ships, with the data only they used. `docker + # compose up` never removes a container whose service has left the compose + # file, so each would keep running on every host until someone removed it by + # hand. Idempotent: a host that never had one, or was cleaned already, passes + # untouched. Containers first; a volume still in use cannot be removed. + # registry (netork-registry-1, netork_registry_data): the bundled registry:2 + # for satellite images. Satellites pull from registry.netork.io; it held one + # old image, nothing had pulled from it in 30 days, and it accepted + # anonymous pushes on port 5000 (NetOrk/netork#763). + local RETIRED_CONTAINERS="netork-registry-1" + local RETIRED_VOLUMES="netork_registry_data" + echo "[${SERVER}] Removing retired services..." + run_remote "$SERVER" "removing ${RETIRED_CONTAINERS} ${RETIRED_VOLUMES}" "" \ + "for c in ${RETIRED_CONTAINERS}; do \ + docker rm -f \$c >/dev/null 2>&1 && echo \"removed container \$c\"; done; \ + for v in ${RETIRED_VOLUMES}; do \ + docker volume rm \$v >/dev/null 2>&1 && echo \"removed volume \$v\"; done; true" || true + # Verify the containers actually run the image we just pulled. `docker # compose up -d` reports "Running" (not "Started") when it decides nothing # changed, and `pull` prints "Pulled" even when the tag was already local — diff --git a/tests/test_deploy_retired_services.py b/tests/test_deploy_retired_services.py new file mode 100644 index 0000000..75eab06 --- /dev/null +++ b/tests/test_deploy_retired_services.py @@ -0,0 +1,73 @@ +"""A service netOrk no longer ships is removed from the host (NetOrk/netork#763). + +`docker compose up` never removes a container whose service has left the compose +file, so it would keep running on every host until somebody removed it by hand. +The first of these is the bundled `registry:2`: it held only an old satellite +image, nothing had pulled from it in 30 days, and it accepted anonymous pushes on +port 5000 of every instance. + +The remote side is a fake `ssh` that records every command it is given. +""" + +from __future__ import annotations + +import os +import subprocess +from pathlib import Path + +DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh" + +_FAKE_SSH = """#!/bin/sh +for arg in "$@"; do cmd="$arg"; done +# The host lock's holder runs for real, here (test_deploy_host_lock.py). +case "$cmd" in *.deploy.lock*) exec sh -c "$cmd" ;; esac +cat > /dev/null +printf '%s\\n----\\n' "$cmd" >> "{state}/commands" +exit 0 +""" + + +def _deploy(tmp_path: Path) -> tuple[subprocess.CompletedProcess[str], list[str]]: + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + fake = bin_dir / "ssh" + fake.write_text(_FAKE_SSH.replace("{state}", str(tmp_path))) + fake.chmod(0o755) + result = subprocess.run( + ["bash", str(DEPLOY), "testhost"], + env={ + "PATH": f"{bin_dir}:{os.environ['PATH']}", + "HOME": str(tmp_path), + "DEPLOY_ENV_FILE": "/dev/null", + "REGISTRY_HOST": "registry.example", + "NETORK_VERSION": "latest-dev", + }, + stdin=subprocess.DEVNULL, + capture_output=True, + text=True, + timeout=60, + check=False, + ) + commands = (tmp_path / "commands").read_text().split("\n----\n") + return result, commands + + +def test_a_deploy_removes_the_retired_registry_and_its_volume(tmp_path: Path) -> None: + result, commands = _deploy(tmp_path) + + assert result.returncode == 0, result.stderr + removal = [c for c in commands if "netork-registry-1" in c] + assert removal, "no command removes the netork-registry-1 container" + assert "docker rm -f" in removal[0] + assert "docker volume rm" in removal[0] and "netork_registry_data" in removal[0] + # The container goes first: a volume still in use cannot be removed. + assert removal[0].index("docker rm -f") < removal[0].index("docker volume rm") + + +def test_the_infrastructure_reconcile_no_longer_starts_the_registry(tmp_path: Path) -> None: + result, commands = _deploy(tmp_path) + + assert result.returncode == 0, result.stderr + reconcile = [c for c in commands if "up -d --no-deps" in c and "--force-recreate" not in c] + assert reconcile, "the infrastructure reconcile step is missing" + assert " registry" not in reconcile[0].split("up -d --no-deps", 1)[1]