fix: remove the bundled registry from every host, and stop starting it #5

Merged
christianmanivong merged 1 commits from fix/retire-registry into main 2026-10-07 14:19:29 +00:00
3 changed files with 95 additions and 2 deletions
+3 -1
View File
@@ -81,8 +81,10 @@ started earlier pulls whatever image the registry held before, which is stale.
more after 5 s (`DEPLOY_RECREATE_RETRY_DELAY`). Compose's parallel recreate can lose
a container it just renamed and leave the rest stopped. If the second attempt fails
too, the container states are printed and the deploy fails.
4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its
4. Reconciles `apt-cacher-ng` and `signal-api`: each is recreated only if its
definition changed. It never touches `postgres` or `redis`.
It then removes services netOrk no longer ships, container and volume, where a host
still has them: today the bundled `registry` (NetOrk/netork#763).
5. Verifies that the containers run exactly the image that was pulled. If they don't, the
deploy fails.
6. Records `REGISTRY_HOST` and `NETORK_VERSION` in `~/netork/.env`, so that a
+19 -1
View File
@@ -252,7 +252,7 @@ deploy_steps() {
# Containers on upstream images. Reconciled, not force-recreated (below).
# postgres and redis are deliberately absent: restarting a database on every
# deploy would be worse than any compose change one could miss.
local INFRA_SERVICES="registry apt-cacher-ng signal-api"
local INFRA_SERVICES="apt-cacher-ng signal-api"
echo "[${SERVER}] Pulling images..."
run_remote "$SERVER" "pulling images for ${SERVER_VERSION}" 'Pulled|Already|Error' \
@@ -311,6 +311,24 @@ deploy_steps() {
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps ${INFRA_SERVICES}" || true
# Services netOrk no longer ships, with the data only they used. `docker
# compose up` never removes a container whose service has left the compose
# file, so each would keep running on every host until someone removed it by
# hand. Idempotent: a host that never had one, or was cleaned already, passes
# untouched. Containers first; a volume still in use cannot be removed.
# registry (netork-registry-1, netork_registry_data): the bundled registry:2
# for satellite images. Satellites pull from registry.netork.io; it held one
# old image, nothing had pulled from it in 30 days, and it accepted
# anonymous pushes on port 5000 (NetOrk/netork#763).
local RETIRED_CONTAINERS="netork-registry-1"
local RETIRED_VOLUMES="netork_registry_data"
echo "[${SERVER}] Removing retired services..."
run_remote "$SERVER" "removing ${RETIRED_CONTAINERS} ${RETIRED_VOLUMES}" "" \
"for c in ${RETIRED_CONTAINERS}; do \
docker rm -f \$c >/dev/null 2>&1 && echo \"removed container \$c\"; done; \
for v in ${RETIRED_VOLUMES}; do \
docker volume rm \$v >/dev/null 2>&1 && echo \"removed volume \$v\"; done; true" || true
# Verify the containers actually run the image we just pulled. `docker
# compose up -d` reports "Running" (not "Started") when it decides nothing
# changed, and `pull` prints "Pulled" even when the tag was already local —
+73
View File
@@ -0,0 +1,73 @@
"""A service netOrk no longer ships is removed from the host (NetOrk/netork#763).
`docker compose up` never removes a container whose service has left the compose
file, so it would keep running on every host until somebody removed it by hand.
The first of these is the bundled `registry:2`: it held only an old satellite
image, nothing had pulled from it in 30 days, and it accepted anonymous pushes on
port 5000 of every instance.
The remote side is a fake `ssh` that records every command it is given.
"""
from __future__ import annotations
import os
import subprocess
from pathlib import Path
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
_FAKE_SSH = """#!/bin/sh
for arg in "$@"; do cmd="$arg"; done
# The host lock's holder runs for real, here (test_deploy_host_lock.py).
case "$cmd" in *.deploy.lock*) exec sh -c "$cmd" ;; esac
cat > /dev/null
printf '%s\\n----\\n' "$cmd" >> "{state}/commands"
exit 0
"""
def _deploy(tmp_path: Path) -> tuple[subprocess.CompletedProcess[str], list[str]]:
bin_dir = tmp_path / "bin"
bin_dir.mkdir()
fake = bin_dir / "ssh"
fake.write_text(_FAKE_SSH.replace("{state}", str(tmp_path)))
fake.chmod(0o755)
result = subprocess.run(
["bash", str(DEPLOY), "testhost"],
env={
"PATH": f"{bin_dir}:{os.environ['PATH']}",
"HOME": str(tmp_path),
"DEPLOY_ENV_FILE": "/dev/null",
"REGISTRY_HOST": "registry.example",
"NETORK_VERSION": "latest-dev",
},
stdin=subprocess.DEVNULL,
capture_output=True,
text=True,
timeout=60,
check=False,
)
commands = (tmp_path / "commands").read_text().split("\n----\n")
return result, commands
def test_a_deploy_removes_the_retired_registry_and_its_volume(tmp_path: Path) -> None:
result, commands = _deploy(tmp_path)
assert result.returncode == 0, result.stderr
removal = [c for c in commands if "netork-registry-1" in c]
assert removal, "no command removes the netork-registry-1 container"
assert "docker rm -f" in removal[0]
assert "docker volume rm" in removal[0] and "netork_registry_data" in removal[0]
# The container goes first: a volume still in use cannot be removed.
assert removal[0].index("docker rm -f") < removal[0].index("docker volume rm")
def test_the_infrastructure_reconcile_no_longer_starts_the_registry(tmp_path: Path) -> None:
result, commands = _deploy(tmp_path)
assert result.returncode == 0, result.stderr
reconcile = [c for c in commands if "up -d --no-deps" in c and "--force-recreate" not in c]
assert reconcile, "the infrastructure reconcile step is missing"
assert " registry" not in reconcile[0].split("up -d --no-deps", 1)[1]