diff --git a/README.md b/README.md index 17173af..15bc601 100644 --- a/README.md +++ b/README.md @@ -70,7 +70,10 @@ started earlier pulls whatever image the registry held before, which is stale. 2. Pulls the engine image and copies `docker-compose.yml` and `docker-compose.registry.yml` out of it into `~/netork/`. 3. Pulls the netOrk images, then force-recreates the API, the workers, `netork-beat`, - `flower` and, on UI servers, `netork-ui`. + `flower` and, on UI servers, `netork-ui`. If the recreate fails, it is tried once + more after 5 s (`DEPLOY_RECREATE_RETRY_DELAY`). Compose's parallel recreate can lose + a container it just renamed and leave the rest stopped. If the second attempt fails + too, the container states are printed and the deploy fails. 4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its definition changed. It never touches `postgres` or `redis`. 5. Verifies that the containers run exactly the image that was pulled. If they don't, the diff --git a/deploy.sh b/deploy.sh index 52a0911..dddd34e 100644 --- a/deploy.sh +++ b/deploy.sh @@ -194,12 +194,31 @@ deploy_server() { # observed on netork-ui, which kept serving a stale image after its tag had # already advanced. Recreating unconditionally costs a restart per deploy, # which a deliberate rollout wants anyway. See NetOrk/netork#90. + # + # Tried twice. compose recreates the services in parallel, and on 2026-10-05 it + # lost a container it had just renamed ("No such container") and stopped with + # every engine container Created and none running: the old API was gone, so + # netOrk was down until someone ran the deploy again, which then went through + # (NetOrk/netork#586). The second attempt is that rerun. Should it fail as well, + # the container states go to the log and the deploy fails as before. echo "[${SERVER}] Starting containers..." - run_remote "$SERVER" "recreating ${SERVICES}" "" \ - "cd ~/netork && \ + local RECREATE="cd ~/netork && \ REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ docker compose -f docker-compose.yml -f docker-compose.registry.yml \ up -d --no-deps --force-recreate ${SERVICES}" + if ! run_remote "$SERVER" "recreating ${SERVICES}" "" "$RECREATE"; then + local delay="${DEPLOY_RECREATE_RETRY_DELAY:-5}" + echo "[${SERVER}] Recreating failed; trying once more in ${delay}s..." >&2 + sleep "$delay" + if ! run_remote "$SERVER" "recreating ${SERVICES} (second attempt)" "" "$RECREATE"; then + echo "[${SERVER}] Containers after two failed attempts:" >&2 + ssh -n "$SERVER" "cd ~/netork && \ + REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ + docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -a" 2>&1 \ + | sed "s/^/[${SERVER}] /" >&2 || true + return 1 + fi + fi # Infrastructure containers, deliberately left out of the force-recreate # above. Plain `up -d`: compose compares each service definition against the diff --git a/tests/test_deploy_recreate_retry.py b/tests/test_deploy_recreate_retry.py new file mode 100644 index 0000000..b2cc68d --- /dev/null +++ b/tests/test_deploy_recreate_retry.py @@ -0,0 +1,106 @@ +"""A failed container recreate is tried once more before the deploy gives up. + +On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers": +`docker compose up -d --force-recreate` recreates the services in parallel, lost +a container it had just renamed ("No such container: 02586df7…"), and stopped +with every engine container `Created` and none running. The old API was already +gone, so netOrk was down until somebody ran the same deploy again, which then +went through cleanly (NetOrk/netork#586). + +So the script runs that step a second time on its own. Should the second attempt +fail too, it shows what the containers look like and fails loudly, as before. + +The remote side is a fake `ssh` that answers each step and can fail the recreate +a given number of times. +""" + +from __future__ import annotations + +import os +import subprocess +from pathlib import Path + +import pytest + +DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh" + +_FAKE_SSH = """#!/bin/sh +# Swallow stdin (docker login, the .env heredoc), then answer by command. +cat > /dev/null +for arg in "$@"; do cmd="$arg"; done +case "$cmd" in + *--force-recreate*) + n=$(cat "{state}/recreates" 2>/dev/null || echo 0) + n=$((n + 1)) + echo "$n" > "{state}/recreates" + if [ "$n" -le "{failures}" ]; then + echo "Error response from daemon: No such container: 02586df7e659" + exit 1 + fi + echo " Container netork-netork-api-1 Started" + ;; + *" ps -a"*) + echo "netork-netork-api-1 Created" + ;; +esac +exit 0 +""" + + +@pytest.fixture +def deploy(tmp_path: Path): + """Run a deploy against the fake ssh; *failures* recreates fail before one works.""" + + def _deploy(failures: int) -> tuple[subprocess.CompletedProcess[str], int]: + bin_dir = tmp_path / "bin" + bin_dir.mkdir(exist_ok=True) + fake = bin_dir / "ssh" + fake.write_text( + _FAKE_SSH.replace("{state}", str(tmp_path)).replace("{failures}", str(failures)) + ) + fake.chmod(0o755) + result = subprocess.run( + ["bash", str(DEPLOY), "testhost"], + env={ + "PATH": f"{bin_dir}:{os.environ['PATH']}", + "HOME": str(tmp_path), + "DEPLOY_ENV_FILE": "/dev/null", + "REGISTRY_HOST": "registry.example", + "NETORK_VERSION": "latest-dev", + "DEPLOY_RECREATE_RETRY_DELAY": "0", + }, + stdin=subprocess.DEVNULL, + capture_output=True, + text=True, + timeout=60, + check=False, + ) + recreates = tmp_path / "recreates" + return result, int(recreates.read_text()) if recreates.exists() else 0 + + return _deploy + + +def test_a_recreate_that_works_runs_once(deploy) -> None: + result, recreates = deploy(failures=0) + + assert result.returncode == 0, result.stderr + assert recreates == 1 + + +def test_a_recreate_that_fails_once_is_tried_again(deploy) -> None: + result, recreates = deploy(failures=1) + + assert result.returncode == 0, result.stderr + assert recreates == 2 + assert "trying once more" in result.stderr + assert "All servers deployed successfully." in result.stdout + + +def test_two_failed_recreates_fail_the_deploy_and_show_the_containers(deploy) -> None: + result, recreates = deploy(failures=2) + + assert result.returncode != 0 + assert recreates == 2 + assert "netork-netork-api-1 Created" in result.stderr + assert "FAILED on: testhost" in result.stderr