From 923ec4ce659b13fa3b54edcff728d62757208672 Mon Sep 17 00:00:00 2001 From: Christian Manivong Date: Mon, 5 Oct 2026 21:58:36 +0200 Subject: [PATCH] fix: a failed container recreate is tried once more before the deploy gives up MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers": docker compose up -d --force-recreate recreates the services in parallel, lost a container it had just renamed ("No such container: 02586df7…") and stopped with every engine container Created and none running. The old API was already gone, so netOrk was down for about two minutes, until the same deploy was run again and went through cleanly. The script now does that second run itself, after DEPLOY_RECREATE_RETRY_DELAY seconds (5 by default). If the second attempt fails too, it prints the container states (docker compose ps -a) and fails as before. Tests run the script against a fake ssh that fails the recreate zero, one or two times. README step 3 says what happens. Refs NetOrk/netork#586 --- README.md | 5 +- deploy.sh | 23 +++++- tests/test_deploy_recreate_retry.py | 106 ++++++++++++++++++++++++++++ 3 files changed, 131 insertions(+), 3 deletions(-) create mode 100644 tests/test_deploy_recreate_retry.py diff --git a/README.md b/README.md index 17173af..15bc601 100644 --- a/README.md +++ b/README.md @@ -70,7 +70,10 @@ started earlier pulls whatever image the registry held before, which is stale. 2. Pulls the engine image and copies `docker-compose.yml` and `docker-compose.registry.yml` out of it into `~/netork/`. 3. Pulls the netOrk images, then force-recreates the API, the workers, `netork-beat`, - `flower` and, on UI servers, `netork-ui`. + `flower` and, on UI servers, `netork-ui`. If the recreate fails, it is tried once + more after 5 s (`DEPLOY_RECREATE_RETRY_DELAY`). Compose's parallel recreate can lose + a container it just renamed and leave the rest stopped. If the second attempt fails + too, the container states are printed and the deploy fails. 4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its definition changed. It never touches `postgres` or `redis`. 5. Verifies that the containers run exactly the image that was pulled. If they don't, the diff --git a/deploy.sh b/deploy.sh index 52a0911..dddd34e 100644 --- a/deploy.sh +++ b/deploy.sh @@ -194,12 +194,31 @@ deploy_server() { # observed on netork-ui, which kept serving a stale image after its tag had # already advanced. Recreating unconditionally costs a restart per deploy, # which a deliberate rollout wants anyway. See NetOrk/netork#90. + # + # Tried twice. compose recreates the services in parallel, and on 2026-10-05 it + # lost a container it had just renamed ("No such container") and stopped with + # every engine container Created and none running: the old API was gone, so + # netOrk was down until someone ran the deploy again, which then went through + # (NetOrk/netork#586). The second attempt is that rerun. Should it fail as well, + # the container states go to the log and the deploy fails as before. echo "[${SERVER}] Starting containers..." - run_remote "$SERVER" "recreating ${SERVICES}" "" \ - "cd ~/netork && \ + local RECREATE="cd ~/netork && \ REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ docker compose -f docker-compose.yml -f docker-compose.registry.yml \ up -d --no-deps --force-recreate ${SERVICES}" + if ! run_remote "$SERVER" "recreating ${SERVICES}" "" "$RECREATE"; then + local delay="${DEPLOY_RECREATE_RETRY_DELAY:-5}" + echo "[${SERVER}] Recreating failed; trying once more in ${delay}s..." >&2 + sleep "$delay" + if ! run_remote "$SERVER" "recreating ${SERVICES} (second attempt)" "" "$RECREATE"; then + echo "[${SERVER}] Containers after two failed attempts:" >&2 + ssh -n "$SERVER" "cd ~/netork && \ + REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ + docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -a" 2>&1 \ + | sed "s/^/[${SERVER}] /" >&2 || true + return 1 + fi + fi # Infrastructure containers, deliberately left out of the force-recreate # above. Plain `up -d`: compose compares each service definition against the diff --git a/tests/test_deploy_recreate_retry.py b/tests/test_deploy_recreate_retry.py new file mode 100644 index 0000000..b2cc68d --- /dev/null +++ b/tests/test_deploy_recreate_retry.py @@ -0,0 +1,106 @@ +"""A failed container recreate is tried once more before the deploy gives up. + +On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers": +`docker compose up -d --force-recreate` recreates the services in parallel, lost +a container it had just renamed ("No such container: 02586df7…"), and stopped +with every engine container `Created` and none running. The old API was already +gone, so netOrk was down until somebody ran the same deploy again, which then +went through cleanly (NetOrk/netork#586). + +So the script runs that step a second time on its own. Should the second attempt +fail too, it shows what the containers look like and fails loudly, as before. + +The remote side is a fake `ssh` that answers each step and can fail the recreate +a given number of times. +""" + +from __future__ import annotations + +import os +import subprocess +from pathlib import Path + +import pytest + +DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh" + +_FAKE_SSH = """#!/bin/sh +# Swallow stdin (docker login, the .env heredoc), then answer by command. +cat > /dev/null +for arg in "$@"; do cmd="$arg"; done +case "$cmd" in + *--force-recreate*) + n=$(cat "{state}/recreates" 2>/dev/null || echo 0) + n=$((n + 1)) + echo "$n" > "{state}/recreates" + if [ "$n" -le "{failures}" ]; then + echo "Error response from daemon: No such container: 02586df7e659" + exit 1 + fi + echo " Container netork-netork-api-1 Started" + ;; + *" ps -a"*) + echo "netork-netork-api-1 Created" + ;; +esac +exit 0 +""" + + +@pytest.fixture +def deploy(tmp_path: Path): + """Run a deploy against the fake ssh; *failures* recreates fail before one works.""" + + def _deploy(failures: int) -> tuple[subprocess.CompletedProcess[str], int]: + bin_dir = tmp_path / "bin" + bin_dir.mkdir(exist_ok=True) + fake = bin_dir / "ssh" + fake.write_text( + _FAKE_SSH.replace("{state}", str(tmp_path)).replace("{failures}", str(failures)) + ) + fake.chmod(0o755) + result = subprocess.run( + ["bash", str(DEPLOY), "testhost"], + env={ + "PATH": f"{bin_dir}:{os.environ['PATH']}", + "HOME": str(tmp_path), + "DEPLOY_ENV_FILE": "/dev/null", + "REGISTRY_HOST": "registry.example", + "NETORK_VERSION": "latest-dev", + "DEPLOY_RECREATE_RETRY_DELAY": "0", + }, + stdin=subprocess.DEVNULL, + capture_output=True, + text=True, + timeout=60, + check=False, + ) + recreates = tmp_path / "recreates" + return result, int(recreates.read_text()) if recreates.exists() else 0 + + return _deploy + + +def test_a_recreate_that_works_runs_once(deploy) -> None: + result, recreates = deploy(failures=0) + + assert result.returncode == 0, result.stderr + assert recreates == 1 + + +def test_a_recreate_that_fails_once_is_tried_again(deploy) -> None: + result, recreates = deploy(failures=1) + + assert result.returncode == 0, result.stderr + assert recreates == 2 + assert "trying once more" in result.stderr + assert "All servers deployed successfully." in result.stdout + + +def test_two_failed_recreates_fail_the_deploy_and_show_the_containers(deploy) -> None: + result, recreates = deploy(failures=2) + + assert result.returncode != 0 + assert recreates == 2 + assert "netork-netork-api-1 Created" in result.stderr + assert "FAILED on: testhost" in result.stderr -- 2.54.0