fix: a failed container recreate is tried once more before the deploy gives up #2

Merged
christianmanivong merged 1 commits from fix/recreate-retry into main 2026-10-05 20:03:55 +00:00
3 changed files with 131 additions and 3 deletions
Showing only changes of commit 923ec4ce65 - Show all commits
+4 -1
View File
@@ -70,7 +70,10 @@ started earlier pulls whatever image the registry held before, which is stale.
2. Pulls the engine image and copies `docker-compose.yml` and 2. Pulls the engine image and copies `docker-compose.yml` and
`docker-compose.registry.yml` out of it into `~/netork/`. `docker-compose.registry.yml` out of it into `~/netork/`.
3. Pulls the netOrk images, then force-recreates the API, the workers, `netork-beat`, 3. Pulls the netOrk images, then force-recreates the API, the workers, `netork-beat`,
`flower` and, on UI servers, `netork-ui`. `flower` and, on UI servers, `netork-ui`. If the recreate fails, it is tried once
more after 5 s (`DEPLOY_RECREATE_RETRY_DELAY`). Compose's parallel recreate can lose
a container it just renamed and leave the rest stopped. If the second attempt fails
too, the container states are printed and the deploy fails.
4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its 4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its
definition changed. It never touches `postgres` or `redis`. definition changed. It never touches `postgres` or `redis`.
5. Verifies that the containers run exactly the image that was pulled. If they don't, the 5. Verifies that the containers run exactly the image that was pulled. If they don't, the
+21 -2
View File
@@ -194,12 +194,31 @@ deploy_server() {
# observed on netork-ui, which kept serving a stale image after its tag had # observed on netork-ui, which kept serving a stale image after its tag had
# already advanced. Recreating unconditionally costs a restart per deploy, # already advanced. Recreating unconditionally costs a restart per deploy,
# which a deliberate rollout wants anyway. See NetOrk/netork#90. # which a deliberate rollout wants anyway. See NetOrk/netork#90.
#
# Tried twice. compose recreates the services in parallel, and on 2026-10-05 it
# lost a container it had just renamed ("No such container") and stopped with
# every engine container Created and none running: the old API was gone, so
# netOrk was down until someone ran the deploy again, which then went through
# (NetOrk/netork#586). The second attempt is that rerun. Should it fail as well,
# the container states go to the log and the deploy fails as before.
echo "[${SERVER}] Starting containers..." echo "[${SERVER}] Starting containers..."
run_remote "$SERVER" "recreating ${SERVICES}" "" \ local RECREATE="cd ~/netork && \
"cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \ REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml \ docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps --force-recreate ${SERVICES}" up -d --no-deps --force-recreate ${SERVICES}"
if ! run_remote "$SERVER" "recreating ${SERVICES}" "" "$RECREATE"; then
local delay="${DEPLOY_RECREATE_RETRY_DELAY:-5}"
echo "[${SERVER}] Recreating failed; trying once more in ${delay}s..." >&2
sleep "$delay"
if ! run_remote "$SERVER" "recreating ${SERVICES} (second attempt)" "" "$RECREATE"; then
echo "[${SERVER}] Containers after two failed attempts:" >&2
ssh -n "$SERVER" "cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -a" 2>&1 \
| sed "s/^/[${SERVER}] /" >&2 || true
return 1
fi
fi
# Infrastructure containers, deliberately left out of the force-recreate # Infrastructure containers, deliberately left out of the force-recreate
# above. Plain `up -d`: compose compares each service definition against the # above. Plain `up -d`: compose compares each service definition against the
+106
View File
@@ -0,0 +1,106 @@
"""A failed container recreate is tried once more before the deploy gives up.
On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers":
`docker compose up -d --force-recreate` recreates the services in parallel, lost
a container it had just renamed ("No such container: 02586df7…"), and stopped
with every engine container `Created` and none running. The old API was already
gone, so netOrk was down until somebody ran the same deploy again, which then
went through cleanly (NetOrk/netork#586).
So the script runs that step a second time on its own. Should the second attempt
fail too, it shows what the containers look like and fails loudly, as before.
The remote side is a fake `ssh` that answers each step and can fail the recreate
a given number of times.
"""
from __future__ import annotations
import os
import subprocess
from pathlib import Path
import pytest
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
_FAKE_SSH = """#!/bin/sh
# Swallow stdin (docker login, the .env heredoc), then answer by command.
cat > /dev/null
for arg in "$@"; do cmd="$arg"; done
case "$cmd" in
*--force-recreate*)
n=$(cat "{state}/recreates" 2>/dev/null || echo 0)
n=$((n + 1))
echo "$n" > "{state}/recreates"
if [ "$n" -le "{failures}" ]; then
echo "Error response from daemon: No such container: 02586df7e659"
exit 1
fi
echo " Container netork-netork-api-1 Started"
;;
*" ps -a"*)
echo "netork-netork-api-1 Created"
;;
esac
exit 0
"""
@pytest.fixture
def deploy(tmp_path: Path):
"""Run a deploy against the fake ssh; *failures* recreates fail before one works."""
def _deploy(failures: int) -> tuple[subprocess.CompletedProcess[str], int]:
bin_dir = tmp_path / "bin"
bin_dir.mkdir(exist_ok=True)
fake = bin_dir / "ssh"
fake.write_text(
_FAKE_SSH.replace("{state}", str(tmp_path)).replace("{failures}", str(failures))
)
fake.chmod(0o755)
result = subprocess.run(
["bash", str(DEPLOY), "testhost"],
env={
"PATH": f"{bin_dir}:{os.environ['PATH']}",
"HOME": str(tmp_path),
"DEPLOY_ENV_FILE": "/dev/null",
"REGISTRY_HOST": "registry.example",
"NETORK_VERSION": "latest-dev",
"DEPLOY_RECREATE_RETRY_DELAY": "0",
},
stdin=subprocess.DEVNULL,
capture_output=True,
text=True,
timeout=60,
check=False,
)
recreates = tmp_path / "recreates"
return result, int(recreates.read_text()) if recreates.exists() else 0
return _deploy
def test_a_recreate_that_works_runs_once(deploy) -> None:
result, recreates = deploy(failures=0)
assert result.returncode == 0, result.stderr
assert recreates == 1
def test_a_recreate_that_fails_once_is_tried_again(deploy) -> None:
result, recreates = deploy(failures=1)
assert result.returncode == 0, result.stderr
assert recreates == 2
assert "trying once more" in result.stderr
assert "All servers deployed successfully." in result.stdout
def test_two_failed_recreates_fail_the_deploy_and_show_the_containers(deploy) -> None:
result, recreates = deploy(failures=2)
assert result.returncode != 0
assert recreates == 2
assert "netork-netork-api-1 Created" in result.stderr
assert "FAILED on: testhost" in result.stderr