Merge pull request 'fix: a failed container recreate is tried once more before the deploy gives up' (#2) from fix/recreate-retry into main
CI / check (push) Successful in 12s

This commit was merged in pull request #2.
This commit is contained in:
2026-10-05 20:03:55 +00:00
3 changed files with 131 additions and 3 deletions
+4 -1
View File
@@ -70,7 +70,10 @@ started earlier pulls whatever image the registry held before, which is stale.
2. Pulls the engine image and copies `docker-compose.yml` and
`docker-compose.registry.yml` out of it into `~/netork/`.
3. Pulls the netOrk images, then force-recreates the API, the workers, `netork-beat`,
`flower` and, on UI servers, `netork-ui`.
`flower` and, on UI servers, `netork-ui`. If the recreate fails, it is tried once
more after 5 s (`DEPLOY_RECREATE_RETRY_DELAY`). Compose's parallel recreate can lose
a container it just renamed and leave the rest stopped. If the second attempt fails
too, the container states are printed and the deploy fails.
4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its
definition changed. It never touches `postgres` or `redis`.
5. Verifies that the containers run exactly the image that was pulled. If they don't, the
+21 -2
View File
@@ -194,12 +194,31 @@ deploy_server() {
# observed on netork-ui, which kept serving a stale image after its tag had
# already advanced. Recreating unconditionally costs a restart per deploy,
# which a deliberate rollout wants anyway. See NetOrk/netork#90.
#
# Tried twice. compose recreates the services in parallel, and on 2026-10-05 it
# lost a container it had just renamed ("No such container") and stopped with
# every engine container Created and none running: the old API was gone, so
# netOrk was down until someone ran the deploy again, which then went through
# (NetOrk/netork#586). The second attempt is that rerun. Should it fail as well,
# the container states go to the log and the deploy fails as before.
echo "[${SERVER}] Starting containers..."
run_remote "$SERVER" "recreating ${SERVICES}" "" \
"cd ~/netork && \
local RECREATE="cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
up -d --no-deps --force-recreate ${SERVICES}"
if ! run_remote "$SERVER" "recreating ${SERVICES}" "" "$RECREATE"; then
local delay="${DEPLOY_RECREATE_RETRY_DELAY:-5}"
echo "[${SERVER}] Recreating failed; trying once more in ${delay}s..." >&2
sleep "$delay"
if ! run_remote "$SERVER" "recreating ${SERVICES} (second attempt)" "" "$RECREATE"; then
echo "[${SERVER}] Containers after two failed attempts:" >&2
ssh -n "$SERVER" "cd ~/netork && \
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -a" 2>&1 \
| sed "s/^/[${SERVER}] /" >&2 || true
return 1
fi
fi
# Infrastructure containers, deliberately left out of the force-recreate
# above. Plain `up -d`: compose compares each service definition against the
+106
View File
@@ -0,0 +1,106 @@
"""A failed container recreate is tried once more before the deploy gives up.
On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers":
`docker compose up -d --force-recreate` recreates the services in parallel, lost
a container it had just renamed ("No such container: 02586df7…"), and stopped
with every engine container `Created` and none running. The old API was already
gone, so netOrk was down until somebody ran the same deploy again, which then
went through cleanly (NetOrk/netork#586).
So the script runs that step a second time on its own. Should the second attempt
fail too, it shows what the containers look like and fails loudly, as before.
The remote side is a fake `ssh` that answers each step and can fail the recreate
a given number of times.
"""
from __future__ import annotations
import os
import subprocess
from pathlib import Path
import pytest
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
_FAKE_SSH = """#!/bin/sh
# Swallow stdin (docker login, the .env heredoc), then answer by command.
cat > /dev/null
for arg in "$@"; do cmd="$arg"; done
case "$cmd" in
*--force-recreate*)
n=$(cat "{state}/recreates" 2>/dev/null || echo 0)
n=$((n + 1))
echo "$n" > "{state}/recreates"
if [ "$n" -le "{failures}" ]; then
echo "Error response from daemon: No such container: 02586df7e659"
exit 1
fi
echo " Container netork-netork-api-1 Started"
;;
*" ps -a"*)
echo "netork-netork-api-1 Created"
;;
esac
exit 0
"""
@pytest.fixture
def deploy(tmp_path: Path):
"""Run a deploy against the fake ssh; *failures* recreates fail before one works."""
def _deploy(failures: int) -> tuple[subprocess.CompletedProcess[str], int]:
bin_dir = tmp_path / "bin"
bin_dir.mkdir(exist_ok=True)
fake = bin_dir / "ssh"
fake.write_text(
_FAKE_SSH.replace("{state}", str(tmp_path)).replace("{failures}", str(failures))
)
fake.chmod(0o755)
result = subprocess.run(
["bash", str(DEPLOY), "testhost"],
env={
"PATH": f"{bin_dir}:{os.environ['PATH']}",
"HOME": str(tmp_path),
"DEPLOY_ENV_FILE": "/dev/null",
"REGISTRY_HOST": "registry.example",
"NETORK_VERSION": "latest-dev",
"DEPLOY_RECREATE_RETRY_DELAY": "0",
},
stdin=subprocess.DEVNULL,
capture_output=True,
text=True,
timeout=60,
check=False,
)
recreates = tmp_path / "recreates"
return result, int(recreates.read_text()) if recreates.exists() else 0
return _deploy
def test_a_recreate_that_works_runs_once(deploy) -> None:
result, recreates = deploy(failures=0)
assert result.returncode == 0, result.stderr
assert recreates == 1
def test_a_recreate_that_fails_once_is_tried_again(deploy) -> None:
result, recreates = deploy(failures=1)
assert result.returncode == 0, result.stderr
assert recreates == 2
assert "trying once more" in result.stderr
assert "All servers deployed successfully." in result.stdout
def test_two_failed_recreates_fail_the_deploy_and_show_the_containers(deploy) -> None:
result, recreates = deploy(failures=2)
assert result.returncode != 0
assert recreates == 2
assert "netork-netork-api-1 Created" in result.stderr
assert "FAILED on: testhost" in result.stderr