Files
deploy/tests/test_deploy_recreate_retry.py
Christian Manivong 923ec4ce65
CI / check (pull_request) Successful in 13s
fix: a failed container recreate is tried once more before the deploy gives up
On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers":
docker compose up -d --force-recreate recreates the services in parallel,
lost a container it had just renamed ("No such container: 02586df7…") and
stopped with every engine container Created and none running. The old API
was already gone, so netOrk was down for about two minutes, until the same
deploy was run again and went through cleanly.

The script now does that second run itself, after DEPLOY_RECREATE_RETRY_DELAY
seconds (5 by default). If the second attempt fails too, it prints the
container states (docker compose ps -a) and fails as before.

Tests run the script against a fake ssh that fails the recreate zero, one
or two times. README step 3 says what happens.

Refs NetOrk/netork#586
2026-10-05 21:58:36 +02:00

107 lines
3.4 KiB
Python

"""A failed container recreate is tried once more before the deploy gives up.
On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers":
`docker compose up -d --force-recreate` recreates the services in parallel, lost
a container it had just renamed ("No such container: 02586df7…"), and stopped
with every engine container `Created` and none running. The old API was already
gone, so netOrk was down until somebody ran the same deploy again, which then
went through cleanly (NetOrk/netork#586).
So the script runs that step a second time on its own. Should the second attempt
fail too, it shows what the containers look like and fails loudly, as before.
The remote side is a fake `ssh` that answers each step and can fail the recreate
a given number of times.
"""
from __future__ import annotations
import os
import subprocess
from pathlib import Path
import pytest
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
_FAKE_SSH = """#!/bin/sh
# Swallow stdin (docker login, the .env heredoc), then answer by command.
cat > /dev/null
for arg in "$@"; do cmd="$arg"; done
case "$cmd" in
*--force-recreate*)
n=$(cat "{state}/recreates" 2>/dev/null || echo 0)
n=$((n + 1))
echo "$n" > "{state}/recreates"
if [ "$n" -le "{failures}" ]; then
echo "Error response from daemon: No such container: 02586df7e659"
exit 1
fi
echo " Container netork-netork-api-1 Started"
;;
*" ps -a"*)
echo "netork-netork-api-1 Created"
;;
esac
exit 0
"""
@pytest.fixture
def deploy(tmp_path: Path):
"""Run a deploy against the fake ssh; *failures* recreates fail before one works."""
def _deploy(failures: int) -> tuple[subprocess.CompletedProcess[str], int]:
bin_dir = tmp_path / "bin"
bin_dir.mkdir(exist_ok=True)
fake = bin_dir / "ssh"
fake.write_text(
_FAKE_SSH.replace("{state}", str(tmp_path)).replace("{failures}", str(failures))
)
fake.chmod(0o755)
result = subprocess.run(
["bash", str(DEPLOY), "testhost"],
env={
"PATH": f"{bin_dir}:{os.environ['PATH']}",
"HOME": str(tmp_path),
"DEPLOY_ENV_FILE": "/dev/null",
"REGISTRY_HOST": "registry.example",
"NETORK_VERSION": "latest-dev",
"DEPLOY_RECREATE_RETRY_DELAY": "0",
},
stdin=subprocess.DEVNULL,
capture_output=True,
text=True,
timeout=60,
check=False,
)
recreates = tmp_path / "recreates"
return result, int(recreates.read_text()) if recreates.exists() else 0
return _deploy
def test_a_recreate_that_works_runs_once(deploy) -> None:
result, recreates = deploy(failures=0)
assert result.returncode == 0, result.stderr
assert recreates == 1
def test_a_recreate_that_fails_once_is_tried_again(deploy) -> None:
result, recreates = deploy(failures=1)
assert result.returncode == 0, result.stderr
assert recreates == 2
assert "trying once more" in result.stderr
assert "All servers deployed successfully." in result.stdout
def test_two_failed_recreates_fail_the_deploy_and_show_the_containers(deploy) -> None:
result, recreates = deploy(failures=2)
assert result.returncode != 0
assert recreates == 2
assert "netork-netork-api-1 Created" in result.stderr
assert "FAILED on: testhost" in result.stderr