Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6444aa97f3 | ||
|
|
923ec4ce65 |
@@ -70,7 +70,10 @@ started earlier pulls whatever image the registry held before, which is stale.
|
||||
2. Pulls the engine image and copies `docker-compose.yml` and
|
||||
`docker-compose.registry.yml` out of it into `~/netork/`.
|
||||
3. Pulls the netOrk images, then force-recreates the API, the workers, `netork-beat`,
|
||||
`flower` and, on UI servers, `netork-ui`.
|
||||
`flower` and, on UI servers, `netork-ui`. If the recreate fails, it is tried once
|
||||
more after 5 s (`DEPLOY_RECREATE_RETRY_DELAY`). Compose's parallel recreate can lose
|
||||
a container it just renamed and leave the rest stopped. If the second attempt fails
|
||||
too, the container states are printed and the deploy fails.
|
||||
4. Reconciles `registry`, `apt-cacher-ng` and `signal-api`: each is recreated only if its
|
||||
definition changed. It never touches `postgres` or `redis`.
|
||||
5. Verifies that the containers run exactly the image that was pulled. If they don't, the
|
||||
|
||||
@@ -194,12 +194,31 @@ deploy_server() {
|
||||
# observed on netork-ui, which kept serving a stale image after its tag had
|
||||
# already advanced. Recreating unconditionally costs a restart per deploy,
|
||||
# which a deliberate rollout wants anyway. See NetOrk/netork#90.
|
||||
#
|
||||
# Tried twice. compose recreates the services in parallel, and on 2026-10-05 it
|
||||
# lost a container it had just renamed ("No such container") and stopped with
|
||||
# every engine container Created and none running: the old API was gone, so
|
||||
# netOrk was down until someone ran the deploy again, which then went through
|
||||
# (NetOrk/netork#586). The second attempt is that rerun. Should it fail as well,
|
||||
# the container states go to the log and the deploy fails as before.
|
||||
echo "[${SERVER}] Starting containers..."
|
||||
run_remote "$SERVER" "recreating ${SERVICES}" "" \
|
||||
"cd ~/netork && \
|
||||
local RECREATE="cd ~/netork && \
|
||||
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
|
||||
docker compose -f docker-compose.yml -f docker-compose.registry.yml \
|
||||
up -d --no-deps --force-recreate ${SERVICES}"
|
||||
if ! run_remote "$SERVER" "recreating ${SERVICES}" "" "$RECREATE"; then
|
||||
local delay="${DEPLOY_RECREATE_RETRY_DELAY:-5}"
|
||||
echo "[${SERVER}] Recreating failed; trying once more in ${delay}s..." >&2
|
||||
sleep "$delay"
|
||||
if ! run_remote "$SERVER" "recreating ${SERVICES} (second attempt)" "" "$RECREATE"; then
|
||||
echo "[${SERVER}] Containers after two failed attempts:" >&2
|
||||
ssh -n "$SERVER" "cd ~/netork && \
|
||||
REGISTRY_HOST='${REGISTRY_HOST}' NETORK_VERSION='${SERVER_VERSION}' \
|
||||
docker compose -f docker-compose.yml -f docker-compose.registry.yml ps -a" 2>&1 \
|
||||
| sed "s/^/[${SERVER}] /" >&2 || true
|
||||
return 1
|
||||
fi
|
||||
fi
|
||||
|
||||
# Infrastructure containers, deliberately left out of the force-recreate
|
||||
# above. Plain `up -d`: compose compares each service definition against the
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
"""A failed container recreate is tried once more before the deploy gives up.
|
||||
|
||||
On 2026-10-05 a deploy to 172.22.8.50 failed in "Starting containers":
|
||||
`docker compose up -d --force-recreate` recreates the services in parallel, lost
|
||||
a container it had just renamed ("No such container: 02586df7…"), and stopped
|
||||
with every engine container `Created` and none running. The old API was already
|
||||
gone, so netOrk was down until somebody ran the same deploy again, which then
|
||||
went through cleanly (NetOrk/netork#586).
|
||||
|
||||
So the script runs that step a second time on its own. Should the second attempt
|
||||
fail too, it shows what the containers look like and fails loudly, as before.
|
||||
|
||||
The remote side is a fake `ssh` that answers each step and can fail the recreate
|
||||
a given number of times.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
DEPLOY = Path(__file__).resolve().parent.parent / "deploy.sh"
|
||||
|
||||
_FAKE_SSH = """#!/bin/sh
|
||||
# Swallow stdin (docker login, the .env heredoc), then answer by command.
|
||||
cat > /dev/null
|
||||
for arg in "$@"; do cmd="$arg"; done
|
||||
case "$cmd" in
|
||||
*--force-recreate*)
|
||||
n=$(cat "{state}/recreates" 2>/dev/null || echo 0)
|
||||
n=$((n + 1))
|
||||
echo "$n" > "{state}/recreates"
|
||||
if [ "$n" -le "{failures}" ]; then
|
||||
echo "Error response from daemon: No such container: 02586df7e659"
|
||||
exit 1
|
||||
fi
|
||||
echo " Container netork-netork-api-1 Started"
|
||||
;;
|
||||
*" ps -a"*)
|
||||
echo "netork-netork-api-1 Created"
|
||||
;;
|
||||
esac
|
||||
exit 0
|
||||
"""
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def deploy(tmp_path: Path):
|
||||
"""Run a deploy against the fake ssh; *failures* recreates fail before one works."""
|
||||
|
||||
def _deploy(failures: int) -> tuple[subprocess.CompletedProcess[str], int]:
|
||||
bin_dir = tmp_path / "bin"
|
||||
bin_dir.mkdir(exist_ok=True)
|
||||
fake = bin_dir / "ssh"
|
||||
fake.write_text(
|
||||
_FAKE_SSH.replace("{state}", str(tmp_path)).replace("{failures}", str(failures))
|
||||
)
|
||||
fake.chmod(0o755)
|
||||
result = subprocess.run(
|
||||
["bash", str(DEPLOY), "testhost"],
|
||||
env={
|
||||
"PATH": f"{bin_dir}:{os.environ['PATH']}",
|
||||
"HOME": str(tmp_path),
|
||||
"DEPLOY_ENV_FILE": "/dev/null",
|
||||
"REGISTRY_HOST": "registry.example",
|
||||
"NETORK_VERSION": "latest-dev",
|
||||
"DEPLOY_RECREATE_RETRY_DELAY": "0",
|
||||
},
|
||||
stdin=subprocess.DEVNULL,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=60,
|
||||
check=False,
|
||||
)
|
||||
recreates = tmp_path / "recreates"
|
||||
return result, int(recreates.read_text()) if recreates.exists() else 0
|
||||
|
||||
return _deploy
|
||||
|
||||
|
||||
def test_a_recreate_that_works_runs_once(deploy) -> None:
|
||||
result, recreates = deploy(failures=0)
|
||||
|
||||
assert result.returncode == 0, result.stderr
|
||||
assert recreates == 1
|
||||
|
||||
|
||||
def test_a_recreate_that_fails_once_is_tried_again(deploy) -> None:
|
||||
result, recreates = deploy(failures=1)
|
||||
|
||||
assert result.returncode == 0, result.stderr
|
||||
assert recreates == 2
|
||||
assert "trying once more" in result.stderr
|
||||
assert "All servers deployed successfully." in result.stdout
|
||||
|
||||
|
||||
def test_two_failed_recreates_fail_the_deploy_and_show_the_containers(deploy) -> None:
|
||||
result, recreates = deploy(failures=2)
|
||||
|
||||
assert result.returncode != 0
|
||||
assert recreates == 2
|
||||
assert "netork-netork-api-1 Created" in result.stderr
|
||||
assert "FAILED on: testhost" in result.stderr
|
||||
Reference in New Issue
Block a user