mirror of
https://github.com/home-assistant/supervisor.git
synced 2026-09-29 22:12:59 +01:00
* Treat containers with corrupt storage metadata as missing Since Docker 29.4, containers whose RW layer fails to load while the daemon restores its state at startup (e.g. after storage metadata was corrupted by an unclean shutdown) are kept registered so they can still be removed, instead of being dropped (moby/moby#51724). Every other operation on such a container fails with a 500 "RWLayer of container <id> is unexpectedly nil" error, including the inspect at the start of nearly every Supervisor container operation. A single corrupt container record therefore permanently breaks starting, restarting, rebuilding and backing up of the affected add-on, plugin or Home Assistant Core: the start path errors out on the initial state check, before reaching the cleanup that would remove the broken record. Older Docker versions dropped such containers at daemon startup, so Supervisor saw them as missing and recreated them, healing the installation as a side effect. Restore that behavior explicitly by reporting a container as missing when a Docker API call fails with this error. Recovery paths then recreate the container as before, with stop_container's force delete (which no longer inspects first since #7175) removing the broken record along the way. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * Broaden corrupt container detection and heal at start/restart The detection matched only one shape of the nil-RW-layer message, the inspect-time one built from the container id. Docker phrases the same condition differently per call site: with the container name instead of its id, with reversed word order on the containerd image store, and bare on Windows. Loosen the pattern to cover them; "unexpectedly nil" appears nowhere else in Docker, so it stays just as narrow. With the containerd image store the inspect of such a container succeeds entirely (the RW layer check in daemon/inspect.go is graph driver only), so none of the inspect-time handling fires: the error only surfaces from the actual start or restart call. Handle it there as well by removing the broken record and reporting the container as missing, so the next start takes the recreate path. Sentry shows this shape in the field (SUPERVISOR-1B6Q). Also soften the corrupt-record comment: the daemon marks the record once while restoring state, for any error loading the RW layer, and a daemon restart may recover it. Removal and recreation is still the safe recovery for Supervisor-managed containers, which hold no state. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * Document why corrupt container removal stays at the job-locked sites Review feedback asked to remove the corrupt record right where it is detected instead of reporting the container as missing. The record indeed cannot be used for anything until it is removed, but removal has to go by name, and the read paths (is_running() and friends) run outside the job locks that serialize a container's lifecycle. Names are only unique at a given instant — the heal deletes the broken record and then recreates the container under the same name — so a delete-by-name from an unlocked path after a stale inspect could hit the freshly recreated container rather than the broken one. Removal therefore stays where the job lock is held: the inner start and restart failures (added in the previous commit for the containerd image store, where only those calls see the error) and stop_container's cleanup, which every recreate flow reaches moments after detection. The read paths keep reporting the container as missing without acting on it. Say so in the code, and pin the split in tests: the read paths must not remove the record, the locked sites must. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
158 lines
5.9 KiB
Python
158 lines
5.9 KiB
Python
"""Test Docker utilities."""
|
|
|
|
import aiodocker
|
|
import pytest
|
|
|
|
from supervisor.docker.const import DOCKER_HUB
|
|
from supervisor.docker.utils import (
|
|
get_registry_from_image,
|
|
is_corrupt_container_error,
|
|
is_registry_domain,
|
|
split_docker_domain,
|
|
split_image_tag,
|
|
)
|
|
|
|
CORRUPT_CONTAINER_MESSAGE = (
|
|
"RWLayer of container "
|
|
"1b56493ca170514364e10113038a16e9d207cb16a229be55ed6139649a39ca4e "
|
|
"is unexpectedly nil"
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("image_ref", "expected"),
|
|
[
|
|
# No registry, hosted on Docker Hub
|
|
("nginx", (None, "nginx")),
|
|
("nginx:latest", (None, "nginx:latest")),
|
|
("library/nginx", (None, "library/nginx")),
|
|
("homeassistant/amd64-supervisor", (None, "homeassistant/amd64-supervisor")),
|
|
# Registry with a dot
|
|
(
|
|
"ghcr.io/home-assistant/amd64-supervisor",
|
|
("ghcr.io", "home-assistant/amd64-supervisor"),
|
|
),
|
|
("registry.example.com/org/image:v1", ("registry.example.com", "org/image:v1")),
|
|
("127.0.0.1/myimage", ("127.0.0.1", "myimage")),
|
|
# Registry with a port
|
|
("myregistry:5000/myimage", ("myregistry:5000", "myimage")),
|
|
("registry.io:5000/org/app:v1", ("registry.io:5000", "org/app:v1")),
|
|
# localhost is a reserved namespace and always a registry
|
|
("localhost/myimage", ("localhost", "myimage")),
|
|
("localhost:5000/myimage:tag", ("localhost:5000", "myimage:tag")),
|
|
# IPv6 registry
|
|
("[::1]:5000/myimage", ("[::1]:5000", "myimage")),
|
|
("[2001:db8::1]:5000/myimage:tag", ("[2001:db8::1]:5000", "myimage:tag")),
|
|
# Legacy Docker Hub domain gets canonicalized
|
|
("index.docker.io/library/nginx", (DOCKER_HUB, "library/nginx")),
|
|
# Uppercase is not allowed in a path component, so it is a registry
|
|
("Foo/bar", ("Foo", "bar")),
|
|
],
|
|
)
|
|
def test_split_docker_domain(image_ref: str, expected: tuple[str | None, str]):
|
|
"""Test splitting an image reference into registry domain and remainder."""
|
|
assert split_docker_domain(image_ref) == expected
|
|
|
|
|
|
def test_get_registry_from_image():
|
|
"""Test get_registry_from_image returns only the registry domain."""
|
|
assert get_registry_from_image("ghcr.io/home-assistant/supervisor") == "ghcr.io"
|
|
assert get_registry_from_image("homeassistant/supervisor") is None
|
|
assert get_registry_from_image("index.docker.io/library/nginx") == DOCKER_HUB
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("domain", "valid"),
|
|
[
|
|
("ghcr.io", True),
|
|
("registry.example.com", True),
|
|
("myregistry:5000", True),
|
|
("localhost", True),
|
|
("localhost:5000", True),
|
|
("127.0.0.1", True),
|
|
("[::1]:5000", True),
|
|
("[2001:db8::1]", True),
|
|
# Malformed domains
|
|
(".ghcr.io", False),
|
|
("ghcr.io.", False),
|
|
("-bad-.com", False),
|
|
("bad-.com", False),
|
|
("....", False),
|
|
("ghcr.io:", False),
|
|
("ghcr.io:port", False),
|
|
("ghcr.io/org", False),
|
|
],
|
|
)
|
|
def test_is_registry_domain(domain: str, valid: bool):
|
|
"""Test validation of registry domains."""
|
|
assert is_registry_domain(domain) is valid
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("image_ref", "expected"),
|
|
[
|
|
# No tag
|
|
("nginx", ("nginx", None)),
|
|
("library/nginx", ("library/nginx", None)),
|
|
(
|
|
"ghcr.io/home-assistant/amd64-supervisor",
|
|
("ghcr.io/home-assistant/amd64-supervisor", None),
|
|
),
|
|
# With tag
|
|
("nginx:latest", ("nginx", "latest")),
|
|
(
|
|
"homeassistant/amd64-supervisor:1.2.3",
|
|
("homeassistant/amd64-supervisor", "1.2.3"),
|
|
),
|
|
# Registry with a port, the port must stay part of the image name
|
|
("myregistry:5000/myimage", ("myregistry:5000/myimage", None)),
|
|
("registry.io:5000/org/app:v1", ("registry.io:5000/org/app", "v1")),
|
|
(
|
|
"gitlab.example.com:5005/org/app/aarch64:0.3.3-dev1",
|
|
("gitlab.example.com:5005/org/app/aarch64", "0.3.3-dev1"),
|
|
),
|
|
# localhost with a port
|
|
("localhost:5000/myimage", ("localhost:5000/myimage", None)),
|
|
("localhost:5000/myimage:tag", ("localhost:5000/myimage", "tag")),
|
|
# IPv6 registry
|
|
("[::1]:5000/myimage", ("[::1]:5000/myimage", None)),
|
|
("[2001:db8::1]:5000/myimage:tag", ("[2001:db8::1]:5000/myimage", "tag")),
|
|
# Digests are stripped along with the tag
|
|
("nginx@sha256:1234abcd", ("nginx", None)),
|
|
("ghcr.io/org/app@sha256:1234abcd", ("ghcr.io/org/app", None)),
|
|
# A bare digest keeps its algorithm prefix as the name
|
|
("sha256:1234abcd", ("sha256", "1234abcd")),
|
|
],
|
|
)
|
|
def test_split_image_tag(image_ref: str, expected: tuple[str, str | None]):
|
|
"""Test splitting an image reference into image name and tag."""
|
|
assert split_image_tag(image_ref) == expected
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("status", "message", "expected"),
|
|
[
|
|
(500, CORRUPT_CONTAINER_MESSAGE, True),
|
|
(500, f"[500] {CORRUPT_CONTAINER_MESSAGE}", True),
|
|
# Docker phrases the message differently per call site: with the
|
|
# container name instead of its id, with reversed word order on the
|
|
# containerd image store, or bare on Windows.
|
|
(500, "RWLayer of container /addon_ssh is unexpectedly nil", True),
|
|
(500, "RWLayer is unexpectedly nil for container 1b56493ca170", True),
|
|
(500, "RWLayer is unexpectedly nil", True),
|
|
(500, "stat /mnt/data/docker/overlay2/abc: no such file or directory", False),
|
|
(
|
|
500,
|
|
"error creating overlay mount to /mnt/data/docker/overlay2/abc/merged: "
|
|
"no such file or directory",
|
|
False,
|
|
),
|
|
(404, CORRUPT_CONTAINER_MESSAGE, False),
|
|
],
|
|
)
|
|
def test_is_corrupt_container_error(status: int, message: str, expected: bool):
|
|
"""Test detection of Docker's corrupt container error."""
|
|
assert (
|
|
is_corrupt_container_error(aiodocker.DockerError(status, message)) == expected
|
|
)
|