From fc228a809c5a0ad3981c60bec2c279e4c991b2fd Mon Sep 17 00:00:00 2001 From: Jason Lernerman Date: Fri, 25 Sep 2026 03:56:23 -0400 Subject: [PATCH] dind: run dockerd in its own cgroup namespace on cgroup v2 Kubernetes does not give privileged containers a cgroup namespace on cgroup v2 (KEP-2254), so in a privileged dind pod /sys/fs/cgroup is the node's root: "dind" sets up its nesting there, moving whatever sits in the node's root cgroup into /init, and dockerd creates its containers in /docker/ next to kubepods.slice, outside the pod's cgroup and so outside its limits and accounting (moby/moby#45378). When the container's cgroup is not the root of its namespace, enter a new cgroup and mount namespace with unshare and re-mount /sys/fs/cgroup so it is rooted there, the same way moby/buildkit#6368 does for the buildkit image. The mount options are carried over because cgroup2 applies them to the whole hierarchy. If that is not possible, warn and carry on as before. Signed-off-by: Jason Lernerman --- 29/dind/Dockerfile | 1 + 29/dind/dockerd-entrypoint.sh | 18 ++++++++++++++++++ Dockerfile-dind.template | 1 + dockerd-entrypoint.sh | 18 ++++++++++++++++++ 4 files changed, 38 insertions(+) diff --git a/29/dind/Dockerfile b/29/dind/Dockerfile index 886fa6c43..2d4a13383 100644 --- a/29/dind/Dockerfile +++ b/29/dind/Dockerfile @@ -16,6 +16,7 @@ RUN set -eux; \ openssl \ pigz \ shadow-uidmap \ + util-linux-misc \ xz \ ; diff --git a/29/dind/dockerd-entrypoint.sh b/29/dind/dockerd-entrypoint.sh index 77ff1aab3..c9b527f1c 100755 --- a/29/dind/dockerd-entrypoint.sh +++ b/29/dind/dockerd-entrypoint.sh @@ -230,6 +230,24 @@ if [ "$1" = 'dockerd' ]; then # if we have the (mostly defunct now) Docker-in-Docker wrapper script, use it set -- '/usr/local/bin/dind' "$@" fi + + # cgroup v2: make sure we are at the root of our own cgroup namespace + # Kubernetes runs privileged containers in the host's cgroup namespace (https://github.com/kubernetes/enhancements/tree/master/keps/sig-node/2254-cgroup-v2#cgroup-namespace), so "dind" would set up nesting at the *host's* cgroup root and dockerd would create its containers in "/docker/" next to "kubepods" -- outside the pod's cgroup, and thus outside its limits and accounting + # see https://github.com/moby/moby/issues/45378 (and https://github.com/moby/buildkit/pull/6368 for the same fix in buildkit's image) + if [ -f /sys/fs/cgroup/cgroup.controllers ] && [ "$(sed -n 's/^0:://p' /proc/self/cgroup)" != '/' ]; then + # re-mount /sys/fs/cgroup so it is rooted at the new namespace, keeping the existing mount options (the kernel applies them to the whole hierarchy) + cgroupns=' + opts="$(grep " /sys/fs/cgroup cgroup2 " /proc/self/mounts | tail -n1 | cut -d" " -f4)" + umount /sys/fs/cgroup + mount -t cgroup2 -o "$opts" cgroup2 /sys/fs/cgroup + exec "$@" + ' + if unshare --cgroup --mount sh -ec "$cgroupns" sh true 2>/dev/null; then + set -- unshare --cgroup --mount sh -ec "$cgroupns" sh "$@" + else + echo >&2 'warning: unable to create a cgroup namespace; containers will be created outside of the cgroup of this container' + fi + fi else # if it isn't `dockerd` we're trying to run, pass it through `docker-entrypoint.sh` so it gets `DOCKER_HOST` set appropriately too set -- docker-entrypoint.sh "$@" diff --git a/Dockerfile-dind.template b/Dockerfile-dind.template index 873144424..845bd2a55 100644 --- a/Dockerfile-dind.template +++ b/Dockerfile-dind.template @@ -11,6 +11,7 @@ RUN set -eux; \ openssl \ pigz \ shadow-uidmap \ + util-linux-misc \ xz \ ; diff --git a/dockerd-entrypoint.sh b/dockerd-entrypoint.sh index 77ff1aab3..c9b527f1c 100755 --- a/dockerd-entrypoint.sh +++ b/dockerd-entrypoint.sh @@ -230,6 +230,24 @@ if [ "$1" = 'dockerd' ]; then # if we have the (mostly defunct now) Docker-in-Docker wrapper script, use it set -- '/usr/local/bin/dind' "$@" fi + + # cgroup v2: make sure we are at the root of our own cgroup namespace + # Kubernetes runs privileged containers in the host's cgroup namespace (https://github.com/kubernetes/enhancements/tree/master/keps/sig-node/2254-cgroup-v2#cgroup-namespace), so "dind" would set up nesting at the *host's* cgroup root and dockerd would create its containers in "/docker/" next to "kubepods" -- outside the pod's cgroup, and thus outside its limits and accounting + # see https://github.com/moby/moby/issues/45378 (and https://github.com/moby/buildkit/pull/6368 for the same fix in buildkit's image) + if [ -f /sys/fs/cgroup/cgroup.controllers ] && [ "$(sed -n 's/^0:://p' /proc/self/cgroup)" != '/' ]; then + # re-mount /sys/fs/cgroup so it is rooted at the new namespace, keeping the existing mount options (the kernel applies them to the whole hierarchy) + cgroupns=' + opts="$(grep " /sys/fs/cgroup cgroup2 " /proc/self/mounts | tail -n1 | cut -d" " -f4)" + umount /sys/fs/cgroup + mount -t cgroup2 -o "$opts" cgroup2 /sys/fs/cgroup + exec "$@" + ' + if unshare --cgroup --mount sh -ec "$cgroupns" sh true 2>/dev/null; then + set -- unshare --cgroup --mount sh -ec "$cgroupns" sh "$@" + else + echo >&2 'warning: unable to create a cgroup namespace; containers will be created outside of the cgroup of this container' + fi + fi else # if it isn't `dockerd` we're trying to run, pass it through `docker-entrypoint.sh` so it gets `DOCKER_HOST` set appropriately too set -- docker-entrypoint.sh "$@"