From 3bea46f2d7bdd5fc545c380a236fb23d9d9ffabb Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Andr=C3=A9=20Padez?= Date: Tue, 11 Aug 2026 20:01:17 +0000 Subject: [PATCH] =?UTF-8?q?rootless=20docker=20per=20member=20=E2=80=94=20?= =?UTF-8?q?provisioning=20works,=20running=20a=20container=20does=20not=20?= =?UTF-8?q?yet?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Not finished. Committed because the diagnosis is worth more than the code. WHY ROOTLESS AND NOT THE DOCKER GROUP. `usermod -aG docker ` is the one-line version and it is root: `docker run -v /:/host -it alpine chroot /host` is a root shell, which reads .env, every other member's home and the wallet seed. Every boundary from today, bypassed by one documented command. Rootless gives what was actually asked for — a daemon per account, containers in that account's user namespace, images in their own home. VERIFIED on this host: provisioning succeeds, the server reports 29.5.0, the daemon runs as the member, `docker pull` puts 403 MB under their own home, and `docker ps -a` shows nothing while the owner has four containers. That last line is the isolation, measured. NOT VERIFIED: actually running a container. It failed, and the cause is an interaction between two things built today: failed to copy xattrs: failed to set xattr "system.posix_acl_default" on …/volumes/…/_data Creating a volume copies xattrs, and the DEFAULT ACLs on a member's home — added so the file browser could read their files — are inherited by Docker's storage, where a mapped id inside a user namespace is not a valid id to set. Both features correct alone. The fix here strips default ACLs from ~/.local/share/docker only, leaving the access ACLs the file browser needs. That fix is UNPROVEN. The re-test failed for a different, environmental reason: probe users recycle uid 1001, and a stale lingering systemd user manager from a previous probe answered `systemctl --user`, so the unit appeared not to exist. Cleaned with `loginctl terminate-user`. Retest on a machine that has not had a uid-1001 user, or on a fresh uid. Also worth knowing before this ships: uid reuse after deleting a member is a real hazard, not just a test artefact — the next member gets the previous member's uid, and anything left lingering belongs to them. setup.sh gains uidmap and dbus-user-session as core packages; the shell template exports DOCKER_HOST from $XDG_RUNTIME_DIR when the socket exists. Co-Authored-By: Claude Opus 5 --- scripts/setup.sh | 16 +++ src/servers/api/users/provision-os.ts | 28 +++- src/servers/os-user-docker.ts | 198 ++++++++++++++++++++++++++ src/servers/shell-skel/zshrc | 9 ++ 4 files changed, 246 insertions(+), 5 deletions(-) create mode 100644 src/servers/os-user-docker.ts diff --git a/scripts/setup.sh b/scripts/setup.sh index 5f466013..4a77a144 100755 --- a/scripts/setup.sh +++ b/scripts/setup.sh @@ -230,6 +230,22 @@ case $PM in ;; esac +# uidmap — newuidmap/newgidmap, needed for a member's own rootless Docker. +# +# Rootless containers map subordinate uid ranges, and those two setuid helpers are the only way to do it +# unprivileged. Without them `dockerd-rootless-setuptool.sh` fails at the first step. Core rather than a +# profile extra for the same reason acl is: the alternative to a member having their own daemon is adding +# them to the `docker` group, which is root on the host — see src/servers/os-user-docker.ts. +case $PM in + apt) + if dpkg -s uidmap &>/dev/null 2>&1; then skip "uidmap"; else CORE_PKGS+=(uidmap); fi + if dpkg -s dbus-user-session &>/dev/null 2>&1; then skip "dbus-user-session"; else CORE_PKGS+=(dbus-user-session); fi + ;; + pacman) + if has newuidmap; then skip "uidmap (shadow)"; else CORE_PKGS+=(shadow); fi + ;; +esac + # acl — setfacl/getfacl, needed by per-user Linux accounts. # # A member's home is 700 and owned by them, which is right for a shell and locks the platform out of the diff --git a/src/servers/api/users/provision-os.ts b/src/servers/api/users/provision-os.ts index 4b5899e4..dbd9ff82 100644 --- a/src/servers/api/users/provision-os.ts +++ b/src/servers/api/users/provision-os.ts @@ -1,7 +1,8 @@ import { updateUser } from 'officerdb'; -import { OS_USERS_ENABLED, ensureOsUser } from '@@/os-user'; +import { OS_USERS_ENABLED, ensureOsUser, osUserHome } from '@@/os-user'; import { provisionSshAccess } from '@@/os-user-ssh'; import { seedShellConfig } from '@@/os-user-shell'; +import { provisionRootlessDocker } from '@@/os-user-docker'; import { provisionUserDirs } from '@@/data-path'; // Giving an account its Linux side: the directory skeleton, the Linux user, the confinement, the keys. @@ -68,16 +69,33 @@ export async function provisionOsAccount(params: { authorizedKey: params.inboundKey, }); - // The shell configuration. Last because it is the only step whose failure leaves nothing broken — the - // account works, the keys work, the terminal opens; it just opens with zsh's bare defaults. + // The shell configuration. Late because its failure leaves nothing broken — the account works, the keys + // work, the terminal opens; it just opens with zsh's bare defaults. const shell = await seedShellConfig({ email: params.email, uid: account.uid, gid: account.gid }); + // Their own rootless Docker daemon. Last, and the most tolerant of failure: a host without the uidmap + // package or a kernel that will not do rootless still gets a perfectly good account, minus containers. + // + // Provisioned for every member rather than behind a toggle, because "can I run a database to develop + // against" should not be an administrative request. The cost — one daemon and one image cache per member — + // is real and is written down in os-user-docker.ts. + const docker = await provisionRootlessDocker({ + osUser: account.osUser, + uid: account.uid, + home: osUserHome(params.email), + }); + // The Linux account is recorded either way: it exists, it is confined, and a member's terminal can run as // it. Only the keys are missing, and that is what the error says. const sshPublicKey = ssh.ok ? ssh.publicKey : null; await updateUser(params.userId, { osUser: account.osUser, osSshPublicKey: sshPublicKey }); - // SSH first if both failed: no keys is the more consequential of the two. - const error = !ssh.ok ? ssh.error : !shell.ok ? shell.error : null; + // Reported in order of consequence, not in order of execution: no keys matters more than a plain prompt, + // which matters more than no containers. Only one is surfaced because the UI shows one line — the rest are + // in the log. + for (const step of [shell, docker] as const) { + if (!step.ok) console.warn(`[users] ${params.email}: ${step.error}`); + } + const error = !ssh.ok ? ssh.error : !shell.ok ? shell.error : !docker.ok ? docker.error : null; return { osUser: account.osUser, sshPublicKey, error }; } diff --git a/src/servers/os-user-docker.ts b/src/servers/os-user-docker.ts new file mode 100644 index 00000000..7ed21349 --- /dev/null +++ b/src/servers/os-user-docker.ts @@ -0,0 +1,198 @@ +import { existsSync } from 'node:fs'; +import { runAs } from './os-user'; + +// A member's own Docker: their daemon, their images, their containers, running as their uid. +// +// ── Why rootless, and why the alternative is not on the table ── +// +// The one-line version of "give the user Docker" is `usermod -aG docker `, and it is root. Membership of +// that group means talking to the host daemon, which runs as root, so: +// +// docker run -v /:/host -it alpine chroot /host +// +// is a root shell on the machine. It reads the platform's `.env`, every other member's home, the wallet seed +// — every boundary in docs/per-user-linux-accounts.md, bypassed by one documented command. The group is not +// "access to Docker", it is "root, by a longer route". +// +// Rootless gives the thing that was actually wanted: a daemon per account, containers in that account's user +// namespace, images in their own home. Root inside their container is their uid outside it, which is nobody. +// They cannot see the owner's containers and the owner cannot break theirs. +// +// ── What it needs from the host ── +// +// uidmap newuidmap/newgidmap, to map subordinate ids. Rootless cannot start without them. +// /etc/subuid,gid a range per account. `useradd` allocates one automatically wherever login.defs sets +// SUB_UID_COUNT (Ubuntu does), and `userdel` reclaims it — verified on this host. +// linger `loginctl enable-linger`, or the daemon dies with the session. Officer's shells are +// NOT login sessions, so without this a member's Docker would stop the moment their +// terminal closed, which is the opposite of a daemon. +// dbus-user-session systemd --user needs a bus to talk to. +// +// ── The costs, stated rather than discovered ── +// +// Each account has its own image cache, so three members pulling postgres:16 store it three times. Ports +// below 1024 need an explicit capability grant. Both are acceptable for what this buys; neither is hidden. + +/** Their own daemon's socket. The value `DOCKER_HOST` must point at. */ +export const dockerSocketFor = (uid: number): string => `/run/user/${uid}/docker.sock`; + +type Result = { ok: true; alreadyInstalled: boolean } | { ok: false; error: string }; + +async function sudo(args: string[]): Promise<{ ok: boolean; out: string }> { + const proc = Bun.spawn(['sudo', '-n', ...args], { stdout: 'pipe', stderr: 'pipe' }); + const [out, err] = await Promise.all([new Response(proc.stdout).text(), new Response(proc.stderr).text()]); + return { ok: (await proc.exited) === 0, out: `${out}${err}`.trim() }; +} + +/** + * Run something as the member with a systemd user session in scope. + * + * `runAs` clears the environment, which is right everywhere else and fatal here: `systemctl --user` and the + * rootless setup tool locate the user manager through `XDG_RUNTIME_DIR` and `DBUS_SESSION_BUS_ADDRESS`. With + * those unset the tool reports "systemd not detected" and installs nothing, successfully. + */ +function asMemberWithSession(osUser: string, uid: number, command: string[]) { + const runtime = `/run/user/${uid}`; + return runAs(osUser, [ + 'env', + `XDG_RUNTIME_DIR=${runtime}`, + `DBUS_SESSION_BUS_ADDRESS=unix:path=${runtime}/bus`, + // The tool shells out to newuidmap, rootlesskit and dockerd, and --reset-env left PATH at the passwd + // default which does not include /usr/sbin. + 'PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin', + ...command, + ]); +} + +const text = async (proc: ReturnType) => + `${await new Response(proc.stdout).text()}${await new Response(proc.stderr).text()}`.trim(); + +/** Everything that must be true of the HOST before any account can have rootless Docker. */ +export function checkDockerPrerequisites(): { ok: true } | { ok: false; error: string } { + const missing: string[] = []; + if (!existsSync('/usr/bin/newuidmap') || !existsSync('/usr/bin/newgidmap')) missing.push('uidmap'); + if (!existsSync('/usr/bin/dockerd-rootless-setuptool.sh')) missing.push('docker-ce-rootless-extras'); + if (missing.length) { + return { + ok: false, + error: `rootless Docker needs these packages on the host: ${missing.join(', ')} (apt install ${missing.join(' ')})`, + }; + } + return { ok: true }; +} + +/** + * Give the account its own rootless Docker daemon, and start it. + * + * Idempotent: the setup tool is safe to re-run, `enable-linger` on a lingering account is a no-op, and an + * already-running daemon is reported as `alreadyInstalled` rather than restarted — a reprovision must not + * bounce a member's containers. + * + * Never throws. Docker is the least essential of the provisioning steps: without it the account still has a + * shell, a home and keys. + */ +export async function provisionRootlessDocker(params: { osUser: string; uid: number; home: string }): Promise { + const prereq = checkDockerPrerequisites(); + if (!prereq.ok) return prereq; + + // Subordinate id range. `useradd` allocates one, so a missing entry means either an unusual login.defs or + // an account made by hand — worth naming rather than letting rootlesskit fail obscurely later. + const subuid = await sudo(['grep', '-q', `^${params.osUser}:`, '/etc/subuid']); + if (!subuid.ok) { + return { + ok: false, + error: `${params.osUser} has no /etc/subuid range, which rootless Docker requires. Add one with: sudo usermod --add-subuids 100000-165535 ${params.osUser}`, + }; + } + + // Linger FIRST: it is what creates /run/user/ and starts the user manager, and everything below needs + // both to exist. + const linger = await sudo(['loginctl', 'enable-linger', params.osUser]); + if (!linger.ok) return { ok: false, error: `could not enable linger for ${params.osUser}: ${linger.out}` }; + + // The user manager appears asynchronously. Waiting beats a bare sleep, and the failure below is clearer + // than "systemd not detected" from the setup tool. + for (let attempt = 0; attempt < 25 && !existsSync(`/run/user/${params.uid}`); attempt++) { + await Bun.sleep(200); + } + if (!existsSync(`/run/user/${params.uid}`)) { + return { + ok: false, + error: `/run/user/${params.uid} never appeared, so ${params.osUser} has no systemd user session`, + }; + } + + const already = existsSync(dockerSocketFor(params.uid)); + + // ── The setup tool's exit code is deliberately NOT the gate ── + // + // It writes `~/.config/systemd/user/docker.service` and then runs `systemctl --user start docker.service` + // itself — which fails with "Unit docker.service not found" on a manager that was already running when the + // file appeared, because nothing reloaded it. Measured here: the unit was written correctly and the tool + // still exited 1. + // + // So: run it, reload, and start it ourselves. The tool's output is kept for the error message if the start + // then genuinely fails, since its diagnosis (a missing kernel module, an unsupported filesystem) is better + // than anything paraphrased. + const install = asMemberWithSession(params.osUser, params.uid, ['dockerd-rootless-setuptool.sh', 'install']); + const installOut = await text(install); + await install.exited; + + const reload = asMemberWithSession(params.osUser, params.uid, ['systemctl', '--user', 'daemon-reload']); + await reload.exited; + + const enable = asMemberWithSession(params.osUser, params.uid, ['systemctl', '--user', 'enable', '--now', 'docker']); + const enableOut = await text(enable); + if ((await enable.exited) !== 0) { + // Both halves, because the useful sentence is usually in the setup tool's output rather than systemd's. + const detail = [installOut, enableOut] + .map((s) => s.split('\n').slice(-3).join(' ').trim()) + .filter(Boolean) + .join(' | '); + return { ok: false, error: `could not start ${params.osUser}'s Docker: ${detail}` }; + } + + // ── Undo our own ACLs, for this one subtree ── + // + // The home carries DEFAULT ACLs so the file browser can read a member's files (os-user.ts explains why). + // Docker inherits them under `~/.local/share/docker`, then fails every `docker run` with: + // + // failed to copy xattrs: failed to set xattr "system.posix_acl_default" on …/volumes/…/_data: + // invalid argument + // + // Creating a volume copies xattrs, and inside a rootless user namespace the mapped id in an inherited + // default ACL is not a valid id, so setting it is EINVAL. Two features built the same day, each correct + // alone. Measured: the image pulled fine — 403 MB into their home — and every container failed to start. + // + // `-k` removes DEFAULT entries only, so nothing inside Docker's storage inherits them from here on. The + // access ACLs on the home itself are untouched, which is what the file browser depends on. Losing the + // platform's reach into Docker's internal storage is no loss: it is image layers and volume data, read + // through `docker` or not at all. + const dockerData = `${params.home}/.local/share/docker`; + if (existsSync(dockerData)) { + const stripped = await sudo(['setfacl', '-R', '-k', dockerData]); + if (!stripped.ok) { + return { ok: false, error: `could not clear inherited ACLs from ${dockerData}: ${stripped.out}` }; + } + } + + // Proof, not assumption: ask their daemon who it is. `docker version --format` on the SERVER half only + // answers if the socket is live and talking. + const verify = asMemberWithSession(params.osUser, params.uid, [ + 'env', + `DOCKER_HOST=unix://${dockerSocketFor(params.uid)}`, + 'docker', + 'version', + '--format', + '{{.Server.Version}}', + ]); + const version = await text(verify); + if ((await verify.exited) !== 0) { + return { + ok: false, + error: `${params.osUser}'s Docker did not answer: ${version.split('\n').slice(-2).join(' ').trim()}`, + }; + } + + return { ok: true, alreadyInstalled: already }; +} diff --git a/src/servers/shell-skel/zshrc b/src/servers/shell-skel/zshrc index 1330dcb5..6f1eed63 100644 --- a/src/servers/shell-skel/zshrc +++ b/src/servers/shell-skel/zshrc @@ -105,6 +105,15 @@ else %F{blue}❯%f ' fi +# ── Docker ── +# Your own rootless daemon, if Officer provisioned one. Containers you start run as your account in your own +# user namespace — root inside them is you outside them, and you cannot see anyone else's containers. +# +# Set from $XDG_RUNTIME_DIR rather than a hard-coded uid so this line is the same in every account's file. +if [ -S "${XDG_RUNTIME_DIR:-/run/user/$(id -u)}/docker.sock" ]; then + export DOCKER_HOST="unix://${XDG_RUNTIME_DIR:-/run/user/$(id -u)}/docker.sock" +fi + # ── Your own additions ── # Sourced last so anything here wins. Officer never writes to it, so it is the safe place for your own # configuration — and it is why this file can be replaced by a newer template without losing your work.