diff --git a/docker/.env.example b/docker/.env.example index 1027b955..62919d27 100644 --- a/docker/.env.example +++ b/docker/.env.example @@ -44,6 +44,12 @@ CODEMAN_PASSWORD=changeme # Required. Username for Codeman HTTP Basic authentication. CODEMAN_USERNAME=admin +# Optional. Extra Host-header allowlist entries for a reverse-proxied domain +# (comma-separated; a bare `.suffix` matches every subdomain). Without it a +# proxied request is rejected with `403 Forbidden: host not allowed`. See +# README.md, "Reverse-proxy host allowlist". +# CODEMAN_ALLOWED_HOSTS=codeman.example.com,.internal.example.com + # Optional: authenticate Gemini CLI without an interactive login. GEMINI_API_KEY= diff --git a/docker/docker-compose.yaml b/docker/docker-compose.yaml index 1e888676..d26b2a97 100644 --- a/docker/docker-compose.yaml +++ b/docker/docker-compose.yaml @@ -32,6 +32,10 @@ services: CODEMAN_DOCKER_HOST_HOME: ${CODEMAN_APPDATA_PATH} CODEMAN_DOCKER_DISABLE_SWAP_LIMIT: ${CODEMAN_DOCKER_DISABLE_SWAP_LIMIT} CODEMAN_CASES_PATH: ${CODEMAN_CASES_PATH} + # Extra Host-header allowlist entries for a reverse-proxied deployment + # (docker/README.md, "Reverse-proxy host allowlist"). Optional, so it + # defaults to empty rather than requiring a line in every .env. + CODEMAN_ALLOWED_HOSTS: ${CODEMAN_ALLOWED_HOSTS:-} CODEMAN_HOST: ${CODEMAN_HOST} CODEMAN_PASSWORD: ${CODEMAN_PASSWORD} CODEMAN_PORT: ${CODEMAN_PORT} @@ -94,8 +98,18 @@ services: cap_add: # The entrypoint corrects bind-mount ownership as root before dropping to # PUID:PGID. Everything not listed here remains dropped by cap_drop above. + # test/docker-entrypoint.test.ts pins this list against what the + # entrypoint and `init: true` actually need, so a capability cannot go + # missing silently again. - CHOWN - DAC_OVERRIDE + # `init: true` makes tini PID 1, and tini stays ROOT while the entrypoint + # drops the server to PUID. Signalling a process of a different uid needs + # CAP_KILL; without it tini's SIGTERM forward fails ("Unexpected error + # when forwarding signal: 'Operation not permitted'"), tini dies, and the + # PID namespace teardown SIGKILLs the server instead of letting + # `server.stop()` flush state on every `docker compose down`/`restart`. + - KILL - SETGID - SETUID healthcheck: diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh index be4912e4..7cfd9111 100755 --- a/docker/entrypoint.sh +++ b/docker/entrypoint.sh @@ -15,6 +15,18 @@ # /opt/codeman-cli's ownership on every start, not just at image build time - # see the comment at that chown below for why that matters for anyone who # runs the compose file directly rather than through Start-Codeman.sh. +# +# Capabilities this script needs against the compose file's `cap_drop: ALL` +# (test/docker-entrypoint.test.ts pins the list against docker-compose.yaml): +# CHOWN + DAC_OVERRIDE the chown of a root-owned bind source below +# SETUID + SETGID the setpriv drop itself +# KILL NOT used here, but required by the container: with +# `init: true` tini is PID 1 and runs as root while the +# server runs as PUID, and signalling a process of a +# different uid needs CAP_KILL. Without it every +# `docker compose down`/`restart` ends in tini dying with +# "Unexpected error when forwarding signal" and the +# server being SIGKILLed instead of stopping cleanly. set -eu @@ -24,42 +36,95 @@ if [ "$(id -u)" -ne 0 ]; then exec "$@" fi +# Everything below runs as root and calls stat, chown, id, setpriv and friends +# by bare name, so the lookup path must not contain a directory the runtime +# account can write to. /opt/codeman-cli/bin is exactly that (it is chowned to +# PUID:PGID so sessions can update the agent CLIs in place), and the image +# appends it to PATH for the server's sake. Resolve root's commands through the +# system directories only, and hand the image's full PATH back to the server at +# the exec below, since Codeman resolves the agent CLIs through it. +runtime_path=$PATH +PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin +export PATH + : "${PUID:=1000}" : "${PGID:=1000}" +# The capabilities the compose file must grant, named in the diagnosis below so +# an out-of-tree compose file (Unraid's Compose Manager, a hand-written unit) +# fails with a one-line fix instead of a restart loop. +required_caps='CHOWN, DAC_OVERRIDE, KILL, SETGID, SETUID' + +# Pre-flight the drop itself before touching anything. A container started with +# `cap_drop: ALL` and none of the additions above fails here, and would otherwise +# die at the final exec with a bare "setpriv: setresuid failed: Operation not +# permitted" after chown had already failed, or worse, misreport a perfectly +# writable directory as unwritable because the probe below could not drop +# privileges to test it. +if ! setpriv --reuid "$PUID" --regid "$PGID" --clear-groups true 2>/dev/null; then + printf 'entrypoint: cannot drop privileges to PUID:PGID (%s:%s).\n' "$PUID" "$PGID" >&2 + printf 'entrypoint: this image starts as root and drops with setpriv, which needs\n' >&2 + printf 'entrypoint: cap_add: [%s]\n' "$required_caps" >&2 + printf 'entrypoint: on top of cap_drop: ALL (see docker/docker-compose.yaml). Add them to the\n' >&2 + printf 'entrypoint: compose file that started this container, or set `user:` to skip the drop entirely.\n' >&2 + exit 1 +fi + +# Preserve the supplementary groups Compose granted through group_add - that is +# how the Docker socket stays reachable - while discarding root's own group. +supplementary=$(id -G | tr ' ' '\n' | grep -vx 0 | paste -sd, -) +[ -n "$supplementary" ] || supplementary="$PGID" + +# Writable as the account the server is about to become? A real probe, run as +# exactly the identity the final exec below produces (PUID, PGID, the same +# supplementary groups, capabilities dropped), rather than a comparison of +# owners: ownership is not writability. A group-writable tree owned by another +# account, an ACL, or a CIFS/NFS mount that reports some unrelated uid are all +# fine to run on and would all fail an owner check. +writable_as_runtime() { + setpriv --reuid "$PUID" --regid "$PGID" --groups "$supplementary" test -w "$1" 2>/dev/null +} + for target in "${HOME:-}" "${CODEMAN_CASES_PATH:-}"; do [ -n "$target" ] && [ -d "$target" ] || continue owner=$(stat -c '%u:%g' "$target") [ "$owner" = "${PUID}:${PGID}" ] && continue - # Only ever correct a directory the DAEMON created (root-owned, because + # Only ever correct a directory the DAEMON created: root-owned, because # neither PUID nor PGID existed yet when it materialised the missing bind - # source). Anything else - a host tree that legitimately belongs to some + # source. Anything else - a host tree that legitimately belongs to some # OTHER account, such as an existing CODEMAN_CASES_PATH the README already # allows pointing at a normal project directory - is not this container's # to reassign; recursively chowning it on every mismatch silently rewrote # a credentials tree or a projects directory to PUID:PGID with one log - # line to explain it. Refuse instead, the same way Start-Codeman.sh already - # refuses to touch a root-owned appdata directory it did not expect. - if [ "${owner%%:*}" != '0' ]; then - printf 'entrypoint: %s is owned by %s, which is neither root nor PUID:PGID (%s:%s).\n' \ - "$target" "$owner" "$PUID" "$PGID" >&2 - printf 'entrypoint: refusing to change ownership of a directory this container did not create.\n' >&2 - printf 'entrypoint: either chown it on the host, or set PUID/PGID to match its current owner.\n' >&2 - exit 1 + # line to explain it. Such a directory is left alone and only PROBED below. + # + # The chown is deliberately not fatal. A bind mount backed by NFS, CIFS or a + # rootless daemon can refuse chown while still being perfectly writable, and + # the probe below is what decides whether the server can run on it. + if [ "${owner%%:*}" = '0' ]; then + if chown -R "${PUID}:${PGID}" "$target" 2>/dev/null; then + printf 'entrypoint: corrected ownership of %s to %s:%s\n' "$target" "$PUID" "$PGID" + else + printf 'entrypoint: warning: cannot change ownership of %s to %s:%s; checking whether it is writable anyway\n' \ + "$target" "$PUID" "$PGID" >&2 + fi fi - # Deliberately not fatal for a root-owned directory. A bind mount backed by - # NFS, CIFS or a rootless daemon can refuse chown while still being - # perfectly writable, and those deployments must keep working. A warning is - # more useful than a container that will not start. - if chown -R "${PUID}:${PGID}" "$target" 2>/dev/null; then - printf 'entrypoint: corrected ownership of %s to %s:%s\n' "$target" "$PUID" "$PGID" - else - printf 'entrypoint: warning: cannot change ownership of %s to %s:%s\n' \ - "$target" "$PUID" "$PGID" >&2 - printf 'entrypoint: warning: continuing; set the ownership on the host if startup fails\n' >&2 + if writable_as_runtime "$target"; then + if [ "${owner%%:*}" != '0' ]; then + printf 'entrypoint: %s is owned by %s, not %s:%s, but is writable as the runtime account; leaving its ownership alone\n' \ + "$target" "$owner" "$PUID" "$PGID" + fi + continue fi + + printf 'entrypoint: %s is not writable as PUID:PGID (%s:%s); it is owned by %s.\n' \ + "$target" "$PUID" "$PGID" "$owner" >&2 + printf 'entrypoint: refusing to change ownership of a directory this container did not create.\n' >&2 + printf 'entrypoint: either chown it on the host, make it writable to %s:%s, or set PUID/PGID to match its owner.\n' \ + "$PUID" "$PGID" >&2 + exit 1 done # /opt/codeman-cli (the four agent CLIs) is chowned to PUID:PGID once, at @@ -82,15 +147,19 @@ if [ -d /opt/codeman-cli ] && [ "$(stat -c '%u:%g' /opt/codeman-cli)" != "${PUID chown -R "${PUID}:${PGID}" /opt/codeman-cli fi -# Preserve the supplementary groups Compose granted through group_add - that is -# how the Docker socket stays reachable - while discarding root's own group. -supplementary=$(id -G | tr ' ' '\n' | grep -vx 0 | paste -sd, -) -[ -n "$supplementary" ] || supplementary="$PGID" +# Discarding group 0 is right for root's own group, but it also discards a +# `group_add: 0` that was there to reach a Docker socket owned by root:root. +# The previous image ran as PUID with that group kept, so say so rather than +# letting Docker-case support vanish silently on such a host. +if [ -S /var/run/docker.sock ] && [ "$(stat -c '%g' /var/run/docker.sock)" = '0' ]; then + printf 'entrypoint: warning: /var/run/docker.sock is owned by group 0, which is dropped along with root;\n' >&2 + printf 'entrypoint: warning: Docker cases will not work from this container. Give the socket a dedicated\n' >&2 + printf 'entrypoint: warning: group on the host and set DOCKER_SOCKET_GID to it.\n' >&2 +fi -# --bounding-set -all: with the reuid/regid drop above, CapPrm/CapEff are -# already empty, but the bounding set otherwise still lists everything -# cap_add granted (visible as a nonzero CapBnd even post-drop). no-new-privileges -# already makes that moot - nothing can regain a capability outside the -# bounding set - but clearing it too is free and matches what "drops to -# PUID:PGID" actually promises. -exec setpriv --reuid "$PUID" --regid "$PGID" --groups "$supplementary" --bounding-set -all "$@" +# No `--bounding-set -all` here: it is a silent no-op without CAP_SETPCAP, which +# the compose file deliberately does not grant, and `no-new-privileges` already +# makes the bounding set moot. The reuid/regid drop leaves CapPrm/CapEff empty. +# The image's full PATH goes back to the server here; see the top of the file. +exec setpriv --reuid "$PUID" --regid "$PGID" --groups "$supplementary" \ + env PATH="$runtime_path" "$@" diff --git a/docker/server.Dockerfile b/docker/server.Dockerfile index 27ec4a43..7fff784d 100644 --- a/docker/server.Dockerfile +++ b/docker/server.Dockerfile @@ -100,8 +100,15 @@ COPY --from=docker:29-cli \ # # Bump these deliberately, in a release. `--no-cache` is still needed to rebuild # this layer when only the pins change upstream. +# The prefix is APPENDED to PATH, never prepended: it is chowned to the runtime +# account below, and entrypoint.sh runs as root calling stat/chown/setpriv by +# bare name. A prefix ahead of /usr/bin would let a session drop a `setpriv` +# there and have it run as root at the next container start (measured with a +# minimal image of this exact shape). The four CLIs live only in this prefix, +# so they still resolve; entrypoint.sh additionally pins its own PATH to the +# system directories for the root part of the start. ENV NPM_CONFIG_PREFIX=/opt/codeman-cli -ENV PATH=/opt/codeman-cli/bin:$PATH +ENV PATH=$PATH:/opt/codeman-cli/bin RUN npm install --global \ @anthropic-ai/claude-code@2.1.258 \ @google/gemini-cli@0.58.0 \