#!/bin/bash # ============================================================================= # CodeRaft Platform — One-line installer # Usage: curl -fsSL https://install.coderaft.io | bash # # Installs the CodeRaft Dashboard. The dashboard handles everything else: # • License activation # • Product deployment (WolfGuard, Ravenscan, RedFox) # • Configuration & updates # ============================================================================= set -e # ── F-037 (2026-07-31): installer self-integrity check (SHA256) ───────────── # Goal: detect a script tampered with in transit or at rest between # publication and execution (e.g. a MITM proxy, a compromised CDN edge, or a # stale/corrupted cache rewriting the response to # `curl https://install.coderaft.io/install.sh`) before a single command # from this file runs. # # Mechanism: recompute this script's own SHA256 — excluding the # EXPECTED_SHA256 line itself, to avoid the chicken-and-egg problem of a # file needing to embed a hash of its own content — and compare it against # the digest published at install.sh.sha256, a sidecar file served # alongside this one but generated/updated independently (see # scripts/generate-install-checksums.sh, run at every release cut). # EXPECTED_SHA256 below is a secondary, embedded offline pin: a frozen # snapshot of the hash taken at the same time as the last regeneration, # used only as a fallback when the network fetch of the published digest # fails outright (fully offline environment, endpoint down). Regenerate # both together with scripts/generate-install-checksums.sh whenever this # file changes — never edit either by hand. # # KNOWN LIMITATION (a structural bash limit, not a bug in this check): when # this file is piped straight into an interpreter — # `curl -fsSL https://install.coderaft.io/install.sh | bash`, this file's # own documented usage at the top of this header — "$0" is literally the # string "bash", not a path to this script's bytes, and stdin has already # been consumed by bash parsing the script by the time this code runs. A # script cannot read "itself" back out of a pipe once its interpreter has # started consuming it. Under that invocation (and under process # substitution, e.g. `bash <(curl ...)`) the check below degrades to a # clear warning instead of silently pretending to protect the user. It IS # fully effective for the safer download-then-run flow: # curl -fsSL https://install.coderaft.io/install.sh -o install.sh # bash install.sh # self-checks against install.sh.sha256 # See deploy/docs/installer-integrity-verification.md for the full # writeup, threat model, and the (separate, not-yet-scheduled) work needed # to make the public `install.coderaft.io` endpoint actually serve this # monorepo's install.sh/install.sh.sha256 at all — today it still proxies # the legacy `coderaft-installer` repo. EXPECTED_SHA256="7a0118d2d3ea8f42ef70a7a14022ae7c7555ef1276c284462e69e8bf51d95b0a" # CODERAFT_INSTALL_SHA256_URL is overridable purely so this mechanism can be # tested end-to-end against a throwaway local HTTP server instead of the # live production endpoint (see deploy/docs/installer-integrity-verification.md, # "Testing" section). Real installs never need to set it. CODERAFT_INSTALL_SHA256_URL="${CODERAFT_INSTALL_SHA256_URL:-https://install.coderaft.io/install.sh.sha256}" _coderaft_sha256_stdin() { if command -v sha256sum &>/dev/null; then sha256sum | awk '{print $1}' elif command -v shasum &>/dev/null; then shasum -a 256 | awk '{print $1}' elif command -v openssl &>/dev/null; then openssl dgst -sha256 | awk '{print $NF}' else return 1 fi } _coderaft_self_verify() { if [ -n "${SKIP_SELF_VERIFY:-}" ]; then echo " ⚠ SKIP_SELF_VERIFY set — installer integrity check bypassed (dev only)." >&2 return 0 fi # "$0" only points to this script's real bytes when it was invoked as a # file (bash install.sh / sh install.sh / ./install.sh). Under a raw # pipe or process substitution it is "bash"/"sh"/a drained fd — not a # regular, re-readable file. See the KNOWN LIMITATION note above. if [ ! -f "$0" ] || [ ! -r "$0" ]; then echo "" >&2 echo " ⚠ Installer integrity check skipped: this script has no readable file" >&2 echo " path to hash (running from a pipe — see deploy/docs/" >&2 echo " installer-integrity-verification.md). For a verified install:" >&2 echo "" >&2 echo " curl -fsSL https://install.coderaft.io/install.sh -o install.sh" >&2 echo " bash install.sh" >&2 echo "" >&2 return 0 fi local actual actual="$(sed '/^EXPECTED_SHA256=/d' "$0" | _coderaft_sha256_stdin)" || actual="" if [ -z "$actual" ]; then echo " ⚠ Installer integrity check skipped: no sha256sum/shasum/openssl available (or hashing failed) to check this script." >&2 return 0 fi local published="" published="$(curl -fsSL --max-time 10 "$CODERAFT_INSTALL_SHA256_URL" 2>/dev/null | tr -d '[:space:]')" || published="" local expected="$published" local source="install.sh.sha256 (network)" if [ -z "$expected" ]; then if [ -n "$EXPECTED_SHA256" ]; then expected="$EXPECTED_SHA256" source="embedded offline pin" echo " ⚠ Could not reach ${CODERAFT_INSTALL_SHA256_URL} — falling back to the embedded offline pin." >&2 else echo " ⚠ Could not reach ${CODERAFT_INSTALL_SHA256_URL} and no embedded pin is set — skipping integrity check." >&2 return 0 fi fi if [ "$actual" != "$expected" ]; then echo "" >&2 echo " FATAL: installer integrity check failed" >&2 echo " computed (local) : ${actual}" >&2 echo " expected (${source}): ${expected}" >&2 echo "" >&2 echo " This script's content does not match its published digest — it may" >&2 echo " have been tampered with in transit or at rest. Aborting." >&2 echo " Set SKIP_SELF_VERIFY=1 to bypass (development only)." >&2 echo "" >&2 exit 1 fi echo " ✓ Installer integrity verified (SHA256 matches ${source})" } _coderaft_self_verify INSTALL_DIR="${INSTALL_DIR:-./coderaft}" echo "" echo " ╔══════════════════════════════════════════╗" echo " ║ CodeRaft Platform — Installer ║" echo " ║ Security. Identity. Access. Unified. ║" echo " ╚══════════════════════════════════════════╝" echo "" # ── OS detection ───────────────────────────────────────────────────────────── # Coderaft itself runs in Docker on every OS. The native capture daemon # (live packet inspection for Ravenscan) is the exception: on Docker # Desktop (macOS/Windows) containers cannot see the host's real NICs, # so we install a native binary on the host instead. On Linux the # Docker sidecar with network_mode: host works natively — no extra # binary needed. case "$(uname -s)" in Darwin) CODERAFT_OS="macos" ; CODERAFT_NEEDS_NATIVE_CAPTURE=1 ;; Linux) CODERAFT_OS="linux" ; CODERAFT_NEEDS_NATIVE_CAPTURE=0 ;; *) CODERAFT_OS="unknown" ; CODERAFT_NEEDS_NATIVE_CAPTURE=0 ;; esac case "$(uname -m)" in arm64|aarch64) CODERAFT_ARCH="arm64" ;; x86_64|amd64) CODERAFT_ARCH="amd64" ;; *) CODERAFT_ARCH="unknown" ;; esac echo " Detected: ${CODERAFT_OS}/${CODERAFT_ARCH}" echo "" # B33 (2026-06-01): NE PAS forcer DOCKER_DEFAULT_PLATFORM. # Forcer "linux/arm64" (bare) cassait les pulls d'images publiques qui # n'exposent que linux/arm64/v8 (postgres:16-alpine, redis:7-alpine, # caddy:2-alpine, etc.) → "no matching manifest". Docker Desktop ≥ 4.20 # résout correctement par lui-même. Si un user a un Docker très ancien, il # peut toujours forcer en exportant DOCKER_DEFAULT_PLATFORM=linux/arm64/v8 # avant de lancer l'install. # ── Prerequisites ──────────────────────────────────────────────────────────── check_command() { if ! command -v "$1" &> /dev/null; then echo " ✗ $1 is required but not installed." exit 1 fi echo " ✓ $1 found" } echo " Checking prerequisites..." check_command docker if ! docker compose version &> /dev/null; then echo " ✗ Docker Compose v2 is required." echo " https://docs.docker.com/compose/install/" exit 1 fi echo " ✓ docker compose found" # B-DAEMON-CHECK (2026-06-10): `docker compose version` only validates the # CLI plugin — the daemon may still be unreachable. `docker info` round-trips # through the daemon so we fail fast with a clear message instead of pulling # blindly and producing cryptic errors halfway through the install. if ! _docker_info=$(docker info --format '{{.ServerVersion}}' 2>&1); then echo " ✗ Docker daemon is not reachable." echo "" echo " docker info: ${_docker_info}" echo "" echo " Start Docker / Docker Desktop and wait for it to be running," echo " then re-run this installer." exit 1 fi echo " ✓ Docker daemon reachable (server ${_docker_info})" echo "" # ── Install ────────────────────────────────────────────────────────────────── echo " Installing to: ${INSTALL_DIR}" mkdir -p "${INSTALL_DIR}" cd "${INSTALL_DIR}" # ── age key setup (SOPS Phase 2 secrets management) ───────────────────────── # We keep TWO paths: # * AGE_KEY_PATH = /etc/coderaft/age.key (system-wide, root-owned, legacy) # * AGE_KEY_LOCAL = ${INSTALL_DIR}/.coderaft-age.key (bind-mount source for # dashboard-api → /keys/age.key) # The compose bind-mount uses AGE_KEY_LOCAL so it works on macOS / Windows # Docker Desktop without giving the container root access to /etc/coderaft. # When AGE_KEY_PATH already exists (legacy install), we copy it to AGE_KEY_LOCAL # so the dashboard-api can decrypt .env.enc without changing convention. AGE_KEY_DIR="/etc/coderaft" AGE_KEY_PATH="${AGE_KEY_DIR}/age.key" AGE_KEY_LOCAL="$(pwd)/.coderaft-age.key" ensure_age_binary() { if command -v age-keygen &>/dev/null; then return 0 fi echo " Downloading age-keygen..." AGE_VERSION="v1.2.1" AGE_OS="${CODERAFT_OS/macos/darwin}" AGE_TMP="$(mktemp -d)" trap 'rm -rf "$AGE_TMP"' EXIT # --max-time 60: mirrors install.ps1's Invoke-WebRequest -TimeoutSec 60 for # this same age release download. curl -fsSL --max-time 60 "https://github.com/FiloSottile/age/releases/download/${AGE_VERSION}/age-${AGE_VERSION}-${AGE_OS}-${CODERAFT_ARCH}.tar.gz" \ -o "${AGE_TMP}/age.tar.gz" 2>/dev/null || return 1 tar -xzf "${AGE_TMP}/age.tar.gz" -C "${AGE_TMP}" 2>/dev/null sudo install -m 755 "${AGE_TMP}/age/age-keygen" /usr/local/bin/age-keygen 2>/dev/null \ || install -m 755 "${AGE_TMP}/age/age-keygen" "$HOME/.local/bin/age-keygen" 2>/dev/null \ || return 1 sudo install -m 755 "${AGE_TMP}/age/age" /usr/local/bin/age 2>/dev/null \ || install -m 755 "${AGE_TMP}/age/age" "$HOME/.local/bin/age" 2>/dev/null \ || true echo " ✓ age installed" return 0 } setup_age_key() { # 1. If a legacy /etc/coderaft/age.key exists and the local one doesn't, # mirror it so the dashboard-api bind-mount works without sudo. if [ -f "${AGE_KEY_PATH}" ] && [ ! -f "${AGE_KEY_LOCAL}" ]; then echo " ✓ Reusing legacy age key from ${AGE_KEY_PATH}" sudo cat "${AGE_KEY_PATH}" 2>/dev/null > "${AGE_KEY_LOCAL}" \ || cat "${AGE_KEY_PATH}" 2>/dev/null > "${AGE_KEY_LOCAL}" \ || { echo " ⚠ Could not read ${AGE_KEY_PATH} — generating a new local key."; rm -f "${AGE_KEY_LOCAL}"; } if [ -s "${AGE_KEY_LOCAL}" ]; then chmod 400 "${AGE_KEY_LOCAL}" return 0 fi fi if [ -f "${AGE_KEY_LOCAL}" ]; then echo " ✓ age key already exists at ${AGE_KEY_LOCAL}" return 0 fi if ! ensure_age_binary; then echo " ⚠ Could not install age — SOPS encryption deferred to migrate-to-sops.sh" return 1 fi echo " Generating age key at ${AGE_KEY_LOCAL}..." age-keygen -o "${AGE_KEY_LOCAL}" 2>/dev/null || return 1 chmod 400 "${AGE_KEY_LOCAL}" echo " ✓ age key generated (${AGE_KEY_LOCAL})" echo "" echo " IMPORTANT: Back up ${AGE_KEY_LOCAL} to an encrypted USB or secure vault." echo " If this key is lost, all encrypted .env.enc secrets are unrecoverable." echo "" return 0 } # Try to set up age key on every OS. Failure is non-fatal: the dashboard # falls back to plaintext .env with a loud warning, and the operator can # run scripts/migrate-to-sops.sh later. setup_age_key || true # Generate secrets on first install gen_hex() { openssl rand -hex "$1" 2>/dev/null || head -c "$1" /dev/urandom | od -An -tx1 | tr -d ' \n'; } # ── LAN IP auto-detection (RELAY_ADVERTISE_HOST — FalconOne Remote Assist) ── # See the matching PowerShell parity comment in deploy/install.ps1's # Get-LanIPAddress for the full root-cause writeup (commit 471f0b7 + # dashboard-api/server.js's "falconone-relay" service block). Linux/macOS # parity for the RELAY_ADVERTISE_HOST cascade further down this file. # # `ip route get` performs a route LOOKUP only (no packet ever sent) and # reports the source IP the kernel would use to reach a public address — # i.e. the LAN-facing IP of whichever interface carries the default route. # This naturally skips docker0/veth/br-* bridges, which are never on the # default route. macOS has no `ip` by default (no iproute2), so `route get` # resolves the outbound interface first, then `ipconfig getifaddr` reads # its IPv4 address. get_lan_ip() { local ip="" case "$(uname -s)" in Linux) if command -v ip >/dev/null 2>&1; then ip=$(ip route get 1.1.1.1 2>/dev/null | sed -n 's/.* src \([0-9.]*\).*/\1/p' | head -1) fi if [ -z "$ip" ] && command -v hostname >/dev/null 2>&1; then # Fallback for minimal images without iproute2. ip=$(hostname -I 2>/dev/null | awk '{print $1}') fi ;; Darwin) local iface iface=$(route -n get 1.1.1.1 2>/dev/null | awk '/interface:/{print $2}') if [ -n "$iface" ]; then ip=$(ipconfig getifaddr "$iface" 2>/dev/null) fi ;; esac printf '%s' "$ip" } ABSOLUTE_INSTALL_DIR="$(pwd)" # ── install-config.env: install-time / public config, NEVER encrypted ────── # Task #150 (2026-07-31): HOST_PROJECT_DIR / CODERAFT_HOST_OS / CODERAFT_HOST_ARCH # are not secrets — they used to be written straight into .env, which then got # swept into .env.enc by SOPS below, conflating install-time topology with real # credentials (root cause referenced literally as #150 in dashboard-api's own # comments). They now live in their own plaintext file, mode 600, that is # NEVER passed to `sops --encrypt`. docker-compose interpolation reads it via # `--env-file install-config.env --env-file .env` (see every `docker compose` # invocation below and in start.sh/stop.sh/update.sh). upsert_install_config() { local key="$1" val="$2" if [ -f install-config.env ] && grep -q "^${key}=" install-config.env 2>/dev/null; then grep -v "^${key}=" install-config.env > install-config.env.tmp \ && printf '%s=%s\n' "$key" "$val" >> install-config.env.tmp \ && mv install-config.env.tmp install-config.env else printf '%s=%s\n' "$key" "$val" >> install-config.env fi # Defense-in-depth: `> file.tmp && mv` creates file.tmp under the current # umask, not the original mode — belt-and-braces even though every call # site today already re-asserts chmod 600 right after. chmod 600 install-config.env 2>/dev/null || true } # Task #148 (2026-07-31): postgres migrates to Docker's native `secrets:` + # POSTGRES_PASSWORD_FILE (the official postgres image already supports this) # instead of a plaintext POSTGRES_PASSWORD= env var baked into Config.Env. # Unlike `.env` (read client-side by the `docker compose` CLI process running # INSIDE dashboard-api's own container — proven safe on a tmpfs, see the # dashboard-api service's `tmpfs:` mount below), a compose `secrets: file:` # source IS resolved by the actual Docker DAEMON as a literal bind-mount # source (confirmed experimentally 2026-07-31: a secrets file living only on # a container-internal tmpfs made `docker compose up` fail on the daemon side # with "bind source path does not exist") — so this file MUST live on real, # persistent, host-visible disk, same tree as docker-compose.yml itself. # Written here (mirroring the .env bootstrap right below) so it already # exists before the very first `docker compose up` — postgres's `secrets:` # block would otherwise fail outright on a truly fresh install. write_postgres_secret_file() { local pg_pw="$1" mkdir -p secrets chmod 700 secrets printf '%s' "$pg_pw" > secrets/postgres_password chmod 600 secrets/postgres_password } # Task #219 (2026-07-31, logical follow-up of #148 Phase 3): redis has no # native `_FILE` env var convention like postgres's image does, but its # entrypoint is a real shell (redis:7-alpine, unlike the distroless vault) — # so `--requirepass` is read from this file via a `sh -c` command override # instead (see the redis service block below). Same host-visible-disk # constraint as postgres's secrets file: Compose's non-swarm `secrets: file:` # is resolved by the Docker DAEMON itself. write_redis_secret_file() { local redis_pw="$1" mkdir -p secrets chmod 700 secrets printf '%s' "$redis_pw" > secrets/redis_password chmod 600 secrets/redis_password } if [ -f ".env" ] && grep -q '^POSTGRES_PASSWORD=' .env 2>/dev/null; then # Existing install. Always (re)write HOST_PROJECT_DIR with the current # install dir — the location may have changed since the previous install, # and a stale or missing value breaks docker-compose interpolation # (warning + empty bind-mount path → dashboard-api cannot reach .env.enc # → fake "first run"). touch install-config.env upsert_install_config HOST_PROJECT_DIR "${ABSOLUTE_INSTALL_DIR}" # Backfill CODERAFT_HOST_OS/ARCH from install-config.env if already there # (re-run of the new installer), else from .env (upgrade from a version # that still wrote them there), else from the OS/arch detected above. if grep -q '^CODERAFT_HOST_OS=' install-config.env 2>/dev/null; then : elif grep -q '^CODERAFT_HOST_OS=' .env 2>/dev/null; then upsert_install_config CODERAFT_HOST_OS "$(grep '^CODERAFT_HOST_OS=' .env | head -1 | cut -d= -f2-)" upsert_install_config CODERAFT_HOST_ARCH "$(grep '^CODERAFT_HOST_ARCH=' .env | head -1 | cut -d= -f2- || echo "${CODERAFT_ARCH}")" else upsert_install_config CODERAFT_HOST_OS "${CODERAFT_OS}" upsert_install_config CODERAFT_HOST_ARCH "${CODERAFT_ARCH}" fi # Strip any legacy copies from .env now that install-config.env is authoritative. if grep -qE '^(HOST_PROJECT_DIR|CODERAFT_HOST_OS|CODERAFT_HOST_ARCH)=' .env 2>/dev/null; then grep -vE '^(HOST_PROJECT_DIR|CODERAFT_HOST_OS|CODERAFT_HOST_ARCH)=' .env > .env.tmp \ && mv .env.tmp .env fi chmod 600 .env install-config.env # Task #148: upgrade from a pre-#148 install never had secrets/postgres_password — # backfill it from the EXISTING .env value so postgres's `secrets:` block # (added by this version) resolves to the SAME password postgres already # has, instead of a fresh one that would mismatch the running cluster. if [ ! -f secrets/postgres_password ]; then write_postgres_secret_file "$(grep '^POSTGRES_PASSWORD=' .env | head -1 | cut -d= -f2-)" echo " ✓ secrets/postgres_password backfilled from existing .env (task #148)" fi # Task #219: same backfill, redis. if [ ! -f secrets/redis_password ] && grep -q '^REDIS_PASSWORD=' .env 2>/dev/null; then write_redis_secret_file "$(grep '^REDIS_PASSWORD=' .env | head -1 | cut -d= -f2-)" echo " ✓ secrets/redis_password backfilled from existing .env (task #219)" fi # Backward compat: legacy install without .env.enc — show warning in dashboard if [ ! -f "${AGE_KEY_LOCAL}" ] && [ ! -f "${AGE_KEY_PATH}" ]; then echo " ⚠ Legacy install detected: no age key found." echo " Secrets are currently stored as plaintext in .env." echo " Run the Setup Wizard to migrate to encrypted .env.enc." fi echo " ✓ Existing config preserved" else echo " Generating secrets..." PG_PASSWORD_BOOTSTRAP="$(gen_hex 24)" REDIS_PASSWORD_BOOTSTRAP="$(gen_hex 24)" cat > .env << ENVFILE # CodeRaft Dashboard — $(date -u +"%Y-%m-%d") POSTGRES_PASSWORD=${PG_PASSWORD_BOOTSTRAP} REDIS_PASSWORD=${REDIS_PASSWORD_BOOTSTRAP} DASHBOARD_SECRET=$(gen_hex 32) RAVENSCAN_CAPTURE_TOKEN=$(gen_hex 32) ENVFILE chmod 600 .env # Task #148: same value as .env's POSTGRES_PASSWORD above, materialized # as a standalone file for postgres's `secrets:`/POSTGRES_PASSWORD_FILE — # both must agree at the container's very first initdb. write_postgres_secret_file "${PG_PASSWORD_BOOTSTRAP}" unset PG_PASSWORD_BOOTSTRAP # Task #219: same idea, redis. write_redis_secret_file "${REDIS_PASSWORD_BOOTSTRAP}" unset REDIS_PASSWORD_BOOTSTRAP cat > install-config.env << CONFIGFILE # CodeRaft — install-time / public config (NOT a secret, NEVER SOPS-encrypted) # $(date -u +"%Y-%m-%d") HOST_PROJECT_DIR=${ABSOLUTE_INSTALL_DIR} CODERAFT_HOST_OS=${CODERAFT_OS} CODERAFT_HOST_ARCH=${CODERAFT_ARCH} CONFIGFILE chmod 600 install-config.env echo " ✓ Secrets generated" echo " ✓ install-config.env generated" fi # ── RELAY_ADVERTISE_HOST auto-detection (FalconOne Remote Assist relay) ──── # See get_lan_ip's header comment above for the full root-cause writeup. # Cascade, resolved once and stored in install-config.env (same home as # HOST_PROJECT_DIR/CODERAFT_HOST_OS/CODERAFT_HOST_ARCH — never .env, this is # public config, never a secret): # 0) A value already migrated/written to install-config.env on a previous # run — including a manual correction — is NEVER touched again. # 1) A legacy RELAY_ADVERTISE_HOST left in .env by a manual edit is # migrated as-is, then stripped from .env — install-config.env is # authoritative. # 2) CODERAFT_HOSTNAME configured in .env (Setup Wizard TLS step) and not # just the internal LAN-only default "coderaft.local" → reuse it, the # operator already told the platform its externally-reachable name. # 3) Otherwise, auto-detect this machine's own LAN IP (interface carrying # the default route) — correct for the common case. # NOTE: this auto-detection can be wrong on a multi-homed machine or a # deployment that WAN-port-forwards 21116-21117 to a different address than # the one detected here. In that case, a value set by hand directly in # install-config.env (RELAY_ADVERTISE_HOST=:21117) ALWAYS wins over the # cascade above — see step 0 — but this is an escape hatch, never a # prerequisite for a normal deployment to work. if ! grep -q '^RELAY_ADVERTISE_HOST=' install-config.env 2>/dev/null; then LEGACY_RELAY_HOST="" if [ -f .env ] && grep -qE '^RELAY_ADVERTISE_HOST=' .env 2>/dev/null; then LEGACY_RELAY_HOST="$(grep -E '^RELAY_ADVERTISE_HOST=' .env | tail -1 | cut -d= -f2- | tr -d '"' | tr -d "'" | xargs)" fi if [ -n "$LEGACY_RELAY_HOST" ]; then upsert_install_config RELAY_ADVERTISE_HOST "$LEGACY_RELAY_HOST" grep -vE '^RELAY_ADVERTISE_HOST=' .env > .env.tmp && mv .env.tmp .env chmod 600 .env echo " ✓ RELAY_ADVERTISE_HOST migré .env → install-config.env ($LEGACY_RELAY_HOST)" else RELAY_HOSTNAME_VAL="" if [ -f .env ] && grep -qE '^CODERAFT_HOSTNAME=' .env 2>/dev/null; then RELAY_HOSTNAME_VAL="$(grep -E '^CODERAFT_HOSTNAME=' .env | tail -1 | cut -d= -f2- | tr -d '"' | tr -d "'" | xargs)" fi if [ -n "$RELAY_HOSTNAME_VAL" ] && [ "$RELAY_HOSTNAME_VAL" != "coderaft.local" ] && [ "$RELAY_HOSTNAME_VAL" != "localhost" ]; then upsert_install_config RELAY_ADVERTISE_HOST "${RELAY_HOSTNAME_VAL}:21117" echo " ✓ RELAY_ADVERTISE_HOST auto-détecté : ${RELAY_HOSTNAME_VAL}:21117 (CODERAFT_HOSTNAME configuré — nom externe défini via le Setup Wizard)" else RELAY_LAN_IP="$(get_lan_ip)" if [ -n "$RELAY_LAN_IP" ]; then upsert_install_config RELAY_ADVERTISE_HOST "${RELAY_LAN_IP}:21117" echo " ✓ RELAY_ADVERTISE_HOST auto-détecté : ${RELAY_LAN_IP}:21117 (IP LAN auto-détectée de cette machine — aucun CODERAFT_HOSTNAME externe configuré)" else echo " ⚠ RELAY_ADVERTISE_HOST n'a pas pu être auto-détecté (aucune IP LAN trouvée) — FalconOne Remote Assist restera injoignable tant qu'une valeur n'est pas ajoutée manuellement à install-config.env (RELAY_ADVERTISE_HOST=:21117)." fi fi fi fi # Every `docker compose` invocation from here on must read BOTH files — # install-config.env (plaintext config) first, then .env (secrets) so a # stray key collision (should never happen, see #150 var lists) resolves in # favor of the secrets file. Order has no practical effect today since the # two files are disjoint by construction. COMPOSE_ENV_ARGS=(--env-file install-config.env --env-file .env) # ── Encrypt .env → .env.enc and purge plaintext (banking-grade) ───────────── # Runs on first install OR on a re-install where plaintext is still around. The # dashboard-api still needs a plaintext .env at `docker compose up` time for # variable interpolation, so we keep .env present here and let the dashboard # decrypt .env.enc on subsequent boots. The migrate-to-sops.sh --finalize step # is what eventually purges plaintext on running deployments. For NEW installs # we can be more aggressive: write .env.enc immediately so readHostEnv() never # touches plaintext after the first `docker compose up` settles. encrypt_env_to_enc() { local age_key="" if [ -f "${AGE_KEY_LOCAL}" ]; then age_key="${AGE_KEY_LOCAL}" elif [ -f "${AGE_KEY_PATH}" ]; then # Mirror legacy key into install dir so the bind-mount works. sudo cat "${AGE_KEY_PATH}" 2>/dev/null > "${AGE_KEY_LOCAL}" \ || cat "${AGE_KEY_PATH}" 2>/dev/null > "${AGE_KEY_LOCAL}" \ || return 1 chmod 400 "${AGE_KEY_LOCAL}" age_key="${AGE_KEY_LOCAL}" else return 1 fi if ! command -v sops &>/dev/null; then # Try to install sops on the fly; non-fatal. SOPS_VERSION="v3.8.1" SOPS_OS="${CODERAFT_OS/macos/darwin}" # --max-time 60: no direct install.ps1 equivalent (the Windows install # path doesn't fetch a standalone sops binary), so by analogy with the # sibling age release download above (same file, same GitHub-release # binary category). curl -fsSL --max-time 60 "https://github.com/getsops/sops/releases/download/${SOPS_VERSION}/sops-${SOPS_VERSION}.${SOPS_OS}.${CODERAFT_ARCH}" \ -o /tmp/sops-coderaft 2>/dev/null || return 1 sudo install -m 755 /tmp/sops-coderaft /usr/local/bin/sops 2>/dev/null \ || install -m 755 /tmp/sops-coderaft "$HOME/.local/bin/sops" 2>/dev/null \ || return 1 rm -f /tmp/sops-coderaft fi local age_pub age_pub=$(grep "# public key:" "${age_key}" 2>/dev/null | head -1 | awk '{print $NF}') [ -n "${age_pub}" ] || return 1 SOPS_AGE_KEY_FILE="${age_key}" sops --encrypt --age "${age_pub}" --output .env.enc .env \ || return 1 chmod 600 .env.enc echo " ✓ .env.enc created (sops + age)" return 0 } if [ -f .env ] && [ ! -f .env.enc ]; then if encrypt_env_to_enc; then echo " ✓ Secrets encrypted to .env.enc" # On a fresh install we keep .env (compose interpolation needs it for # the very first `up`). The dashboard-api re-encrypts on every deploy # and will refuse to read plaintext at runtime when CODERAFT_REJECT_PLAINTEXT_ENV=1. # Operators can run `bash scripts/migrate-to-sops.sh --finalize` after # the first successful boot to purge plaintext for good. echo "" echo " ⚠ Plaintext .env still on disk (needed by 'docker compose up' for" echo " variable interpolation). Run this AFTER the dashboard boots OK:" echo " curl -fsSL https://install.coderaft.io/migrate.sh | bash -s -- --finalize" echo "" else echo " ⚠ Could not encrypt .env to .env.enc (age/sops unavailable)." echo " Plaintext .env will be used as fallback." echo " Run scripts/migrate-to-sops.sh later to fix." fi fi # Read the capture token back so we can pass it to the native daemon # install step (only relevant on macOS). RAVENSCAN_CAPTURE_TOKEN_VALUE="$(grep '^RAVENSCAN_CAPTURE_TOKEN=' .env | cut -d= -f2)" # ── Vault master-key bootstrap (D2 + D3) ──────────────────────────────────── # Runs once on fresh install. Skipped if vault-keys/age.key already exists. # The vault age key is SEPARATE from the SOPS age key (.coderaft-age.key). # .coderaft-age.key = SOPS legacy path (kept for backward compat per Phase 0.5) # vault-keys/age.key = master key that encrypts the vault's envelope DEK vault_bootstrap() { mkdir -p vault-keys vault-tls vault-config # ── Step 1: Generate vault age key (or reuse existing) ────────────────── # B-VAULT-BOOT (2026-06-09): le return 0 anticipé skipait aussi TLS PKI + # config.yaml quand age.key existait déjà → vault container fail healthcheck # car /etc/coderaft-vault/config.yaml manquant. Skip uniquement la génération # de la key et toujours continuer vers TLS + config (ces étapes sont # idempotentes via leurs propres "skip si déjà existant" en interne). if [ -f vault-keys/age.key ]; then echo " ✓ Vault age key already exists — skipping key generation" vault_bootstrap_tls return 0 fi if ! ensure_age_binary; then echo " ✗ Cannot generate vault age key — age-keygen not available" echo " Install age from https://github.com/FiloSottile/age/releases" echo " then re-run the installer." exit 1 fi # B-VAULT-DEK (2026-06-22): wipe a stale coderaft_vault_data volume left # by a prior install attempt. Its wrapped DEK was sealed with the previous # age key; a fresh age.key here will produce "unseal failed: bad master # key" on the next /v1/unseal call. age.key didn't exist yet so there's # no legitimate data to preserve. if docker volume inspect coderaft_vault_data >/dev/null 2>&1; then echo " Removing stale coderaft_vault_data volume from a previous install attempt..." if docker volume rm -f coderaft_vault_data >/dev/null 2>&1; then echo " ✓ stale vault data removed" else echo " ⚠ could not remove coderaft_vault_data — vault unseal may fail with 'bad master key'" echo " Manual fix: docker compose down; docker volume rm coderaft_vault_data; docker compose up -d" fi fi echo " Generating vault master key..." age-keygen -o vault-keys/age.key 2>/dev/null || { echo " ✗ age-keygen failed" exit 1 } chmod 400 vault-keys/age.key # ── Step 2: Generate mTLS PKI (D3) ────────────────────────────────────── # AUDIT-SECU-2026-08-04 (Vault H1 follow-up): this used to be "Step 2: # Compute BIP39 recovery phrase" / "Step 3: Display recovery phrase with # big warning" — a `docker run ... -mnemonic-from-key` call that NEVER # existed as a real coderaft-vault sub-command (confirmed against # cmd/coderaft-vault/main.go: the binary only accepts -config and # -health-check), always silently fell through to a raw key-fingerprint # placeholder, and displayed that under a banner claiming "This 24-word # phrase is the ONLY way to recover your vault" — factually false even # before this fix (the fingerprint isn't a recovery mechanism at all) and # doubly so now: coderaft-vault no longer reads vault-keys/age.key for # ANYTHING (keyprovider.NewAgeMasterKeyProvider() is stateless — see # coderaft-vault/internal/keyprovider). Removed entirely rather than # patched, per the "KNOWN FOLLOW-UP" this file already flagged in # vault_check_seal_state() above. vault-keys/age.key itself is still # generated (see Step 1) purely as this function's own "already # bootstrapped" idempotency marker — update.sh/vault_seed_bootstrap_secrets # key off its presence — but it is not, and was never really, a vault # recovery secret. # # The REAL recovery mechanism is the Shamir ceremony (POST /v1/init once # the vault container is actually running, POST /v1/unseal with a # threshold of the returned shares) — it cannot run yet at this point in # the script (the vault container doesn't exist until later). Once it is # up, vault_check_seal_state() below prints the exact commands and this # is also documented in deploy/docs/vault.md ("Init + unseal ceremony"). echo "" echo " ℹ Vault master key bootstrap complete (vault-keys/age.key)." echo " This is NOT a vault recovery secret — coderaft-vault generates" echo " its own master key internally and splits it into real Shamir" echo " shares the first time an operator runs the init ceremony" echo " (POST /v1/init) against the running container. This script will" echo " print the exact commands once the vault is up. WRITE DOWN and" echo " separately distribute every share when that happens — it is" echo " the ONLY way to recover the vault; there is no other back door." echo "" vault_bootstrap_tls } vault_bootstrap_tls() { # B12 fix: correct cert filenames: vault.crt / vault.key / client-ca.crt # B11 fix: correct ACL field names: name, cert_san, permissions (NOT san/role/allow) # B7 fix: server cert SAN includes localhost + 127.0.0.1 for mTLS hostname verification # Prefer host openssl if available; otherwise use an alpine container (no host dep). if command -v openssl &>/dev/null; then _vault_bootstrap_tls_host else _vault_bootstrap_tls_alpine fi } _vault_bootstrap_tls_host() { # Skip if already generated (re-run safety) if [ -f vault-tls/client-ca.crt ] && [ -f vault-tls/vault.crt ]; then echo " ✓ Vault mTLS PKI already exists — skipping cert generation" return 0 fi echo " Generating vault mTLS PKI (host openssl)..." chmod 700 vault-tls # CA (client-ca.crt — not ca.crt, the vault config.yaml expects this exact name) openssl req -x509 -newkey rsa:4096 -days 3650 -nodes -sha256 \ -keyout vault-tls/client-ca.key \ -out vault-tls/client-ca.crt \ -subj "/CN=coderaft-vault-ca" \ -addext "basicConstraints=critical,CA:TRUE" \ 2>/dev/null chmod 600 vault-tls/client-ca.key vault-tls/client-ca.crt # Server cert — SAN must include coderaft-vault, localhost, 127.0.0.1 # so the curlimages/curl sidecar can verify --cacert client-ca.crt against # https://coderaft-vault:8200 (hostname check fails without the DNS SAN). openssl req -newkey rsa:2048 -nodes -sha256 \ -keyout vault-tls/vault.key \ -out vault-tls/vault.csr \ -subj "/CN=coderaft-vault" \ 2>/dev/null openssl x509 -req -days 3650 -sha256 \ -in vault-tls/vault.csr \ -CA vault-tls/client-ca.crt -CAkey vault-tls/client-ca.key -CAcreateserial \ -out vault-tls/vault.crt \ -extfile <(printf "subjectAltName=DNS:coderaft-vault,DNS:localhost,IP:127.0.0.1\nbasicConstraints=CA:FALSE") \ 2>/dev/null rm -f vault-tls/vault.csr chmod 600 vault-tls/vault.crt vault-tls/vault.key # Per-product client certs _vault_client_cert "dashboard-api" "dashboard-api.coderaft.local" _vault_client_cert "entraguard" "entraguard.coderaft.local" _vault_client_cert "ravenscan" "ravenscan.coderaft.local" _vault_client_cert "redfox" "redfox.coderaft.local" _vault_client_cert "falconone" "falconone.coderaft.local" _vault_client_cert "cve-proxy" "cve-proxy.coderaft.local" _vault_write_config } _vault_bootstrap_tls_alpine() { # Fallback: run openssl inside an alpine one-shot container (no host dep). # Mirrors the approach in update.ps1 4c.2. if [ -f vault-tls/client-ca.crt ] && [ -f vault-tls/vault.crt ]; then echo " ✓ Vault mTLS PKI already exists — skipping cert generation" return 0 fi echo " Generating vault mTLS PKI (alpine container — host openssl absent)..." chmod 700 vault-tls local openssl_script openssl_script=$(cat <<'SCRIPT' set -e apk add --no-cache openssl >/dev/null cd /work openssl req -x509 -newkey rsa:4096 -days 3650 -nodes -sha256 \ -keyout client-ca.key -out client-ca.crt \ -subj "/CN=coderaft-vault-ca" \ -addext "basicConstraints=critical,CA:TRUE" 2>/dev/null openssl req -newkey rsa:2048 -nodes -sha256 \ -keyout vault.key -out vault.csr \ -subj "/CN=coderaft-vault" 2>/dev/null cat > /tmp/server.ext </dev/null rm -f vault.csr /tmp/server.ext for pair in "dashboard-api:dashboard-api.coderaft.local" \ "entraguard:entraguard.coderaft.local" \ "ravenscan:ravenscan.coderaft.local" \ "redfox:redfox.coderaft.local" \ "falconone:falconone.coderaft.local" \ "cve-proxy:cve-proxy.coderaft.local"; do name="${pair%%:*}" san="${pair##*:}" openssl req -newkey rsa:2048 -nodes -sha256 \ -keyout "${name}-client.key" -out "${name}-client.csr" \ -subj "/CN=${san}" 2>/dev/null cat > /tmp/client.ext </dev/null rm -f "${name}-client.csr" /tmp/client.ext done chmod 600 *.key 2>/dev/null || true # falconone-api / coderaft-cve-proxy nonroot fix (distroless uid 65532). chmod 644 falconone-client.key falconone-client.crt 2>/dev/null || true chmod 644 cve-proxy-client.key cve-proxy-client.crt 2>/dev/null || true SCRIPT ) local abs_tls_dir abs_tls_dir="$(cd vault-tls && pwd)" echo "$openssl_script" | docker run --rm -i \ -v "${abs_tls_dir}:/work" \ alpine:3.20 sh 2>&1 if [ ! -f vault-tls/vault.crt ]; then echo " ✗ Alpine openssl cert generation failed" >&2 exit 1 fi _vault_write_config } _vault_write_config() { # vault config.yaml — references /tls/vault.crt (not server.crt) and /tls/client-ca.crt (not ca.crt) cat > vault-config/config.yaml << 'CFGEOF' server: addr: "0.0.0.0:8200" tls_cert: "/tls/vault.crt" tls_key: "/tls/vault.key" client_ca: "/tls/client-ca.crt" storage: path: "/data/vault.db" keys: age_key_path: "/keys/age.key" audit: log_path: "/data/audit.log" acl_path: "/etc/coderaft-vault/acl.yaml" CFGEOF chmod 600 vault-config/config.yaml # B11 fix: correct ACL field names: name, cert_san, permissions cat > vault-config/acl.yaml << 'ACLEOF' # coderaft-vault ACL — controls which client cert SAN can access which secrets. # Field names: name, cert_san, permissions (NOT san/role/allow). clients: - name: dashboard-api cert_san: "dashboard-api.coderaft.local" permissions: ["*"] - name: entraguard cert_san: "entraguard.coderaft.local" permissions: ["read:azure_*","read:license_key","read:entraguard_*","read:platform/identity/oidc","read:platform/identity/graph-tools","write:platform/identity/graph-tools","read:credentials/*","read:tenant/*","read:m365dsc-exo-cert/*","write:m365dsc-exo-cert/*","delete:m365dsc-exo-cert/*"] - name: ravenscan cert_san: "ravenscan.coderaft.local" permissions: ["read:ravenscan_*","read:neo4j_*","read:license_key","read:platform/identity/oidc"] - name: redfox cert_san: "redfox.coderaft.local" permissions: ["read:redfox_*","read:license_key","read:platform/identity/oidc","read:platform/identity/graph-tools","read:redfox/connections/*","write:redfox/connections/*","delete:redfox/connections/*","read:redfox/k8s/*","write:redfox/k8s/*","delete:redfox/k8s/*"] - name: falconone cert_san: "falconone.coderaft.local" permissions: ["read:license_key","read:falconone_*","read:platform/identity/oidc","read:platform/identity/graph-tools","sign:falconone_agent_cert","read:falconone/nvd_api_key","read:falconone/audit_hmac_key","write:falconone/audit_hmac_key","read:falconone/pki/agents-ca/cert","read:pki/falconone-agents-ca*","write:pki/falconone-agents-ca*","read:falconone/scripts_ca*","write:falconone/scripts_ca*"] - name: cve-proxy cert_san: "cve-proxy.coderaft.local" permissions: ["read:cve-proxy/*", "write:cve-proxy/*"] ACLEOF chmod 600 vault-config/acl.yaml echo " ✓ Vault mTLS PKI generated" echo " CA: vault-tls/client-ca.crt" echo " Server: vault-tls/vault.{crt,key}" echo " Clients: vault-tls/{dashboard-api,entraguard,ravenscan,redfox,falconone,cve-proxy}-client.{crt,key}" } _vault_client_cert() { local name="$1" san="$2" openssl req -newkey rsa:2048 -nodes -sha256 \ -keyout "vault-tls/${name}-client.key" \ -out "vault-tls/${name}-client.csr" \ -subj "/CN=${san}" \ 2>/dev/null openssl x509 -req -days 3650 -sha256 \ -in "vault-tls/${name}-client.csr" \ -CA vault-tls/client-ca.crt -CAkey vault-tls/client-ca.key -CAcreateserial \ -out "vault-tls/${name}-client.crt" \ -extfile <(printf "subjectAltName=DNS:%s\nbasicConstraints=CA:FALSE" "$san") \ 2>/dev/null rm -f "vault-tls/${name}-client.csr" if [ "$name" = "falconone" ] || [ "$name" = "cve-proxy" ]; then # falconone-api / coderaft-cve-proxy run distroless nonroot # (uid 65532) — 600 would be unreadable. See update.sh for the same fix. chmod 644 "vault-tls/${name}-client.crt" "vault-tls/${name}-client.key" else chmod 600 "vault-tls/${name}-client.crt" "vault-tls/${name}-client.key" fi } # ── FalconOne agents mTLS PKI (#170, #174, #226) ───────────────────────────── # Historically (#170) this function ALSO generated a private, self-signed # CA + server leaf under falconone-tls/ as the :8443 agent listener's trust # chain. Task #174 moved that trust chain to the Coderaft Vault instead # (buildAgentTLSFromVault, cmd/falconone-api/main.go: fetches the Vault's # falconone-agents-ca and mints the :8443 server leaf from that SAME CA) # whenever Vault is reachable — the normal case — so this installer-generated # file was never actually part of a successful mTLS handshake once #174 # shipped; main.go's preferred path never reads it. # # Worse, its mere presence on disk caused a real incident: task #223 found # the deployment-bundle builder (internal/experience/bundle.go) was still # unconditionally pinning THIS file's CA into every new agent's # server-ca.pem while the :8443 listener actually presented a Vault-signed # leaf — fresh enrolment failed with "x509: certificate signed by unknown # authority", live-reported by an operator 2026-08-11 (root-caused + fixed # in bundle.go, commit 3f60303). The installer CA looks legitimate (it's a # real, validly-formed 10-year CA) but corresponds to nothing any agent # actually trusts — exactly the kind of on-disk artifact that keeps # resurfacing as a source of confusion in any code path that falls back to # reading it. # # #226 fix: stop generating it, here and in install.ps1 / update.sh / # update.ps1. main.go's buildAgentTLS keeps its Path B fallback (reads # cfg.AgentServerCrt/Key/AgentsCA from disk when Vault is unreachable at # boot) UNCHANGED — that stays available for an operator who deliberately # places override material at those paths — but it is no longer fed # automatically by this installer. Net effect: if Vault is down at first # boot and nobody has manually provided override files, falconone-api now # fails closed (refuses to start rather than serve a TLS trust chain that # matches no real agent) instead of degrading silently. See buildAgentTLS's # own doc comment in main.go for that tradeoff. # # Any file left over from a pre-#226 install/update is backed up (not just # deleted — recoverable, same convention as the acl.yaml self-heals below) # and removed here too, so upgrading an EXISTING deployment also closes the # hole. This is safe: as explained above, that file was never what a # successfully-registered agent actually trusts once Vault has ever been # reachable, so removing it cannot break a live, working handshake — it can # only change what happens on a FUTURE boot where Vault is unreachable, from # "silently wrong trust" to "fail closed" (or "trust the manual override", # if one is present). _falconone_tls_bootstrap() { local install_dir="${1:?install_dir required}" local fo_tls_dir="${install_dir}/falconone-tls" mkdir -p "$fo_tls_dir" chmod 755 "$fo_tls_dir" local f ts ts="$(date +%Y%m%d%H%M%S)" for f in agents-ca.crt agents-ca.key agents-ca.srl server.crt server.key; do if [ -f "${fo_tls_dir}/${f}" ]; then cp "${fo_tls_dir}/${f}" "${fo_tls_dir}/${f}.bak-${ts}" rm -f "${fo_tls_dir}/${f}" _rotate_backups "${fo_tls_dir}/${f}" fi done echo " ✓ FalconOne agent TLS: no installer-generated PKI — trust sourced from Vault at boot (#226)" } # ── Backup rotation (security hardening, 2026-07-31) ───────────────────────── # Every self-heal path below does `cp X X.bak-` before touching X # (acl.yaml, and in update.sh also docker-compose.yml/override.yml). On a # deployment that runs unattended for months/years across many install/update # runs, these accumulate without bound. Keep only the $2 most recent (default # 5) backups sharing $1's basename; delete anything older. Safe/idempotent: # a no-op when there are $2 or fewer. _rotate_backups() { local base_path="$1" keep="${2:-5}" local dir base dir="$(dirname "$base_path")" base="$(basename "$base_path")" ls -t "${dir}/${base}".bak-* 2>/dev/null | tail -n "+$((keep + 1))" | while IFS= read -r f; do rm -f -- "$f" done } # ── ACL self-heal: falconone entry/permissions (#172) ──────────────────────── # The static acl.yaml written by _vault_write_config only runs once (its # caller, vault_bootstrap_tls, returns early when vault-tls already exists — # see the "already exists — skipping" guards above). That means any install # that provisioned vault before this fix shipped will never get the # falconone entry/permissions rewritten. This self-heal is additive-only, # idempotent, and safe to call on every install/update run: if the # "falconone" client entry is missing, it appends the full canonical block; # if present, it appends only whichever required permissions are missing, # leaving everything else in the file untouched. Backs up acl.yaml before # any modification. _falconone_acl_selfheal() { local acl_path="$1" if [ ! -f "$acl_path" ]; then echo " [install] ACL self-heal: $acl_path not found — skipping (vault not provisioned yet)" return 0 fi local required_perms=( "read:license_key" "read:falconone_*" "read:platform/identity/oidc" "read:platform/identity/graph-tools" "sign:falconone_agent_cert" "read:falconone/nvd_api_key" "read:falconone/audit_hmac_key" "write:falconone/audit_hmac_key" "read:falconone/pki/agents-ca/cert" "read:pki/falconone-agents-ca*" "write:pki/falconone-agents-ca*" "read:falconone/scripts_ca*" "write:falconone/scripts_ca*" ) local ts ts="$(date -u +"%Y%m%dT%H%M%SZ")" if ! grep -qE '^[[:space:]]*-[[:space:]]*name:[[:space:]]*falconone[[:space:]]*$' "$acl_path"; then cp "$acl_path" "${acl_path}.bak-${ts}" _rotate_backups "$acl_path" cat >> "$acl_path" <<'FALCONONEACL' - name: falconone cert_san: "falconone.coderaft.local" permissions: - "read:license_key" - "read:falconone_*" - "read:platform/identity/oidc" - "read:platform/identity/graph-tools" - "sign:falconone_agent_cert" - "read:falconone/nvd_api_key" - "read:falconone/audit_hmac_key" - "write:falconone/audit_hmac_key" - "read:falconone/pki/agents-ca/cert" - "read:pki/falconone-agents-ca*" - "write:pki/falconone-agents-ca*" - "read:falconone/scripts_ca*" - "write:falconone/scripts_ca*" FALCONONEACL echo " [install] Self-heal ACL: falconone permissions updated (+${#required_perms[@]} added, entry created)" return 0 fi local start_line end_line start_line=$(grep -nE '^[[:space:]]*-[[:space:]]*name:[[:space:]]*falconone[[:space:]]*$' "$acl_path" | head -1 | cut -d: -f1) end_line=$(awk -v s="$start_line" 'NR>s && /^[[:space:]]*-[[:space:]]*name:/{print NR; exit}' "$acl_path") if [ -z "$end_line" ]; then end_line=$(( $(wc -l < "$acl_path") + 1 )) fi local block block=$(sed -n "${start_line},$((end_line - 1))p" "$acl_path") local missing=() local p for p in "${required_perms[@]}"; do if ! grep -qF "\"${p}\"" <<< "$block"; then missing+=("$p") fi done if [ "${#missing[@]}" -eq 0 ]; then echo " [install] ACL falconone already up-to-date" return 0 fi cp "$acl_path" "${acl_path}.bak-${ts}" _rotate_backups "$acl_path" if grep -qE '^[[:space:]]*permissions:[[:space:]]*\[.*\][[:space:]]*$' <<< "$block"; then local additions="" for p in "${missing[@]}"; do additions="${additions},\"${p}\""; done awk -v s="$start_line" -v e="$end_line" -v add="$additions" ' NR>=s && NR "${acl_path}.tmp" && mv "${acl_path}.tmp" "$acl_path" else local addition_block="" for p in "${missing[@]}"; do addition_block="${addition_block} - \"${p}\""$'\n'; done local insert_line=$(( end_line - 1 )) awk -v ins="$insert_line" -v add="$addition_block" ' { print } NR==ins { printf "%s", add } ' "$acl_path" > "${acl_path}.tmp" && mv "${acl_path}.tmp" "$acl_path" fi echo " [install] Self-heal ACL: falconone permissions updated (+${#missing[@]} added)" } # ── ACL self-heal: mantisstrike entry/permissions (Phase 1 platform wiring, # same pattern as _falconone_acl_selfheal above) ───────────────────────────── # The static acl.yaml written by _vault_write_config predates MantisStrike # and has no `mantisstrike` entry — this self-heal is additive-only, # idempotent, and safe on every install/update run (fresh installs included: # it runs unconditionally right after vault_bootstrap, same as falconone's # own self-heal). Permissions mirror the CURRENT real # coderaft-vault/configs/acl.yaml (verified directly against that file, # 2026-08-04) — deliberately NOT copying _vault_write_config's stale # embedded template above, which still includes the since-removed (2026-07- # 31) `read:license_key` grant. _mantisstrike_acl_selfheal() { local acl_path="$1" if [ ! -f "$acl_path" ]; then echo " [install] ACL self-heal: $acl_path not found — skipping (vault not provisioned yet)" return 0 fi local required_perms=( "read:mantisstrike_*" "read:platform/identity/oidc" "read:platform/identity/graph-tools" ) local ts ts="$(date -u +"%Y%m%dT%H%M%SZ")" if ! grep -qE '^[[:space:]]*-[[:space:]]*name:[[:space:]]*mantisstrike[[:space:]]*$' "$acl_path"; then cp "$acl_path" "${acl_path}.bak-${ts}" _rotate_backups "$acl_path" cat >> "$acl_path" <<'MANTISSTRIKEACL' - name: mantisstrike cert_san: "mantisstrike.coderaft.local" permissions: - "read:mantisstrike_*" - "read:platform/identity/oidc" - "read:platform/identity/graph-tools" MANTISSTRIKEACL echo " [install] Self-heal ACL: mantisstrike permissions updated (+${#required_perms[@]} added, entry created)" return 0 fi local start_line end_line start_line=$(grep -nE '^[[:space:]]*-[[:space:]]*name:[[:space:]]*mantisstrike[[:space:]]*$' "$acl_path" | head -1 | cut -d: -f1) end_line=$(awk -v s="$start_line" 'NR>s && /^[[:space:]]*-[[:space:]]*name:/{print NR; exit}' "$acl_path") if [ -z "$end_line" ]; then end_line=$(( $(wc -l < "$acl_path") + 1 )) fi local block block=$(sed -n "${start_line},$((end_line - 1))p" "$acl_path") local missing=() local p for p in "${required_perms[@]}"; do if ! grep -qF "\"${p}\"" <<< "$block"; then missing+=("$p") fi done if [ "${#missing[@]}" -eq 0 ]; then echo " [install] ACL mantisstrike already up-to-date" return 0 fi cp "$acl_path" "${acl_path}.bak-${ts}" _rotate_backups "$acl_path" if grep -qE '^[[:space:]]*permissions:[[:space:]]*\[.*\][[:space:]]*$' <<< "$block"; then local additions="" for p in "${missing[@]}"; do additions="${additions},\"${p}\""; done awk -v s="$start_line" -v e="$end_line" -v add="$additions" ' NR>=s && NR "${acl_path}.tmp" && mv "${acl_path}.tmp" "$acl_path" else local addition_block="" for p in "${missing[@]}"; do addition_block="${addition_block} - \"${p}\""$'\n'; done local insert_line=$(( end_line - 1 )) awk -v ins="$insert_line" -v add="$addition_block" ' { print } NR==ins { printf "%s", add } ' "$acl_path" > "${acl_path}.tmp" && mv "${acl_path}.tmp" "$acl_path" fi echo " [install] Self-heal ACL: mantisstrike permissions updated (+${#missing[@]} added)" } # ── ACL self-heal: redfox connections/k8s vault-backed credentials ────────── # Zero-Knowledge Credential Architecture Palier 1 # (coderaft-platform/docs/redfox-zero-knowledge-scoping.md): target # connection credentials and k8s cluster auth data move from RedFox's own # Postgres into this vault, under redfox/connections/* and redfox/k8s/*. # Same additive-only, idempotent merge-into-existing-entry pattern as # _falconone_acl_selfheal above — any install provisioned before this ships # already has a "redfox" entry (it existed for platform/identity/oidc), so # without this it would never gain the new permissions. _redfox_acl_selfheal() { local acl_path="$1" if [ ! -f "$acl_path" ]; then echo " [install] ACL self-heal: $acl_path not found — skipping (vault not provisioned yet)" return 0 fi local required_perms=( "read:license_key" "read:redfox_*" "read:platform/identity/oidc" "read:platform/identity/graph-tools" "read:redfox/connections/*" "write:redfox/connections/*" "delete:redfox/connections/*" "read:redfox/k8s/*" "write:redfox/k8s/*" "delete:redfox/k8s/*" ) local ts ts="$(date -u +"%Y%m%dT%H%M%SZ")" if ! grep -qE '^[[:space:]]*-[[:space:]]*name:[[:space:]]*redfox[[:space:]]*$' "$acl_path"; then cp "$acl_path" "${acl_path}.bak-${ts}" _rotate_backups "$acl_path" cat >> "$acl_path" <<'REDFOXACL' - name: redfox cert_san: "redfox.coderaft.local" permissions: - "read:license_key" - "read:redfox_*" - "read:platform/identity/oidc" - "read:platform/identity/graph-tools" - "read:redfox/connections/*" - "write:redfox/connections/*" - "delete:redfox/connections/*" - "read:redfox/k8s/*" - "write:redfox/k8s/*" - "delete:redfox/k8s/*" REDFOXACL echo " [install] Self-heal ACL: redfox permissions updated (+${#required_perms[@]} added, entry created)" return 0 fi local start_line end_line start_line=$(grep -nE '^[[:space:]]*-[[:space:]]*name:[[:space:]]*redfox[[:space:]]*$' "$acl_path" | head -1 | cut -d: -f1) end_line=$(awk -v s="$start_line" 'NR>s && /^[[:space:]]*-[[:space:]]*name:/{print NR; exit}' "$acl_path") if [ -z "$end_line" ]; then end_line=$(( $(wc -l < "$acl_path") + 1 )) fi local block block=$(sed -n "${start_line},$((end_line - 1))p" "$acl_path") local missing=() local p for p in "${required_perms[@]}"; do if ! grep -qF "\"${p}\"" <<< "$block"; then missing+=("$p") fi done if [ "${#missing[@]}" -eq 0 ]; then echo " [install] ACL redfox already up-to-date" return 0 fi cp "$acl_path" "${acl_path}.bak-${ts}" _rotate_backups "$acl_path" if grep -qE '^[[:space:]]*permissions:[[:space:]]*\[.*\][[:space:]]*$' <<< "$block"; then local additions="" for p in "${missing[@]}"; do additions="${additions},\"${p}\""; done awk -v s="$start_line" -v e="$end_line" -v add="$additions" ' NR>=s && NR "${acl_path}.tmp" && mv "${acl_path}.tmp" "$acl_path" else local addition_block="" for p in "${missing[@]}"; do addition_block="${addition_block} - \"${p}\""$'\n'; done local insert_line=$(( end_line - 1 )) awk -v ins="$insert_line" -v add="$addition_block" ' { print } NR==ins { printf "%s", add } ' "$acl_path" > "${acl_path}.tmp" && mv "${acl_path}.tmp" "$acl_path" fi echo " [install] Self-heal ACL: redfox permissions updated (+${#missing[@]} added)" } # ── ACL self-heal: cve-proxy entry (coderaft-cve-engine sidecar) ──────────── # Same additive-only, idempotent pattern as _falconone_acl_selfheal above — # any install provisioned before this ships never gets the cve-proxy entry # otherwise, since vault_bootstrap_tls's static acl.yaml write only runs once. _cveproxy_acl_selfheal() { local acl_path="$1" if [ ! -f "$acl_path" ]; then echo " [install] ACL self-heal: $acl_path not found — skipping (vault not provisioned yet)" return 0 fi if grep -qE '^[[:space:]]*-[[:space:]]*name:[[:space:]]*cve-proxy[[:space:]]*$' "$acl_path"; then echo " [install] ACL cve-proxy already present" return 0 fi local ts ts="$(date -u +"%Y%m%dT%H%M%SZ")" cp "$acl_path" "${acl_path}.bak-${ts}" _rotate_backups "$acl_path" cat >> "$acl_path" <<'CVEPROXYACL' - name: cve-proxy cert_san: "cve-proxy.coderaft.local" permissions: - "read:cve-proxy/*" - "write:cve-proxy/*" CVEPROXYACL echo " [install] Self-heal ACL: cve-proxy entry created" } # ── Vault client cert self-heal (any product whose cert was never generated, # e.g. falconone/cve-proxy on installs provisioned before they shipped) ───── # Additive-only: does nothing to certs that already exist. Requires the CA # private key (vault-tls/client-ca.key) to still be present — deliberately # does NOT attempt a full CA rotation (that's a separate, disruptive # operation, not a self-heal). Silently no-ops with a clear message if the # CA key is gone, rather than failing the whole install/update. _vault_client_cert_selfheal() { local name="$1" san="$2" if [ -f "vault-tls/${name}-client.crt" ]; then return 0 fi if [ ! -f vault-tls/client-ca.key ] || [ ! -f vault-tls/client-ca.crt ]; then echo " [install] Cert self-heal: vault-tls/client-ca.key missing — cannot mint ${name}-client cert (needs a full CA rotation, not a self-heal)" return 0 fi echo " [install] Cert self-heal: generating vault-tls/${name}-client (was missing)" if command -v openssl &>/dev/null; then _vault_client_cert "$name" "$san" else local abs_tls_dir abs_tls_dir="$(cd vault-tls && pwd)" docker run --rm -i \ -v "${abs_tls_dir}:/work" \ alpine:3.20 sh -c " set -e apk add --no-cache openssl >/dev/null cd /work openssl req -newkey rsa:2048 -nodes -sha256 \ -keyout '${name}-client.key' -out '${name}-client.csr' \ -subj '/CN=${san}' 2>/dev/null printf 'subjectAltName=DNS:%s\nbasicConstraints=CA:FALSE' '${san}' > /tmp/client.ext openssl x509 -req -days 3650 -sha256 \ -in '${name}-client.csr' -CA client-ca.crt -CAkey client-ca.key -CAcreateserial \ -out '${name}-client.crt' -extfile /tmp/client.ext 2>/dev/null rm -f '${name}-client.csr' /tmp/client.ext if [ '${name}' = 'falconone' ] || [ '${name}' = 'cve-proxy' ]; then chmod 644 '${name}-client.key' '${name}-client.crt' else chmod 600 '${name}-client.key' '${name}-client.crt' fi " 2>&1 fi } vault_bootstrap # ── FalconOne mTLS PKI + ACL self-heal (#170 / #172 / #226) ────────────────── # Always run, independent of vault_bootstrap's internal "already exists — # skipping" guards, so a re-run of this installer on an existing install # still gets the legacy-PKI cleanup (#226) and any missing ACL permissions # healed. _falconone_tls_bootstrap "$PWD" _falconone_acl_selfheal "vault-config/acl.yaml" # ── MantisStrike vault client cert + ACL self-heal (Phase 1 platform wiring, # same pattern as falconone above) ─────────────────────────────────────────── _vault_client_cert_selfheal "mantisstrike" "mantisstrike.coderaft.local" _mantisstrike_acl_selfheal "vault-config/acl.yaml" # ── RedFox connections/k8s vault-backed credentials ACL self-heal ─────────── # Zero-Knowledge Credential Architecture Palier 1 — see _redfox_acl_selfheal # above. redfox's client cert already exists (provisioned for # platform/identity/oidc), so no _vault_client_cert_selfheal call is needed # here — only the ACL entry needs the new permissions. _redfox_acl_selfheal "vault-config/acl.yaml" # ── cve-proxy vault client cert + ACL self-heal ────────────────────────────── # coderaft-cve-proxy is a shared platform sidecar (in front of the central # coderaft-cve-engine), not tied to any single product license — self-healed # unconditionally, same as falconone above. _vault_client_cert_selfheal "falconone" "falconone.coderaft.local" _vault_client_cert_selfheal "cve-proxy" "cve-proxy.coderaft.local" _cveproxy_acl_selfheal "vault-config/acl.yaml" # Append CODERAFT_VAULT_* env vars if not already present _add_env_if_missing() { local key="$1" val="$2" if ! grep -q "^${key}=" .env 2>/dev/null; then printf '%s=%s\n' "$key" "$val" >> .env fi } _add_env_if_missing "CODERAFT_VAULT_URL" "https://coderaft-vault:8200" _add_env_if_missing "CODERAFT_VAULT_AZURE" "0" _add_env_if_missing "CODERAFT_VAULT_LICENSE" "0" _add_env_if_missing "CODERAFT_VAULT_PRODUCTS" "0" _add_env_if_missing "CODERAFT_VAULT_JWT" "0" # Init DB cat > init-db.sql << 'SQL' -- Product databases are created by the dashboard on demand SQL # Docker compose — dashboard + vault echo " Writing docker-compose.yml..." cat > docker-compose.yml << 'COMPOSE' # CodeRaft Dashboard # Products are deployed by the dashboard after license activation. # # F-017 (seccomp, re-verified 2026-07-31): every service below relies on # Docker's IMPLICIT default seccomp profile — deliberately NOT written as # `security_opt: [seccomp=default]`. That literal string is NOT a Docker # keyword; the daemon parses whatever follows `seccomp=` as a PATH to a # custom profile JSON file, so `seccomp=default` fails outright with # `opening seccomp profile (default) failed: open default: no such file or # directory` (re-confirmed live against this exact Docker version, # 2026-07-31 — a prior session hit and reverted the same mistake, this # re-check just confirms it's still true rather than trusting the old # note). Omitting `seccomp:` from `security_opt:` entirely is the correct # way to keep the default profile; it is already a strong default (blocks # ~44 dangerous syscalls including `mount`/`unshare`/`clone3` variants) and # no service here has demonstrated a need to loosen or further restrict it — # the one exception (`ravenscan-capture`, raw packet capture) explicitly # opts OUT via `seccomp:unconfined` for documented libpcap syscall needs, # see that service below. services: # Caddy HTTPS reverse proxy. # TLS mode is env-driven (Setup Wizard → dashboard-api writes .env and # recreates this service): # internal (default) — Caddy's own CA signs the cert (replaces mkcert, # 2026-07). CA root lives in the caddy_data volume: # PRESERVE that volume or agents lose trust. # wildcard — customer cert uploaded to ./caddy_certs/ # acme — Let's Encrypt (public deployments) caddy: image: caddy:2-alpine depends_on: dashboard: { condition: service_started } ports: # CADDY_BIND_ADDR is widened to 0.0.0.0 by the Exposure wizard step. - "${CADDY_BIND_ADDR:-127.0.0.1}:443:443" - "${CADDY_BIND_ADDR:-127.0.0.1}:80:80" environment: - CODERAFT_HOSTNAME=${CODERAFT_HOSTNAME:-coderaft.local} - CODERAFT_TLS_SITES=${CODERAFT_TLS_SITES:-coderaft.local, *.coderaft.local} - CADDY_TLS_MODE_ARGS=${CADDY_TLS_MODE_ARGS:-internal} volumes: - ./caddy_certs:/certs:ro - ./Caddyfile:/etc/caddy/Caddyfile:ro - caddy_data:/data - caddy_config:/config # B-HEALTH (2026-06-09): le healthcheck pointait vers l'admin API Caddy # (port 2019) qu'on désactive volontairement (`admin off` dans Caddyfile, # sécurité). On check le listener HTTP réel (:80) avec 127.0.0.1 pour # éviter le piège IPv6-first (localhost résout ::1 d'abord, nginx/caddy # écoute IPv4 → connection refused). Cf bug récurrent B15 côté Node. healthcheck: test: ["CMD", "wget", "--quiet", "--tries=1", "--spider", "http://127.0.0.1:80/"] interval: 30s timeout: 5s retries: 3 start_period: 10s security_opt: [no-new-privileges:true] # Phase 2 hardening (2026-07-31): verified live in a throwaway compose # project — caddy:2-alpine's default image runs as root and needs # exactly ONE capability back after `cap_drop: [ALL]` to keep binding # :80/:443 (both <1024): NET_BIND_SERVICE. No CHOWN/SETUID/SETGID needed # (unlike the `dashboard` nginx service below) — caddy doesn't fork a # privilege-dropped worker pool the way nginx's `user nginx;` directive # does. Confirmed a full boot + `curl` 200 still works with this exact # cap set. cap_drop: [ALL] cap_add: [NET_BIND_SERVICE] # `/data`/`/config` stay the existing named volumes above (cert/state # persistence — do NOT tmpfs them, see the big comment atop this service # about CA trust). `/tmp` is the only additional writable path caddy # needs under a read_only root; verified live (autosave.json still wrote # to /data, no other FS errors in the log). read_only: true tmpfs: [/tmp] mem_limit: 256m memswap_limit: 256m mem_reservation: 64m cpus: 1 restart: unless-stopped dashboard: image: ghcr.io/liamj74/coderaft-dashboard:latest ports: # Plain HTTP kept on 3000 (loopback only) for fallback when caddy is off # and for the dashboard-api healthchecks/internal helpers. - "127.0.0.1:3000:3000" depends_on: postgres: { condition: service_healthy } redis: { condition: service_healthy } dashboard-api: { condition: service_started } environment: - DATABASE_URL=postgres://coderaft:${POSTGRES_PASSWORD}@postgres:5432/coderaft - REDIS_URL=redis://:${REDIS_PASSWORD}@redis:6379/0 - LICENSE_SERVER_URL=https://license.coderaft.io # Security hardening (2026-07-31): DASHBOARD_SECRET removed from this # service's env — it was PURE DEAD CODE here. This "dashboard" container # is nginx serving a static Vite build (see repo root Dockerfile: # node build → nginx:1.27-alpine, no app server, no envsubst templating # of nginx.conf), it never reads any env var at runtime. Confirmed no # `process.env.DASHBOARD_SECRET` reference anywhere reachable from this # image before removing — dashboard-api (below) is the one that actually # resolves DASHBOARD_SECRET, and it does so from Coderaft Vault directly # (lib/session-auth.js), never from this env var either. Removing a # plaintext secret an image doesn't consume closes a `docker inspect` # exposure window at zero functional cost. # B-DASHBOARD-NET (2026-06-23): nginx inside this image proxies # /api/entraguard/, /api/ravenscan/, /api/redfox/ to the product # containers on coderaft-frontend / coderaft-backend. networks: [default, coderaft-frontend, coderaft-backend] security_opt: [no-new-privileges:true] # Phase 2 hardening (2026-07-31): verified live in a throwaway compose # project. This image's baked-in nginx.conf sets `user nginx;` — the # master process starts as root (this image sets no `USER`) and needs # CAP_CHOWN (to fix up `/var/cache/nginx/*_temp` ownership) + CAP_SETUID/ # CAP_SETGID (to drop the worker processes to the `nginx` account) before # it can serve anything; confirmed `cap_drop: [ALL]` alone fails startup # with `chown("/var/cache/nginx/client_temp", 101) failed (1: Operation # not permitted)`, and adding exactly these 3 caps back fixes it (curl # 200 afterwards). No NET_BIND_SERVICE needed — this image only listens # on :3000 internally (confirmed via its own conf.d/coderaft.conf), never # a privileged port. cap_drop: [ALL] cap_add: [CHOWN, SETUID, SETGID] read_only: true tmpfs: [/tmp, /var/cache/nginx, /var/run] healthcheck: test: ["CMD", "wget", "--quiet", "--tries=1", "--spider", "http://127.0.0.1:3000/"] interval: 30s timeout: 5s retries: 3 start_period: 10s mem_limit: 256m memswap_limit: 256m mem_reservation: 64m cpus: 1 restart: unless-stopped # F-003 (2026-06-21): ACL sidecar for the Docker socket. dashboard-api # talks through this instead of mounting /var/run/docker.sock directly, # so it cannot call POST /exec, /volumes, /secrets, /swarm. docker-proxy: image: tecnativa/docker-socket-proxy:0.3.0 container_name: coderaft-docker-proxy environment: CONTAINERS: 1 IMAGES: 1 NETWORKS: 1 SERVICES: 1 TASKS: 1 INFO: 1 VERSION: 1 POST: 1 # F-003 follow-up (2026-07-16): VOLUMES=1 required so `docker compose up` # can read/create named volumes for the product services (confirmed # 2026-07-31: removing VOLUMES=1 in a throwaway compose project makes # `docker compose up` fail outright with "denied" on POST # /volumes/create for every named volume — postgres_data, redis_data, # vault_data, dashboard_data, etc. — so this stays required). # # CORRECTED (2026-07-31, was inaccurate): the comment that used to sit # here claimed a privileged `Binds:[/:/host]` mount via POST # /containers/create was "still blocked" by this proxy. That is FALSE — # docker-socket-proxy (tecnativa/docker-socket-proxy) authorizes purely # by HTTP method + URL path (its ACL categories are things like # CONTAINERS/POST/VOLUMES), never by inspecting the JSON BODY of the # request. With POST=1 and CONTAINERS=1 already set (required for # `docker compose up` to create ANY container), a POST # /containers/create carrying `HostConfig.Binds:["/:/host"]` sails # through this proxy exactly like a normal container-create call — the # bind-mount privilege escalation is a property of what's IN the # request, which this proxy does not look at. # # Residual risk this creates: dashboard-api's OWN `/host-compose` mount # (see the dashboard-api service below) is a READ-WRITE bind of this # entire install directory, and dashboard-api legitimately WRITES # docker-compose.override.yml there itself (generateOverrideToDir(), # to add/remove product containers per license). If dashboard-api were # ever compromised (RCE in the Node process), the attacker already has # everything needed to add a malicious `Binds:["/:/host"]` (or worse) # to that override file and then trigger `docker compose up` through # this exact proxy — which would honor it, since POST/CONTAINERS/ # VOLUMES are already granted for the legitimate deploy/update/rollback # flows. This is NOT a new hole introduced by this proxy; it is an # inherent consequence of dashboard-api being the thing that actually # authors and executes compose plans for this platform — closing it # completely would require a full compose-plan validator (parse the # generated YAML, allow-list volume sources/binds before ever handing # it to `docker compose up`) that does not exist today and is # deliberately OUT OF SCOPE for this pass (real engineering effort, not # a config tweak — tracked as a follow-up, not silently accepted as # "fine"). What IS mitigated below: the broad `.:/host-compose` bind no # longer redundantly exposes vault-keys/vault-tls/.coderaft-age.key # (see the dashboard-api service's `tmpfs:` entries) — shrinking what # an RCE in dashboard-api can reach, without touching the # deploy/update/rollback-critical override-file write path itself. VOLUMES: 1 # F-003 follow-up 2 (2026-07-22): `docker compose pull` on a multi-arch # image (mandatory for every Coderaft product image) queries # GET /distribution/{name}/json to resolve the manifest for the local # platform BEFORE pulling. That endpoint is its own ACL category in # docker-socket-proxy — gated purely by GET, independent of POST — so # without it every per-product update failed with # "docker compose failed (code 1): denied" even though IMAGES=1 and # POST=1 were already set. Read-only manifest metadata, same exposure # class as IMAGES=1: no RCE surface, EXEC/SECRETS/SWARM/NODES stay 0. DISTRIBUTION: 1 EXEC: 0 SECRETS: 0 SWARM: 0 NODES: 0 ALLOW_START: 1 ALLOW_STOP: 1 ALLOW_RESTARTS: 1 volumes: - /var/run/docker.sock:/var/run/docker.sock:ro # F-014 (2026-07-31): a prior session's note ("read_only:true shadows # haproxy template") had given up on read_only for this service — its # docker-entrypoint.sh templates /usr/local/etc/haproxy/haproxy.cfg # from the ACL env vars above at every boot, `sed`-ing it from a # `haproxy.cfg.template` file baked into the SAME directory. A plain # `tmpfs:` mount there (always empty, unlike a named volume) wipes that # template out from under it — confirmed live: `sed: ... .template: No # such file or directory`. A NAMED VOLUME instead of tmpfs fixes it: # Docker copies the image's existing directory contents (including the # template) into an empty named volume on first mount, so the template # survives and the entrypoint can still write haproxy.cfg alongside it. # Verified live end-to-end: proxy started clean, and a real # `GET /version` through it returned 200 with genuine Docker Engine # version info. - docker_proxy_haproxy_cfg:/usr/local/etc/haproxy networks: - docker-proxy-net security_opt: [no-new-privileges:true] cap_drop: [ALL] cap_add: [CHOWN, SETGID, SETUID, NET_BIND_SERVICE] read_only: true tmpfs: [/tmp, /var/run] mem_limit: 128m memswap_limit: 128m mem_reservation: 32m cpus: 0.5 restart: unless-stopped dashboard-api: image: ghcr.io/liamj74/coderaft-dashboard-api:latest networks: - default - coderaft-vault-net - docker-proxy-net # B-BACKEND-NET (2026-07-16): postgres + redis live on the isolated # coderaft-backend network (see PR #12). Without joining this net, # dashboard-api resolves `postgres` to nothing and every DB call # throws `getaddrinfo EAI_AGAIN postgres`. - coderaft-backend # B-IPV6-KILL (2026-07-22): Docker Desktop on Windows/macOS gives the # container an IPv6 stack with no upstream route; some libraries still # `dns.lookup({family:6})` despite NODE_OPTIONS. Kill IPv6 at the sysctl # level so no lookup can even resolve to an AAAA — root-caused after # Entra callback surfaced ENETUNREACH on 2603:1027:*. sysctls: - net.ipv6.conf.all.disable_ipv6=1 - net.ipv6.conf.default.disable_ipv6=1 depends_on: postgres: { condition: service_healthy } redis: { condition: service_healthy } # B-VAULT-DEP (2026-06-09): coderaft-vault starts sealed and needs the # installer/update script to POST /v1/unseal before it can pass its # own healthcheck. With `service_healthy` here, compose would kill the # whole stack waiting for vault before the unseal step ever runs. # Use `service_started` so dashboard-api boots; it gracefully degrades # to "vault unavailable" until unseal completes, then reconnects. coderaft-vault: { condition: service_started } docker-proxy: { condition: service_started } environment: # F-003 (2026-06-21): talk to Docker via the scoped proxy, not the raw # socket. Blocks POST /exec, /volumes, /secrets, /swarm. - DOCKER_HOST=tcp://docker-proxy:2375 # B15 (2026-05-19): Node.js résout IPv6 d'abord par défaut. Le container # Docker n'a pas d'IPv6 → ENETUNREACH → fallback IPv4 lent ou timeout # sur les appels sortants (license.coderaft.io, login.microsoftonline.com). # Force IPv4-first. - NODE_OPTIONS=--dns-result-order=ipv4first - LICENSE_SERVER_URL=https://license.coderaft.io - DATABASE_URL=postgres://coderaft:${POSTGRES_PASSWORD}@postgres:5432/coderaft - REDIS_URL=redis://:${REDIS_PASSWORD}@redis:6379/0 # Security hardening (2026-07-31): DASHBOARD_SECRET removed — dead code # here too. dashboard-api's own process resolves it directly from # Coderaft Vault (lib/session-auth.js's resolveDashboardSecret(), with a # local-disk cache fallback), never from process.env.DASHBOARD_SECRET — # confirmed via a full grep of dashboard-api/ for real (non-comment) # reads before removing. The OTHER products (WolfGuard/Ravenscan/ # RedFox/FalconOne) that verify JWTs signed with this same secret get # it from a Docker `secrets:` file now (dashboard-api's own compose # generation, generateOverrideToDir() — see # for the full migration of the 7 remaining plaintext # secrets to files). - CONTAINER_COMPOSE_DIR=/host-compose - HOST_PROJECT_DIR=${HOST_PROJECT_DIR} - COMPOSE_PROJECT_NAME=coderaft # Banking-grade: dashboard-api reads .env.enc (sops+age). Plaintext .env # is refused at runtime when CODERAFT_REJECT_PLAINTEXT_ENV=1. # NOTE: Phase 0.5 keeps SOPS path for backward compat; Phase 5 removes it. - CODERAFT_REJECT_PLAINTEXT_ENV=1 - SOPS_AGE_KEY_FILE=/keys/age.key # Vault integration (Phase 0.5 — products read secrets from the vault) - CODERAFT_VAULT_URL=https://coderaft-vault:8200 # B12 fix: correct vault TLS filenames (client-ca.crt, not ca.crt) - CODERAFT_VAULT_CA=/vault-tls/client-ca.crt - CODERAFT_VAULT_CLIENT_CERT=/vault-tls/dashboard-api-client.crt - CODERAFT_VAULT_CLIENT_KEY=/vault-tls/dashboard-api-client.key volumes: # F-003: /var/run/docker.sock mount REMOVED — replaced by # tcp://docker-proxy:2375 over the docker-proxy-net network. - dashboard_data:/data - .:/host-compose # Age private key for SOPS decryption (legacy — kept for backward compat). # The installer creates the host file at .coderaft-age.key on first run. - ./.coderaft-age.key:/keys/age.key:ro # Vault mTLS client cert for dashboard-api (B12: correct filenames) - ./vault-tls/client-ca.crt:/vault-tls/client-ca.crt:ro - ./vault-tls/dashboard-api-client.crt:/vault-tls/dashboard-api-client.crt:ro - ./vault-tls/dashboard-api-client.key:/vault-tls/dashboard-api-client.key:ro # Security hardening (2026-07-31, docker-socket-proxy comment fix # follow-up): the broad `.:/host-compose` bind above gives dashboard-api # RW access to the ENTIRE install directory, redundantly including # vault-keys/ (unseal material), vault-tls/ (mTLS private keys) and # .coderaft-age.key (SOPS decryption key) — none of which # dashboard-api's own code ever reads via CONTAINER_COMPOSE_DIR/ # /host-compose (confirmed via a full grep of server.js: it only # touches docker-compose*.yml, install-config.env, .env.enc, secrets/, # nginx-tls.conf, certbot/ under that path — it reads the real secrets # via the narrow, purpose-specific RO mounts above instead). Blanking # these three paths inside THIS container's view shrinks what an RCE # in dashboard-api can reach, at zero functional cost. Bind-mounting # /dev/null over a FILE works to blank it (reads as empty); tmpfs only # works over DIRECTORIES (verified: tmpfs on a file path fails outright # with "not a directory" — that's why .coderaft-age.key uses the # /dev/null bind below instead of a tmpfs entry). Verified # experimentally 2026-07-31 in an isolated throwaway compose project: # `ls /host-compose` still lists everything (docker-compose.yml, # secrets/, harmless files) with normal RW, `ls /host-compose/vault-keys` # and `/vault-tls` come back empty, `cat /host-compose/.coderaft-age.key` # returns nothing — while deploy/restart/rollback/backup (which only # ever touch the OTHER paths under /host-compose) are unaffected. - /dev/null:/host-compose/.coderaft-age.key # Task #148 (banking-grade runtime exposure reduction, 2026-07-31): the # *working* .env (resolved secret VALUES, passed to `docker compose # --env-file`) is written here instead of the persistent bind-mounted # /host-compose (the `.:/host-compose` volume above). tmpfs is private to # THIS container — wiped on every restart/recreate — and does NOT need to # be host-daemon-visible: the `docker compose` CLI runs INSIDE this same # container (docker-cli-compose baked into the image, talking to the # daemon over tcp://docker-proxy:2375), and --env-file is interpolated # client-side by that CLI process, never resolved by the daemon itself. # Confirmed experimentally 2026-07-31 in an isolated throwaway compose # project (see #148 report). tmpfs: - /run/coderaft-env:size=1m,mode=0700,uid=0 # Shadow vault-keys/vault-tls inside the broad /host-compose bind — see # the long comment on the volumes: block above. - /host-compose/vault-keys:mode=0000,uid=0,size=1m - /host-compose/vault-tls:mode=0000,uid=0,size=1m # Phase 2 hardening (2026-07-31), added alongside `read_only: true` # below: # - /tmp: dashboard-api's own server.js stages 3 things here via # os.tmpdir() — the Ravenscan capture-host installer zip and the # GPO/Intune CA-push zip (both small, a few MB) and, more # importantly, the restic DISASTER-RECOVERY RESTORE scratch dir # (`backup.js`'s `restore()`), which could be multi-gigabyte for a # full platform restore. Rather than size a RAM-backed tmpfs to # cover a worst-case multi-GB restore (risking host OOM under # memory pressure instead of a clean ENOSPC), the two restore call # sites (POST /api/dashboard/backup/restore-platform and # /api/dashboard/backup/restore in server.js) were changed to # target a scratch dir under the disk-backed `/data` volume # instead of falling through to backup.js's own `/tmp` default — # so this tmpfs only ever needs to hold the two small zips. # 512M is generous headroom for those. # - /etc/coderaft: a SEPARATE, legacy local age-key generation path # (AGE_KEY_PATH, unrelated to the vault-tls mTLS certs above) # that self-heals with `mkdir -p /etc/coderaft && age-keygen` on # boot if missing — confirmed live this fails with `ENOENT: no such # file or directory, mkdir '/etc/coderaft'` under read_only without # this tmpfs (and confirmed it succeeds fine once added). This key # was already effectively ephemeral before this change too — no # volume mounts /etc/coderaft today, so every container recreation # already regenerated it from scratch even pre-read_only. # - restic's own metadata cache (`RESTIC_CACHE_DIR`, backup.js) was # ALSO redirected to `/data/.restic-cache` (disk-backed, not tmpfs) # for the same reason as the restore scratch dir above — verified # live that restic hard-fails ("unable to open cache: mkdir # /root/.cache: read-only file system") without this, since this # container runs as root with HOME=/root. - /tmp:size=512m - /etc/coderaft security_opt: [no-new-privileges:true] # F-011/F-014 (2026-07-31): verified live — a full boot (real /data + # /host-compose mounts, DASHBOARD_SECRET set) reached `listening on port # 3001` and GET /healthz → 200 unchanged with ALL caps dropped. No # cap_add needed (this image runs as root by design — Docker socket/CLI # access — but doesn't do any setuid/chown dance at its own startup). cap_drop: [ALL] read_only: true healthcheck: test: ["CMD", "curl", "-fsS", "http://127.0.0.1:3001/healthz"] interval: 30s timeout: 5s retries: 3 start_period: 20s mem_limit: 1g memswap_limit: 1g mem_reservation: 256m cpus: 2 restart: unless-stopped # ── coderaft-vault ────────────────────────────────────────────────────────── # Centralised secret store. All products read/write through it via mTLS. # Port 8200 is internal-only (coderaft-vault-net). No external exposure. # Phase 0.5: deployed from fresh install; existing installs migrate on update. coderaft-vault: image: ghcr.io/liamj74/coderaft-vault:latest # B8 fix: Run as root so the container can write to the Docker-managed # /data volume (SQLite + audit log) and read /tls/*.key files (mode 0600). # Effective security stays equivalent to nonroot because we drop ALL # capabilities AND set no-new-privileges. This is a standard hardening # pattern (root-with-no-caps), not a security regression. # # F-004 re-verified (2026-07-31, Phase 2 hardening pass): core/vault/ # Dockerfile's final stage IS already `gcr.io/distroless/static-debian12: # nonroot` with `USER nonroot:nonroot` (uid/gid 65532) — the image itself # defaults to non-root. This `user: "0:0"` override at the compose layer # is what actually forces it back to root at runtime, and that's # deliberate, not an oversight: `./vault-keys` (the age master key, # chmod 400) and `./vault-tls` (server/client certs+keys, chmod 600) are # HOST bind-mounts created by install.sh under whatever UID/permissions # invoked it (not necessarily root, not necessarily uid 65532 either) — # Docker bind-mounts pass host UID/GID through literally, with no # remapping unless userns-remap is enabled (F-038, separately deferred — # see STATUS.md, itself a stack-wide daemon change). Re-chowning those # host files to 65532 would need CAP_CHOWN on the INSTALLING user, which # a non-root `curl | bash` install does not have (chown to an arbitrary # UID is root-only on Linux) — so it can't be done generically across # every deployment shape this installer supports. Loosening the host # file permissions instead (group/world-readable) to let a non-root # container UID read them would weaken the actual host-level protection # of the vault's own master key, which is a worse trade than the current # one. This is the SAME conclusion two prior sessions reached for the # (now-migrated-away) legacy coderaft-vault repo — re-confirmed here # against the monorepo's actual Dockerfile + install.sh, not just # inherited from old notes. Left unchanged; `cap_drop: [ALL]` below is # the compensating control. user: "0:0" networks: - coderaft-vault-net volumes: - vault_data:/data - ./vault-keys:/keys:ro - ./vault-tls:/tls:ro - ./vault-config:/etc/coderaft-vault:ro healthcheck: test: ["CMD", "/coderaft-vault", "-health-check"] interval: 30s timeout: 5s retries: 3 start_period: 10s security_opt: [no-new-privileges:true] cap_drop: [ALL] mem_limit: 256m memswap_limit: 256m mem_reservation: 64m cpus: 1 restart: unless-stopped # ── coderaft-cve-proxy ─────────────────────────────────────────────────── # Internal sidecar in front of the shared coderaft-cve-engine # (cve.coderaft.io): holds the ONE bearer key for this deployment (read # from vault at boot) and forwards CVE/KEV/EPSS/MSRC lookups from any # product on coderaft-backend. No host port published — reachable only as # coderaft-cve-proxy:8092 inside the docker network. Not tied to any # product license (shared platform service, like coderaft-vault). coderaft-cve-proxy: image: ghcr.io/liamj74/coderaft-cve-proxy:latest networks: - coderaft-vault-net - coderaft-backend - coderaft-frontend depends_on: coderaft-vault: { condition: service_started } environment: - CODERAFT_VAULT_URL=https://coderaft-vault:8200 - CODERAFT_VAULT_CA=/vault-tls/client-ca.crt - CODERAFT_VAULT_CLIENT_CERT=/vault-tls/cve-proxy-client.crt - CODERAFT_VAULT_CLIENT_KEY=/vault-tls/cve-proxy-client.key # ACCEPTED RESIDUAL RISK (security hardening pass, 2026-07-31): every # OTHER consumer of XPRODUCT_INTERNAL_TOKEN/DASHBOARD_SECRET/etc. in # this deployment was migrated to a Docker-native `secrets:` file (see # dashboard-api/server.js's generateOverrideToDir() + the shell # entrypoint.sh loaders / Go envOrFile() helpers in each product repo). # coderaft-cve-proxy could NOT be migrated the same way: # 1. Its source lives in a separate, currently-unmigrated repo not # available in this monorepo — no code we can add native # `_FILE`/`_PATH` support to. # 2. It already already reads this token straight off the vault at # boot for its OWN bearer key ("holds the ONE bearer key for this # deployment", per the comment above) — extending that same # vault-mTLS read to XPRODUCT_INTERNAL_TOKEN instead of an env var # is the RIGHT fix, but requires a code change in that other repo. # 3. It is a distroless static image (confirmed: `docker run # --entrypoint sh ghcr.io/liamj74/coderaft-cve-proxy:latest` fails # with "sh: executable file not found") — no shell-wrapper # workaround is possible either. # Until cve-proxy's own repo adds vault-native or file-based reading of # this token, it stays a plaintext `environment:` value (visible via # `docker inspect`) — a real, deliberate, documented gap, not an # oversight. - XPRODUCT_INTERNAL_TOKEN=${XPRODUCT_INTERNAL_TOKEN} volumes: - ./vault-tls/client-ca.crt:/vault-tls/client-ca.crt:ro - ./vault-tls/cve-proxy-client.crt:/vault-tls/cve-proxy-client.crt:ro - ./vault-tls/cve-proxy-client.key:/vault-tls/cve-proxy-client.key:ro healthcheck: test: ["CMD", "/coderaft-cve-proxy", "-healthcheck"] interval: 30s timeout: 5s retries: 3 security_opt: [no-new-privileges:true] cap_drop: [ALL] mem_limit: 256m memswap_limit: 256m mem_reservation: 64m cpus: 1 restart: unless-stopped # ── self-update-runner ────────────────────────────────────────────────── # Root cause fix (2026-08-13, incident live chez Liam): dashboard-api's own # self-update (`docker compose up -d --force-recreate --no-deps # dashboard-api`) kills the very process issuing that command — Docker/runc # tears down the WHOLE cgroup on container stop (not just PID 1), so the # `docker compose` CLI child spawned by dashboard-api dies mid-sequence, # before it can start the replacement container. This companion, one-off # service runs the SAME pull + force-recreate + healthcheck step for # dashboard-api from a SEPARATE container (different cgroup, unaffected by # dashboard-api's own teardown) — see dashboard-api/routes/platform.js's # delegateSelfUpdate() and dashboard-api/scripts/self-update-runner.js for # the full mechanism. # # `profiles: [self-update-runner]`: never started by a normal `docker # compose up -d` (this installer, a plain restart, update.ps1/update.sh's # own reconcile pass, etc. carry no trace of it) — invoked EXCLUSIVELY via # `docker compose --profile self-update-runner run -d --rm --no-deps # self-update-runner`, emitted by dashboard-api itself right before its own # turn in the update loop, which then returns immediately (full hand-off). # # Same image as dashboard-api (just a different entrypoint) — no separate # build/pull surface to maintain in lockstep. DATABASE_URL / DOCKER_HOST / # CONTAINER_COMPOSE_DIR / the `/host-compose` bind mirror the dashboard-api # service above exactly: this companion calls the identical runCompose() # helper against the identical compose project. `dashboard_data:/data:ro` # is READ-ONLY — this companion only READS the Slack/Teams webhook secrets # from dashboard-api's own on-disk vault (vault.enc under /data, see # vault.js) to send the final update notification; it never writes. # # `networks:`: unlike dashboard-api, this service has no HTTP listener of # its own and never talks to coderaft-vault (its "vault" access is the # local /data:ro file, not the mTLS coderaft-vault container) — it only # needs coderaft-backend (to resolve `postgres`; pg_hba.conf rejects any # peer outside this network's pinned subnet, see the postgres service # below) and docker-proxy-net (to reach `docker-proxy` for DOCKER_HOST). # dashboard-api itself joins both of those networks too, so the healthcheck # HTTP call to `dashboard-api:3001` (see SERVICE_HEALTH_URLS in # routes/platform.js) resolves without any additional network. self-update-runner: image: ghcr.io/liamj74/coderaft-dashboard-api:latest entrypoint: ["node", "scripts/self-update-runner.js"] profiles: ["self-update-runner"] networks: - docker-proxy-net - coderaft-backend environment: - DATABASE_URL=postgres://coderaft:${POSTGRES_PASSWORD}@postgres:5432/coderaft - DOCKER_HOST=tcp://docker-proxy:2375 - CONTAINER_COMPOSE_DIR=/host-compose - COMPOSE_PROJECT_NAME=coderaft volumes: - .:/host-compose - dashboard_data:/data:ro security_opt: [no-new-privileges:true] cap_drop: [ALL] restart: "no" postgres: image: postgres:16-alpine environment: POSTGRES_USER: coderaft # Task #148 (2026-07-31): migrated to Docker's native `secrets:` — the # official postgres image already supports POSTGRES_PASSWORD_FILE, so # the resolved value no longer appears in `docker inspect`/Config.Env. # The file is materialized by install.sh (first boot) and by # dashboard-api's generateOverrideToDir() (every subsequent regen) at # ./secrets/postgres_password — see the `secrets:` top-level block below. POSTGRES_PASSWORD_FILE: /run/secrets/postgres_password POSTGRES_DB: coderaft POSTGRES_INITDB_ARGS: "--data-checksums" secrets: - postgres_password # F-025 (2026-07-31): restrict auth to scram-sha-256 only, and reject any # connection from outside the pinned coderaft-backend subnet # (172.28.42.0/24 — see the `coderaft-backend` network's `ipam:` config # below) at the pg_hba layer itself, not just Docker network isolation. # Verified live in a throwaway compose project: a peer container ON the # pinned subnet connects fine over scram-sha-256; a peer on a SEPARATE # docker network gets a hard `pg_hba.conf rejects connection ... no # encryption` — confirms the reject rule actually fires, not just that # the allow rule works. command: - postgres - -c - hba_file=/etc/postgresql/pg_hba.conf - -c - password_encryption=scram-sha-256 volumes: - postgres_data:/var/lib/postgresql/data - ./postgres/pg_hba.conf:/etc/postgresql/pg_hba.conf:ro # init-db.sql is intentionally NOT bind-mounted — when the # dashboard-api spawns docker-compose from inside a Linux container # against a Windows host, the resolved Windows path contains a # drive-letter colon that the daemon rejects ("too many colons"). # The script was a no-op anyway (just a comment); product databases # are created on demand by the dashboard. healthcheck: test: ["CMD-SHELL", "pg_isready -U coderaft"] interval: 5s timeout: 5s retries: 5 # B-PRODUCT-DB-NET (2026-06-23): products (entraguard, ravenscan, redfox) # are added dynamically by dashboard-api and attached to coderaft-backend. # Without postgres also on that network their first DNS lookup of # "postgres" returns ENOENT and alembic migrations crash with # "Name or service not known". networks: [coderaft-backend] security_opt: [no-new-privileges:true] cap_drop: [ALL] cap_add: [CHOWN, DAC_OVERRIDE, FOWNER, SETGID, SETUID] mem_limit: 1g memswap_limit: 1g mem_reservation: 256m cpus: 2 restart: unless-stopped redis: image: redis:7-alpine # Task #219 (2026-07-31, logical follow-up of #148 Phase 3): redis has # no native `_FILE` env var convention like postgres's image does — read # the password from the mounted secrets file via a shell command # override instead (redis:7-alpine is a real shell image, unlike the # distroless vault). `user: "999:1000"` pins the container to the # image's own unprivileged `redis` account directly (its built-in # UID:GID) — the stock entrypoint only auto-drops root to that user for # its OWN default `redis-server ...` CMD path, and skips that step for # any other command (including this `sh -c` override), which would # otherwise silently run as root. No persistent volume is mounted for # this service, so there is no ownership/chown concern from bypassing # the image's startup fixup step. user: "999:1000" command: ["sh", "-c", "redis-server --requirepass \"$(cat /run/secrets/redis_password)\" --maxmemory 128mb"] healthcheck: test: ["CMD-SHELL", "redis-cli --no-auth-warning -a \"$(cat /run/secrets/redis_password)\" ping"] interval: 5s timeout: 5s retries: 5 secrets: - redis_password # B-PRODUCT-DB-NET: same reason as postgres — product workers also # connect to redis://redis:6379 and must resolve the hostname. networks: [coderaft-backend] # F-011/F-014 (2026-07-31): verified live — PING/SET/BGSAVE all still # work with ALL caps dropped + read_only root + no extra tmpfs. redis: # 7-alpine's own Dockerfile declares `VOLUME /data` — Docker always gives # it a writable volume (anonymous here, since no explicit `/data` bind is # declared) regardless of the read_only root fs, which is what BGSAVE's # periodic RDB snapshot writes into; confirmed a manual `BGSAVE` still # says "Background saving started"/"DB saved on disk" under this config. security_opt: [no-new-privileges:true] cap_drop: [ALL] read_only: true mem_limit: 256m memswap_limit: 256m mem_reservation: 64m cpus: 1 restart: unless-stopped networks: # Internal network for vault ↔ product communication. No external port. coderaft-vault-net: internal: true # F-003 (2026-06-21): private network for the docker-socket-proxy. Only # dashboard-api joins it — external access to the daemon stays impossible. docker-proxy-net: internal: true # B-PRODUCT-DB-NET: backend network shared by data services (postgres, # redis, neo4j) and the dynamically-deployed products. # F-025 (2026-07-31): subnet PINNED (not left to Docker's auto-allocation) # so postgres/pg_hba.conf below can hard-code the exact CIDR it trusts — # otherwise a stack teardown+recreate could get a different auto-assigned # subnet from Docker's pool and silently lock every product out of # Postgres. Chosen to avoid the common Docker default-pool range # (172.17-172.20.0.0/16, where `docker0`/other local stacks often already # sit — see the RAVENSCAN_HOST_PORT comment above re: spineart-traefik # collisions) and to not collide with this repo's OTHER already-used # ranges (coderaft-vault-net/docker-proxy-net/coderaft-frontend/default — # see update.sh / dashboard-api for any of those pinned similarly). coderaft-backend: ipam: config: - subnet: 172.28.42.0/24 # B-DASHBOARD-NET: frontend network where the dashboard nginx and the # product HTTP listeners (entraguard-api, ravenscan, redfox-api) meet. coderaft-frontend: {} volumes: postgres_data: dashboard_data: caddy_data: caddy_config: vault_data: # F-014 (2026-07-31): named volume (not tmpfs) so the docker-socket-proxy # image's baked-in haproxy.cfg.template survives under read_only:true — # see the docker-proxy service's `volumes:` comment above. docker_proxy_haproxy_cfg: # Task #148 (2026-07-31): postgres's password, Docker-native file-based # secret. Must be a real path resolvable by the Docker DAEMON itself (unlike # .env's --env-file, this is a literal bind-mount source, not client-side # interpolation) — ./secrets/postgres_password, relative to this file's # directory (= --project-directory = HOST_PROJECT_DIR). Written by install.sh # on first boot and kept in sync by dashboard-api's generateOverrideToDir(). secrets: postgres_password: file: ./secrets/postgres_password # Task #219 (2026-07-31): same idea, redis — see the redis service's # `secrets:`/`user:`/`command:` above and write_redis_secret_file() above. redis_password: file: ./secrets/redis_password COMPOSE # ── postgres/pg_hba.conf (F-025, 2026-07-31) ───────────────────────────────── # scram-sha-256-only auth, restricted to the pinned coderaft-backend subnet # (172.28.42.0/24 — see the `coderaft-backend` network's `ipam:` block in the # docker-compose.yml heredoc above; the two MUST stay in sync, which is why # both are hard-coded to the same literal here rather than one being derived # from the other — this installer has no templating step between the two # files). Verified in a throwaway compose project: a peer container on the # pinned subnet connects fine; a peer on a different docker network gets a # hard `pg_hba.conf rejects connection ... no encryption`. # # Idempotent — like vault_bootstrap()/Caddyfile below, only written if # missing, so an operator's manual edits (e.g. adding a read replica's # `hostssl replication ... cert` line) survive `update.sh` re-runs. mkdir -p postgres if [ ! -f postgres/pg_hba.conf ]; then cat > postgres/pg_hba.conf << 'PGHBA' # Coderaft — managed by deploy/install.sh (F-025). scram-sha-256 only. # TYPE DATABASE USER ADDRESS METHOD # Unix socket (inside the postgres container itself, e.g. an operator # `docker compose exec postgres psql`). local all all scram-sha-256 # coderaft-backend Docker network only — every consumer (dashboard-api, # entraguard-api/worker/beat, ravenscan, redfox-api/gateway, falconone-*) # joins this network; nothing outside it can reach postgres:5432 at all # (no host port is published), so this is defense-in-depth on top of that # network-level isolation, not the only control. host all all 172.28.42.0/24 scram-sha-256 # Default deny — anything that isn't the exact subnet above (e.g. a stray # container mistakenly joined to more than one network) is rejected here, # not silently allowed by falling through to a permissive default. host all all 0.0.0.0/0 reject host all all ::/0 reject PGHBA chmod 644 postgres/pg_hba.conf fi # ── Caddyfile (env-templated TLS: internal CA / wildcard / ACME) ───────────── # The TLS mode is driven by env placeholders resolved at Caddy start: # CODERAFT_TLS_SITES site list (apex only in ACME mode — HTTP-01 cannot # issue wildcards) # CADDY_TLS_MODE_ARGS "internal" | "/certs/wildcard.crt /certs/wildcard.key" # | "" # dashboard-api (Setup Wizard → TLS step) updates .env and recreates caddy. # # Migration 2026-07: mkcert removed. A legacy Caddyfile referencing # coderaft.local.pem is backed up and regenerated (the old mkcert certs in # caddy_certs/ are simply ignored by the new template). if [ -f Caddyfile ] && grep -q "coderaft.local.pem" Caddyfile 2>/dev/null; then echo " Migrating legacy mkcert Caddyfile → Caddy internal CA (backup: Caddyfile.mkcert.bak)" mv Caddyfile Caddyfile.mkcert.bak fi if [ ! -f Caddyfile ]; then cat > Caddyfile << 'CADDY' { # Admin API stays off (security). auto_https stays ON: Caddy manages the # certificate for the configured sites (internal CA by default). admin off servers { # Trust X-Forwarded-* from private-range proxies (Exposure step: # "external reverse proxy" mode). trusted_proxies static private_ranges } } # F-007 (2026-07-31): baseline security headers for every site block. # # F-036 (2026-07-31) evaluated and closed WITHOUT a code change to the CSP # below — investigated live in a throwaway compose (caddy:2-alpine is # actually v2.11.4; verified against caddyserver/caddy's tplcontext.go), not # just the indicative snippet in REMEDIATION-PROMPT-2026-06-20.md: # - Caddy has NO built-in nonce template function (checked the real # http.handlers.templates function list: include/readFile/import/ # httpInclude/stripHTML/markdown/env/placeholder/fileExists/httpError/ # humanize/maybe/pathEscape — no `nonce`). A dynamic per-request nonce # would need a custom xcaddy build with a third-party plugin — real # ongoing maintenance cost for what `dashboard` is: a pure static Vite # SPA served by nginx (no per-request HTML generation at all — see the # comment on the `dashboard` service below), so there's no natural place # to even mint/thread a nonce server-side. # - Even if we had nonces, they only cover