# syntax=docker/dockerfile:1 # Stage 1: Install dependencies FROM node:22-alpine AS deps RUN apk add --no-cache libc6-compat WORKDIR /app COPY package.json package-lock.json ./ # devDeps are required to build (typescript, tailwind, eslint-config-next). # This whole stage is discarded — none of it reaches the runner. RUN npm ci --no-audit --no-fund # Stage 2: Build the Next.js application FROM node:22-alpine AS builder WORKDIR /app COPY --from=deps /app/node_modules ./node_modules COPY . . ENV NEXT_TELEMETRY_DISABLED=1 ENV NODE_ENV=production # Each Docker build starts from a clean layer, so Turbopack's .next/cache is # written but never restored. Skipping it cuts ~20% of build CPU (the metric # that matters on a 1-vCPU host) and 116 MB off this layer. ENV CI_BUILD=1 RUN npm run build # Stage 3: Production runner with Next.js Standalone FROM node:22-alpine AS runner RUN apk add --no-cache libc6-compat WORKDIR /app ENV NODE_ENV=production ENV NEXT_TELEMETRY_DISABLED=1 ENV PORT=3000 ENV HOSTNAME="0.0.0.0" # ── Runtime configuration ──────────────────────────────────────────────── # # EXACTLY ONE variable must be supplied to this container. That is the whole # deployment contract, and it is one because everything else either ships in # the image or can only have one legal value. # # AUTH_SECRET signs the session cookie and encrypts the platform token # bundle. Generate with: openssl rand -hex 32 # → NOT shipped, and never can be: a secret in the image is # readable with `docker history`, and a secret in git is a # session-forging key for anyone who can read the repo. # → Set it in Dokploy → Environment (the RUNTIME panel — a # BUILD argument is not present when the server runs), or # mount it and set AUTH_SECRET_FILE to its path. # # AUTH_SECRET_FILE optional alternative: a path to read the secret from, the # standard Docker/Swarm secret convention. AUTH_SECRET wins # when both are set. Use this when a dashboard field mangles # the value. # # LOYALY_API_BASE no longer required. Production accepts exactly one origin # (https://mcp.loyaly.ai), so an unset variable could never # have meant anything else; shared/config/platformApi now # resolves it to that origin. Setting it to any OTHER host # is still rejected by name. It also still ships in the .env # copied below, which keeps `docker run` self-describing. # # Hex rather than base64 for the secret, on purpose. `openssl rand -base64 48` # ends in '=' and may contain '+' and '/'. Pasted into a dashboard field or a # KEY=VALUE editor that splits on the first '=', that value can be stored # truncated — or not at all — and the result is indistinguishable from never # having set it. Hex is [0-9a-f] only, so there is nothing for a parser to # mangle. 32 bytes is 256 bits, more than the HMAC and the AES-256 key derived # from it need. # # A container started without the secret does not die and does not 502. It # boots, names the missing variable on stderr (including any environment # variable whose NAME looks like a near-miss for AUTH_SECRET, which is the one # cause invisible from a dashboard), and answers 503 with # `x-loyaly-config: misconfigured` on every gated request. # # ── Where the secret must be set in Dokploy ────────────────────────────── # The "Environment Variables" tab. NOT "Build Arguments" and NOT "Build # Secrets": Dokploy's own documentation is explicit that both of those are # build-time only and are absent from the running container, so a secret placed # there is indistinguishable, from inside the container, from never having been # set at all. The boot log says which of the two happened. # # ── Two probe endpoints, deliberately separate ─────────────────────────── # /api/health LIVENESS — 200 whenever the process answers. Safe to probe # unconditionally; can never remove a serving # container from rotation. # /api/ready READINESS — 503 while a required variable is missing. Meant # for a DEPLOY gate, in Dokploy → Advanced → Swarm # Settings, paired with Update Config # `Order: start-first` + `FailureAction: rollback` # so a misconfigured new task is rolled back while # the previous good one keeps serving. # Run as a non-root user; nextjs owns nothing it does not need to write. RUN addgroup -g 1001 -S nodejs && adduser -u 1001 -S nextjs -G nodejs # Copy public static assets and standalone build output. # These three paths are the ENTIRE runtime payload (~57 MB). Never copy the # whole .next/ directory here — .next/dev and .next/cache are build-host-only # and account for ~1.96 GB. COPY --from=builder --chown=nextjs:nodejs /app/public ./public COPY --from=builder --chown=nextjs:nodejs /app/.next/standalone ./ COPY --from=builder --chown=nextjs:nodejs /app/.next/static ./.next/static # The production environment, as a file the server reads at boot. # # Deliberately redundant, and worth keeping. `next build` already copies .env # (and .env.production, and nothing else — see writeStandaloneDirectory in # next/dist/build/index.js) into .next/standalone, so the line above lands one # at /app/.env on its own. But it only does that when .env was in the BUILD # CONTEXT, and .dockerignore excluded it until recently — which is precisely # how images shipped with no LOYALY_API_BASE at all. # # This line turns that silent outcome into a loud one: exclude .env again and # the Docker build FAILS here with "file not found" instead of producing an # unconfigured image that starts and then rejects every sign-in. # # It does not pin the deployment either way: @next/env never overwrites a # variable already present in process.env, so anything set in Dokploy wins. COPY --chown=nextjs:nodejs .env ./.env USER nextjs EXPOSE 3000 # NO HEALTHCHECK, on purpose. # # One was added here and removed within the hour, because it recreated the # exact 502 it was meant to replace. Dokploy runs applications as Docker Swarm # services, and Swarm does not merely REPORT an unhealthy task — it pulls it # out of the service load balancer and reschedules it. So a healthcheck wired # to /api/health, which answers 503 while a required variable is missing, meant: # # AUTH_SECRET unset -> /api/health 503 -> task unhealthy -> removed from the # load balancer and restarted -> Traefik has no backend -> 502 Bad Gateway on # every url, which is precisely the symptom this whole change exists to end. # # The container would have been up, serving a 503 that names the fault, and # nobody could have reached it. "A broken deploy must not look healthy" is a # real concern, but enforcing it in the orchestrator destroys the diagnostics — # and an outage you cannot see the reason for is the more expensive failure. # # So: the container stays in rotation whenever it can serve HTTP at all, and # the configuration state is reported where it can actually be read — 503 with # `x-loyaly-config: misconfigured` on every gated request, /api/health for a # direct answer, and the named variable in the boot log. # # If a healthcheck is ever added back, it must probe LIVENESS (is the server # answering?) and never configuration, or this comment is being relearned. CMD ["node", "server.js"]