diff --git a/Dockerfile b/Dockerfile index 579d94d..22df9a9 100644 --- a/Dockerfile +++ b/Dockerfile @@ -50,6 +50,14 @@ ENV HOSTNAME="0.0.0.0" # bundle. Generate with: openssl rand -base64 48 # → NOT shipped. Set it as a Dokploy secret. # +# A container started WITHOUT AUTH_SECRET no longer dies. It boots, prints the +# missing variable on stderr, answers 503 with `x-loyaly-config: misconfigured` +# on every request, and fails the HEALTHCHECK below. Exiting instead is what +# made this fault present as a bare 502 Bad Gateway from the reverse proxy on +# every url — /favicon.ico first, in the browser console — with the one line +# that explained it trapped inside a restart-looping container. See +# src/instrumentation-node.ts. +# # AUTH_SECRET used to be an ENV line here with a literal value, which put a # session-forging key in git: anyone who could read the repo could mint a # cookie for any user, and every built image carried it in a layer that @@ -90,4 +98,23 @@ USER nextjs EXPOSE 3000 +# Is this container able to serve, or merely running? +# +# /api/health answers 200 only when every required variable is present and +# acceptable, and 503 otherwise, so a container missing AUTH_SECRET reports +# `unhealthy` rather than presenting itself as a working deploy. That is the +# property the old process.exit(1) was protecting; this keeps it without +# taking the site down to do it. +# +# Probed with node rather than curl or busybox wget: node is already the +# container's entrypoint, so this adds no package and cannot break because a +# base image dropped an applet. `r.ok` is the check — a 503 from /api/health +# exits 1, which is what marks the container unhealthy — and the catch covers +# a server that is not listening at all. +# +# start-period covers first boot: Next reports ready in well under 20s here, +# and an unready server must not be reported as a broken one. +HEALTHCHECK --interval=30s --timeout=5s --start-period=20s --retries=3 \ + CMD node -e "fetch('http://127.0.0.1:'+(process.env.PORT||3000)+'/api/health').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))" + CMD ["node", "server.js"] diff --git a/nginx.conf b/nginx.conf index bb4b57f..a95ed20 100644 --- a/nginx.conf +++ b/nginx.conf @@ -1,3 +1,20 @@ +# ───────────────────────────────────────────────────────────────────────────── +# NOT USED BY THE DEPLOYED IMAGE. Kept only as a reference. +# +# The image ran `node server.js & nginx -g 'daemon off;'` and exposed port 80 +# until commit 28258b5 ("fix docker port"), which dropped nginx and made the +# Next standalone server the container's only process on port 3000. Nothing +# installs nginx any more, and .dockerignore excludes this file from the build +# context, so editing it CANNOT affect production — the TLS termination and +# reverse proxy in front of the container are Dokploy's Traefik, configured in +# the Dokploy dashboard, not here. +# +# That matters when a 502 Bad Gateway shows up: this file is the obvious place +# to look and the wrong one. A 502 means Traefik had no healthy container to +# proxy to. Check the container's log for `[loyaly] configuration problem` and +# `GET /api/health` first. +# ───────────────────────────────────────────────────────────────────────────── + events { worker_connections 1024; } diff --git a/src/app/api/health/route.ts b/src/app/api/health/route.ts new file mode 100644 index 0000000..b195082 --- /dev/null +++ b/src/app/api/health/route.ts @@ -0,0 +1,51 @@ +import {configStatus} from '@/shared/config/configCheck'; + +/** + * GET /api/health — can this container serve? + * + * ── Why a container needs this ─────────────────────────────────────────── + * The boot check used to answer the same question by killing the process, on + * the reasoning that a dead container is the only signal a platform cannot + * misread. It is also the only signal a BROWSER cannot read: the reverse proxy + * in front of it had nothing to connect to and returned 502 for every url, + * which is what a missing AUTH_SECRET looked like from the outside. + * + * This is the half of that trade worth keeping. The container stays up and + * explains itself, while the Dockerfile's HEALTHCHECK polls this route and + * drives the container `unhealthy` when it answers 503 — so a misconfigured + * deploy still cannot present itself as a working one. + * + * ── What it deliberately does not say ──────────────────────────────────── + * Anonymous and public, so it publishes a state and a count, never the problem + * messages: those name environment variables, which is operator information + * (configError.ts states the rule; the login route follows it too). The names + * are printed once at boot, in the container log, where only an operator sees + * them. + * + * Exempt from the proxy's session gate — see HEALTH_PATH in src/proxy.ts — or + * an unauthenticated healthcheck would read 401 as "unhealthy" on a perfectly + * good container. + */ + +export const dynamic = 'force-dynamic'; + +export function GET(): Response { + const {problems} = configStatus(); + const healthy = problems.length === 0; + + return Response.json( + { + status: healthy ? 'ok' : 'misconfigured', + // A count, not the messages. Enough to tell "one variable missing" from + // "this container has nothing set at all" without publishing which. + problems: problems.length, + }, + { + status: healthy ? 200 : 503, + headers: { + 'cache-control': 'no-store', + 'x-loyaly-config': healthy ? 'ok' : 'misconfigured', + }, + }, + ); +} diff --git a/src/instrumentation-node.ts b/src/instrumentation-node.ts index a7abc91..1274000 100644 --- a/src/instrumentation-node.ts +++ b/src/instrumentation-node.ts @@ -1,54 +1,38 @@ -import {resolvePlatformOrigin} from '@/shared/config/platformApi'; -import {ConfigError} from '@/shared/errors/configError'; +import {configStatus} from '@/shared/config/configCheck'; /** - * The Node half of the boot-time configuration check. + * The Node half of the boot-time configuration check: it reports, it does not + * decide. `src/proxy.ts` is what turns a problem into a response. * - * ── Why this is a separate file ────────────────────────────────────────── + * ── Why this is a separate file from instrumentation.ts ────────────────── * `instrumentation.ts` is compiled for EVERY runtime the app uses, and - * src/proxy.ts makes this app compile an Edge one. `process.exit` does not - * exist there, so Turbopack statically flagged it — "A Node.js API is used - * (process.exit) which is not supported in the Edge Runtime" — even though the - * call sits behind a NEXT_RUNTIME guard and could never run in that bundle. + * src/proxy.ts makes this app compile an Edge one. Anything Node-only in there + * is flagged by Turbopack even when it sits behind a NEXT_RUNTIME guard — a + * runtime guard is not a bundling boundary, but a module reached through a + * dynamic import IS one, so the Edge bundle never contains this code at all. * - * A runtime guard is not a bundling boundary. A separate module reached through - * a dynamic import IS one, so the Edge bundle never contains this code at all. + * ── Why it no longer calls process.exit ────────────────────────────────── + * It did, and that is what turned "nobody set AUTH_SECRET" into a site-wide + * 502 Bad Gateway. The chain, measured end to end on a real standalone build: * - * These imports are static rather than dynamic because this module is only ever - * loaded from the Node branch. Neither is `server-only`: apiClient is, and that - * package resolves to a module which THROWS ON IMPORT outside a react-server - * condition, which this bundle is not. shared/config/platformApi exists so the - * boot check and the request path can share one validator without that hazard. + * AUTH_SECRET unset → this function exits 1 → the container dies on boot → + * Dokploy/Traefik has no upstream to proxy to → EVERY url answers 502, + * /favicon.ico included → the only explanation exists on stderr, inside a + * container that is restart-looping. + * + * A browser shown a 502 cannot tell a missing environment variable from a dead + * host, a bad port or an expired certificate, so the fastest fault to fix in + * this app became the slowest to identify. Exiting also made the fix + * unverifiable: there was no server to ask. + * + * The container now starts, and the proxy answers 503 with `x-loyaly-config: + * misconfigured` on every gated request while the Docker HEALTHCHECK (see the + * Dockerfile) drives the container unhealthy off /api/health. That keeps the + * property exiting was chosen for — a broken deploy must never look healthy — + * without taking away the thing that tells you what broke. */ export async function checkConfiguration() { - const isProduction = process.env.NODE_ENV === 'production'; - const problems: string[] = []; - - /** - * Resolve the platform origin exactly the way a request would — same - * function, same validation, same allowlist. A check that reimplemented the - * rules would be a second source of truth and would drift from the real one. - */ - let platform: string | null = null; - try { - platform = resolvePlatformOrigin(); - } catch (err) { - if (!(err instanceof ConfigError)) throw err; - problems.push(err.message); - } - - /** - * AUTH_SECRET is checked by presence rather than by calling the signing - * functions, which would have to be handed a throwaway payload to probe. - * Presence is the whole rule — sessionToken and tokenStore both refuse the - * development key in production and accept anything else. - */ - if (isProduction && !process.env.AUTH_SECRET) { - problems.push( - 'AUTH_SECRET is required in production — it signs the session cookie and ' + - 'encrypts the platform token bundle. Generate one with: openssl rand -base64 48', - ); - } + const {problems, platform} = configStatus(); if (problems.length > 0) { // Numbered, because a container missing its environment is usually missing @@ -56,27 +40,12 @@ export async function checkConfiguration() { // slow way. const detail = problems.map((p, i) => ` ${i + 1}. ${p}`).join('\n'); console.error( - `\n[loyaly] refusing to start — ${problems.length} configuration problem(s):\n${detail}\n` + - '\nSet these as environment variables on the container (Dokploy → Environment). ' + - 'LOYALY_API_BASE also ships in the committed .env; AUTH_SECRET never does.\n', + `\n[loyaly] configuration problem — ${problems.length} problem(s); ` + + `serving 503 until they are fixed:\n${detail}\n` + + '\nSet these as environment variables on the container (Dokploy → Environment), ' + + 'then redeploy. LOYALY_API_BASE also ships in the committed .env; AUTH_SECRET never does.\n', ); - - /** - * Exit, rather than throw. - * - * Throwing from `register()` does NOT stop the server. Measured against a - * real standalone build: the port is already bound by the time this runs, - * so Next logs "Failed to prepare server" and the process STAYS ALIVE, - * answering 500 to every route — /login, /favicon.ico, everything. That is - * the worst of both worlds. A TCP or HTTP healthcheck sees an open port and - * calls the container healthy, so a dead deploy shows up green and keeps - * receiving traffic. - * - * Exiting non-zero makes the container die, which is what a deployment - * platform reads as a failed deploy. The message above is already on - * stderr, so each restart attempt reprints the reason. - */ - process.exit(1); + return; } // The healthy path says what it resolved, so a log reader can confirm the diff --git a/src/instrumentation.ts b/src/instrumentation.ts index 9507839..0f75467 100644 --- a/src/instrumentation.ts +++ b/src/instrumentation.ts @@ -22,12 +22,13 @@ * below run in exactly the situation they are about: a real server, booting, * with a real environment. * - * ── Why it throws ──────────────────────────────────────────────────────── - * A container missing either variable cannot serve a single authenticated - * request. Refusing to start turns that into a failed deploy with a named - * cause in the log pane, which Dokploy surfaces immediately, instead of a - * green healthcheck in front of a console nobody can sign into. It also stops - * a broken image from replacing a working one. + * ── What it does about a problem ───────────────────────────────────────── + * Nothing, by itself. It records the problems and prints them, and the proxy + * turns them into a 503 with a named header on every gated request. It used to + * `process.exit(1)` here instead, which killed the container on boot and made + * the reverse proxy in front of it answer 502 for every url — including + * /favicon.ico, which is usually the first line anyone sees in a browser + * console. See instrumentation-node.ts for why that trade went the wrong way. */ export async function register() { /** diff --git a/src/proxy.ts b/src/proxy.ts index 35d1a28..2ef0def 100644 --- a/src/proxy.ts +++ b/src/proxy.ts @@ -1,5 +1,6 @@ import {NextResponse} from 'next/server'; import type {NextRequest} from 'next/server'; +import {configStatus} from '@/shared/config/configCheck'; import { SESSION_COOKIE, verifySessionToken, @@ -31,6 +32,61 @@ import { const LOGIN_PATH = '/login'; const HOME_PATH = '/dashboard'; +/** + * Never gated, by session or by configuration — it is how a container reports + * which of those two states it is in. See src/app/api/health/route.ts. + */ +const HEALTH_PATH = '/api/health'; + +/** + * A deployment that cannot serve says so, in one place, before anything that + * would need the missing value. + * + * ── Why this branch exists ─────────────────────────────────────────────── + * `verifySessionToken` below reads AUTH_SECRET and THROWS a ConfigError when + * production has none. Thrown from the proxy, that is an unhandled rejection + * per request and Next answers 500 — a status that says "this server has a + * bug", for a server that is merely unconfigured. Checking first replaces a + * 500-shaped accident with a deliberate 503, the status that actually means + * "not able to serve right now, this is expected to be temporary". + * + * ── Why the body does not name the variable ────────────────────────────── + * This endpoint is anonymous and public. The problem messages name environment + * variables, which is operator information — the same rule configError.ts + * states and the login route already follows. The name is printed once at boot + * (instrumentation-node.ts) where only an operator can read it; the response + * carries the header `x-loyaly-config: misconfigured`, which is enough to tell + * this apart from an unrelated outage without publishing anything. + */ +function misconfiguredResponse(pathname: string): NextResponse { + const headers = { + 'cache-control': 'no-store', + // Named so a 503 from THIS app is distinguishable from a 503 produced by + // the reverse proxy in front of it. Grep the container log for + // `[loyaly] configuration problem` for the detail. + 'x-loyaly-config': 'misconfigured', + }; + + if (pathname.startsWith('/api/')) { + return NextResponse.json( + { + error: { + code: 'internal', + message: 'This deployment is not configured. Check the server logs.', + }, + }, + {status: 503, headers}, + ); + } + + return new NextResponse( + 'This deployment is not configured and cannot serve requests.\n' + + 'An operator needs to set the missing environment variables on the ' + + 'container and redeploy; the container log names them.\n', + {status: 503, headers: {...headers, 'content-type': 'text/plain; charset=utf-8'}}, + ); +} + /** * Everything under these prefixes requires a session. Listed explicitly rather * than derived by exclusion: a new public page should have to opt IN to being @@ -43,6 +99,20 @@ const PUBLIC_PATHS = new Set([LOGIN_PATH]); export function proxy(request: NextRequest): NextResponse { const {pathname, search} = request.nextUrl; + // Answers in both states — that is the entire point of it — so it is let + // through ahead of the configuration gate and the session gate alike. + if (pathname === HEALTH_PATH) return NextResponse.next(); + + /** + * Configuration first, session second. A server missing AUTH_SECRET cannot + * verify a cookie at all, so asking it to try would only produce a worse + * error for the same fault. Memoised — see configCheck.ts — so a healthy + * server pays one array-length read per request. + */ + if (configStatus().problems.length > 0) { + return misconfiguredResponse(pathname); + } + const session = verifySessionToken( request.cookies.get(SESSION_COOKIE)?.value, ); diff --git a/src/shared/config/configCheck.ts b/src/shared/config/configCheck.ts new file mode 100644 index 0000000..603620e --- /dev/null +++ b/src/shared/config/configCheck.ts @@ -0,0 +1,79 @@ +/** + * The one list of "what must be true before this server can serve a request", + * shared by everything that needs to ask. + * + * ── Why it is its own module ───────────────────────────────────────────── + * Its three callers live in three different runtimes: + * + * src/instrumentation-node.ts Node, once at boot + * src/proxy.ts Edge, once per request + * src/app/api/health/route.ts Node, on demand + * + * So this file may import nothing runtime-specific. `platformApi` and + * `configError` are both deliberately free of `server-only` and of node: + * builtins for the same reason — see the headers on those two. + * + * Reimplementing the rules in any of the three would create a second source of + * truth, and the first thing it would do is drift: a boot check that passes + * while the request path throws is worse than no boot check at all. + */ + +import {ConfigError} from '@/shared/errors/configError'; +import {resolvePlatformOrigin} from '@/shared/config/platformApi'; + +export interface ConfigStatus { + /** Operator-facing messages. Empty means the server can serve. */ + problems: string[]; + /** The resolved upstream origin, or null when it could not be resolved. */ + platform: string | null; +} + +let cached: ConfigStatus | null = null; + +/** + * Both required variables are read from the real environment, which cannot + * change while the process lives, so this is memoised. A misconfigured server + * therefore pays the check once and answers from a boolean afterwards — which + * is what makes it affordable in the proxy, on every request. + */ +export function configStatus(): ConfigStatus { + if (cached) return cached; + + const problems: string[] = []; + let platform: string | null = null; + + /** + * Resolved through the SAME function a request uses, against the same + * allowlist, so a check that passes here cannot be contradicted by the first + * real call. A non-ConfigError is a bug in the resolver rather than a + * deployment fault, so it propagates. + */ + try { + platform = resolvePlatformOrigin(); + } catch (err) { + if (!(err instanceof ConfigError)) throw err; + problems.push(err.message); + } + + /** + * AUTH_SECRET is checked by presence, because presence is the whole rule: + * sessionToken and tokenStore refuse the development key in production and + * accept anything else. Calling the signing functions to probe would mean + * inventing a throwaway payload to sign. + */ + if (process.env.NODE_ENV === 'production' && !process.env.AUTH_SECRET) { + problems.push( + 'AUTH_SECRET is not set — it signs the session cookie and encrypts the ' + + 'platform token bundle, and production refuses to fall back to the ' + + 'development key. Set it on the container (Dokploy → Environment). ' + + 'Generate one with: openssl rand -base64 48', + ); + } + + return (cached = {problems, platform}); +} + +/** True when every required variable is present and acceptable. */ +export function isConfigured(): boolean { + return configStatus().problems.length === 0; +}