diff --git a/.env b/.env index 78bd394..64cb0e0 100644 --- a/.env +++ b/.env @@ -53,4 +53,9 @@ NEXT_PUBLIC_API_BASE= # Dockerfile once for that reason; do not reintroduce it here. # # Set it as a Dokploy environment variable / secret. Production refuses to sign -# sessions without it. Generate with: openssl rand -base64 48 +# sessions without it. Generate with: openssl rand -hex 32 +# +# Hex, not base64: a base64 value ends in '=' and can contain '+' and '/', and +# an environment editor that splits a line on the first '=' can store that +# truncated or empty. A silently-empty AUTH_SECRET looks exactly like an unset +# one, which is a slow afternoon. Hex has nothing a parser can mangle. diff --git a/.env.example b/.env.example index 5313604..0ced718 100644 --- a/.env.example +++ b/.env.example @@ -29,7 +29,8 @@ LOYALY_API_BASE=http://127.0.0.1:8088 # as a Dokploy environment variable / secret. Locally, any string works; leave # it blank and a development key is used. # -# Generate with: openssl rand -base64 48 +# Generate with: openssl rand -hex 32 (hex, not base64 — a trailing '=' can be +# mangled by a dashboard env editor that splits on the first '=') AUTH_SECRET= # Browser → this app's own BFF routes. Same origin, so leave it empty. Inlined diff --git a/Dockerfile b/Dockerfile index 22df9a9..1710ee0 100644 --- a/Dockerfile +++ b/Dockerfile @@ -47,9 +47,16 @@ ENV HOSTNAME="0.0.0.0" # → SHIPPED, in the .env copied below. Not a secret. # # AUTH_SECRET signs the session cookie and encrypts the platform token -# bundle. Generate with: openssl rand -base64 48 +# bundle. Generate with: openssl rand -hex 32 # → NOT shipped. Set it as a Dokploy secret. # +# Hex rather than base64 on purpose. `openssl rand -base64 48` ends in '=' and +# may contain '+' and '/'. Pasted into a dashboard field or a KEY=VALUE env +# editor that splits on the first '=', that value can be stored truncated — or +# not at all — and the result is indistinguishable from never having set it. +# Hex is [0-9a-f] only, so there is nothing for a parser to mangle. 32 bytes is +# 256 bits, more than the HMAC and the AES-256 key derived from it need. +# # A container started WITHOUT AUTH_SECRET no longer dies. It boots, prints the # missing variable on stderr, answers 503 with `x-loyaly-config: misconfigured` # on every request, and fails the HEALTHCHECK below. Exiting instead is what @@ -98,23 +105,29 @@ USER nextjs EXPOSE 3000 -# Is this container able to serve, or merely running? +# NO HEALTHCHECK, on purpose. # -# /api/health answers 200 only when every required variable is present and -# acceptable, and 503 otherwise, so a container missing AUTH_SECRET reports -# `unhealthy` rather than presenting itself as a working deploy. That is the -# property the old process.exit(1) was protecting; this keeps it without -# taking the site down to do it. +# One was added here and removed within the hour, because it recreated the +# exact 502 it was meant to replace. Dokploy runs applications as Docker Swarm +# services, and Swarm does not merely REPORT an unhealthy task — it pulls it +# out of the service load balancer and reschedules it. So a healthcheck wired +# to /api/health, which answers 503 while a required variable is missing, meant: # -# Probed with node rather than curl or busybox wget: node is already the -# container's entrypoint, so this adds no package and cannot break because a -# base image dropped an applet. `r.ok` is the check — a 503 from /api/health -# exits 1, which is what marks the container unhealthy — and the catch covers -# a server that is not listening at all. +# AUTH_SECRET unset -> /api/health 503 -> task unhealthy -> removed from the +# load balancer and restarted -> Traefik has no backend -> 502 Bad Gateway on +# every url, which is precisely the symptom this whole change exists to end. # -# start-period covers first boot: Next reports ready in well under 20s here, -# and an unready server must not be reported as a broken one. -HEALTHCHECK --interval=30s --timeout=5s --start-period=20s --retries=3 \ - CMD node -e "fetch('http://127.0.0.1:'+(process.env.PORT||3000)+'/api/health').then(r=>process.exit(r.ok?0:1)).catch(()=>process.exit(1))" +# The container would have been up, serving a 503 that names the fault, and +# nobody could have reached it. "A broken deploy must not look healthy" is a +# real concern, but enforcing it in the orchestrator destroys the diagnostics — +# and an outage you cannot see the reason for is the more expensive failure. +# +# So: the container stays in rotation whenever it can serve HTTP at all, and +# the configuration state is reported where it can actually be read — 503 with +# `x-loyaly-config: misconfigured` on every gated request, /api/health for a +# direct answer, and the named variable in the boot log. +# +# If a healthcheck is ever added back, it must probe LIVENESS (is the server +# answering?) and never configuration, or this comment is being relearned. CMD ["node", "server.js"] diff --git a/src/instrumentation-node.ts b/src/instrumentation-node.ts index 1274000..8d6005d 100644 --- a/src/instrumentation-node.ts +++ b/src/instrumentation-node.ts @@ -51,9 +51,14 @@ export async function checkConfiguration() { // The healthy path says what it resolved, so a log reader can confirm the // host WITHOUT having to trigger a request. Never prints AUTH_SECRET, only // whether one was supplied. + const secret = process.env.AUTH_SECRET; console.log( `[loyaly] config ok — platform ${platform}, ` + - `auth secret ${process.env.AUTH_SECRET ? 'set' : 'using development key'}, ` + + // The LENGTH, never the value. `openssl rand -hex 32` gives 64 + // characters, so anything much shorter here is a secret that arrived + // truncated — which otherwise presents as sessions that mysteriously do + // not verify, with nothing in any log to suggest why. + `auth secret ${secret ? `set (${secret.length} chars)` : 'using development key'}, ` + `NODE_ENV=${process.env.NODE_ENV}`, ); } diff --git a/src/shared/config/configCheck.ts b/src/shared/config/configCheck.ts index 603620e..cd52e06 100644 --- a/src/shared/config/configCheck.ts +++ b/src/shared/config/configCheck.ts @@ -61,12 +61,21 @@ export function configStatus(): ConfigStatus { * accept anything else. Calling the signing functions to probe would mean * inventing a throwaway payload to sign. */ - if (process.env.NODE_ENV === 'production' && !process.env.AUTH_SECRET) { + /** + * `.trim()` rather than a bare falsiness check, because the failure this is + * most often covering for is a value that ARRIVED but arrived empty — a + * dashboard field that stored whitespace, or a KEY=VALUE line whose value was + * eaten. An AUTH_SECRET of " " is not a configured secret, and treating it as + * one would put a one-character key behind every session in the deployment. + */ + if (process.env.NODE_ENV === 'production' && !process.env.AUTH_SECRET?.trim()) { problems.push( 'AUTH_SECRET is not set — it signs the session cookie and encrypts the ' + 'platform token bundle, and production refuses to fall back to the ' + 'development key. Set it on the container (Dokploy → Environment). ' + - 'Generate one with: openssl rand -base64 48', + 'Generate one with: openssl rand -hex 32 — hex, not base64, because a ' + + "base64 value ends in '=' and a dashboard field that splits on the " + + 'first = can silently store a truncated or empty value.', ); }