From c078bced04ef1fc9e9be0509193c6a6485a99bb2 Mon Sep 17 00:00:00 2001 From: Suriyakumarvijayanayagam Date: Thu, 13 Aug 2026 13:52:43 +0530 Subject: [PATCH] Stop an unreachable database from presenting as Bad Gateway Two faults compounded into one symptom: with the database unreachable, every route on the service returned 502 - including /docs, which never touches it. _connect() passed no connect_timeout. A host that DROPS packets rather than refusing them, which is what a firewall or a wrong DB_HOST looks like, blocked until the OS gave up - roughly 130 seconds on Linux. Every caller inherited that, /api/health included. Now bounded by DB_CONNECT_TIMEOUT_SECONDS, defaulting to 5. Measured against an unroutable host: /api/health went from hanging past 25s to answering 200 in 5.07s. The container healthcheck then probed /api/health, so that hang timed out the check, the container was marked unhealthy, and the platform stopped routing to it. That is the part that turned a degraded dependency into a total outage, and it was introduced with the healthcheck itself. A healthcheck is a LIVENESS question, because the platform's answer to "no" is to take the container out of service. It may only ask whether the process is still serving HTTP. /api/health is a READINESS report - it dials Postgres and Ollama to say whether they are reachable, and coupling the container's existence to its dependencies is what made a running API unreachable. It now probes "/", which is served from memory and does no I/O, so it can fail only if the app really is gone. Verified with an unroutable DB host: /, /docs and /openapi.json all answer 200, and the healthcheck exits 0. With the app stopped it still exits 1. Co-Authored-By: Claude Opus 5 (1M context) --- Dockerfile | 16 +++++++++------- app/infrastructure/settings.py | 8 ++++++++ app/services/vector_store.py | 15 +++++++++++++-- serve.py | 24 ++++++++++++++++++++---- 4 files changed, 50 insertions(+), 13 deletions(-) diff --git a/Dockerfile b/Dockerfile index 89e3a1b..699a8e1 100644 --- a/Dockerfile +++ b/Dockerfile @@ -81,14 +81,16 @@ VOLUME ["/app/data", "/app/app/intelligence/artifacts"] ENV PORTS=3000,8000 EXPOSE 3000 8000 -# Liveness only, and passes if EITHER port answers. /api/health always returns -# 200 - it reports Postgres and Ollama in the body as "degraded" rather than -# failing - which is deliberate: a check that went red whenever Postgres blinked -# would have Dokploy restart a perfectly healthy API in a loop. +# Liveness only, and passes if EITHER port answers. # -# The 10s timeout is not padding: the handler probes Ollama over HTTP with a 3s -# timeout of its own, so an unreachable Ollama makes every check take ~3s. -# start-period covers first boot, where the venv is still cold. +# It probes "/", which is served from memory, NOT /api/health, which dials +# Postgres and Ollama. That is the whole point: the platform's response to a +# failed healthcheck is to stop routing traffic, so this may only ask "is the +# process still serving HTTP". Tying it to the database meant an unreachable +# Postgres blocked the handler for the OS TCP timeout, the check timed out, the +# container was marked unhealthy, and a perfectly healthy API returned Bad +# Gateway on every route. Use /api/health to ask whether dependencies are up; +# it reports them in the body and always answers 200. HEALTHCHECK --interval=30s --timeout=10s --start-period=40s --retries=3 \ CMD ["python", "serve.py", "--healthcheck"] diff --git a/app/infrastructure/settings.py b/app/infrastructure/settings.py index 309a829..cd94d62 100644 --- a/app/infrastructure/settings.py +++ b/app/infrastructure/settings.py @@ -127,6 +127,14 @@ DB_PORT = os.getenv("DB_PORT", "5432") DB_NAME = os.getenv("DB_NAME", "pgvector") DB_USER = os.getenv("DB_USER", "postgres") DB_PASSWORD = _require("DB_PASSWORD", feature_flag="USE_PGVECTOR") if USE_PGVECTOR else os.getenv("DB_PASSWORD", "") +# How long to wait for the TCP connect before giving up. Matters more than it +# looks: a host that DROPS packets (a firewall, a typo'd DB_HOST) otherwise +# blocks until the OS timeout - about 130 seconds on Linux - and every request +# that touches the database inherits that wait, including /api/health. A short +# ceiling turns "the database is unreachable" into a fast, honest error instead +# of a hung worker and a container the platform decides is unhealthy. +DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5")) + DATABASE_URL = os.getenv( "DATABASE_URL", f"postgresql://{DB_USER}:{DB_PASSWORD}@{DB_HOST}:{DB_PORT}/{DB_NAME}", diff --git a/app/services/vector_store.py b/app/services/vector_store.py index 68af150..e4b3db2 100644 --- a/app/services/vector_store.py +++ b/app/services/vector_store.py @@ -7,7 +7,10 @@ import logging import re import psycopg -from app.infrastructure.settings import DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD +from app.infrastructure.settings import ( + DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD, + DB_CONNECT_TIMEOUT_SECONDS, +) from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand @@ -91,7 +94,15 @@ def _connect() -> Optional[psycopg.Connection]: dbname=DB_NAME, user=DB_USER, password=DB_PASSWORD, - autocommit=True + autocommit=True, + # Without this, a host that DROPS packets rather than refusing them + # - a firewall, a wrong DB_HOST - blocks here until the OS gives up, + # which is around 130 seconds on Linux. Every caller of _connect() + # inherits that: /api/health stops answering, the container's + # healthcheck times out, and the platform pulls the service out of + # its load balancer. "Database unreachable" then presents as a Bad + # Gateway on every route, including ones that never touch the DB. + connect_timeout=DB_CONNECT_TIMEOUT_SECONDS, ) except Exception as e: logger.error(f"Vector DB connection failed: {e}") diff --git a/serve.py b/serve.py index 8699d13..79f8780 100644 --- a/serve.py +++ b/serve.py @@ -129,15 +129,31 @@ def _healthcheck() -> int: Shares configured_ports() with the server so the check cannot drift from what is actually bound - the reason this lives here rather than being a python -c one-liner in the Dockerfile. + + Probes "/" rather than /api/health, and the distinction matters. This is a + LIVENESS check: the only question it may ask is "is this process still + serving HTTP", because the platform's answer to "no" is to stop routing + traffic to the container. + + /api/health is a READINESS report - it dials Postgres and Ollama to say + whether they are reachable. Using it here couples the container's existence + to its dependencies: an unreachable database made /api/health block for the + OS TCP timeout, the check timed out, the container was marked unhealthy, and + a service that was running perfectly well returned Bad Gateway on every + route - including the ones that never touch the database. "/" is served from + memory and does no I/O at all, so it can only fail if the app really is gone. """ ports = configured_ports() for port in ports: try: - with urllib.request.urlopen( - f"http://127.0.0.1:{port}/api/health", timeout=8 - ) as resp: - if 200 <= resp.status < 400: + with urllib.request.urlopen(f"http://127.0.0.1:{port}/", timeout=8) as resp: + if 200 <= resp.status < 500: return 0 + except urllib.error.HTTPError as exc: + # An HTTP status - even 404 - means something is listening and + # routing. That is exactly what liveness asks. + if exc.code < 500: + return 0 except (urllib.error.URLError, OSError, ValueError): continue print(