Two faults compounded into one symptom: with the database unreachable, every route on the service returned 502 - including /docs, which never touches it. _connect() passed no connect_timeout. A host that DROPS packets rather than refusing them, which is what a firewall or a wrong DB_HOST looks like, blocked until the OS gave up - roughly 130 seconds on Linux. Every caller inherited that, /api/health included. Now bounded by DB_CONNECT_TIMEOUT_SECONDS, defaulting to 5. Measured against an unroutable host: /api/health went from hanging past 25s to answering 200 in 5.07s. The container healthcheck then probed /api/health, so that hang timed out the check, the container was marked unhealthy, and the platform stopped routing to it. That is the part that turned a degraded dependency into a total outage, and it was introduced with the healthcheck itself. A healthcheck is a LIVENESS question, because the platform's answer to "no" is to take the container out of service. It may only ask whether the process is still serving HTTP. /api/health is a READINESS report - it dials Postgres and Ollama to say whether they are reachable, and coupling the container's existence to its dependencies is what made a running API unreachable. It now probes "/", which is served from memory and does no I/O, so it can fail only if the app really is gone. Verified with an unroutable DB host: /, /docs and /openapi.json all answer 200, and the healthcheck exits 0. With the app stopped it still exits 1. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
171 lines
6.3 KiB
Python
171 lines
6.3 KiB
Python
"""
|
|
Container entry point: serve the API on every port the platform might route to.
|
|
|
|
The frontend image answers on both 80 and 3000 (``listen 80; listen 3000;`` in
|
|
nginx.conf) so it works whatever port the deployment is configured to hit. This
|
|
does the same for the API, which otherwise has to guess: Dokploy routes a domain
|
|
to one container port, this project's own README, vite.config.js proxy and
|
|
docker-compose mapping all say 8000, and picking wrong produces a 502 with a
|
|
perfectly healthy container behind it.
|
|
|
|
uvicorn's CLI binds a single ``--port``, but ``Server.run()`` accepts a list of
|
|
already-bound sockets, so one process can listen on several. That is what this
|
|
does - no extra worker, no second process to supervise.
|
|
|
|
Usage::
|
|
|
|
python serve.py # binds PORTS (default "3000,8000")
|
|
PORT=8080 python serve.py # binds only 8080
|
|
python serve.py --healthcheck # probe mode, used by HEALTHCHECK
|
|
|
|
Run it directly rather than through ``uvicorn app.main:app`` when you need the
|
|
multi-port behaviour; the plain uvicorn command still works for local
|
|
development where one port is all anyone wants.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import os
|
|
import socket
|
|
import sys
|
|
import urllib.error
|
|
import urllib.request
|
|
|
|
logger = logging.getLogger("serve")
|
|
|
|
# Both of the ports this project actually uses anywhere: 3000 because that is
|
|
# what the platform routes a domain to by default, 8000 because the README,
|
|
# the vite dev proxy and docker-compose all target it.
|
|
DEFAULT_PORTS = "3000,8000"
|
|
|
|
HOST = os.getenv("HOST", "0.0.0.0")
|
|
|
|
|
|
def configured_ports() -> list[int]:
|
|
"""
|
|
The ports to bind, in order.
|
|
|
|
``PORT`` wins over ``PORTS`` and is treated as an explicit single choice:
|
|
setting it means "serve here", not "serve here as well". ``PORTS`` takes a
|
|
comma-separated list for the both-at-once case.
|
|
"""
|
|
raw = os.getenv("PORT") or os.getenv("PORTS") or DEFAULT_PORTS
|
|
|
|
ports: list[int] = []
|
|
for chunk in raw.split(","):
|
|
chunk = chunk.strip()
|
|
if not chunk:
|
|
continue
|
|
try:
|
|
port = int(chunk)
|
|
except ValueError:
|
|
logger.warning("Ignoring non-numeric port %r in %r", chunk, raw)
|
|
continue
|
|
if not 1 <= port <= 65535:
|
|
logger.warning("Ignoring out-of-range port %d", port)
|
|
continue
|
|
if port not in ports: # binding the same port twice would fail
|
|
ports.append(port)
|
|
|
|
if not ports:
|
|
logger.error("No usable port in %r - falling back to %s", raw, DEFAULT_PORTS)
|
|
return [int(p) for p in DEFAULT_PORTS.split(",")]
|
|
return ports
|
|
|
|
|
|
def _bind(port: int) -> socket.socket | None:
|
|
"""Bind one listening socket, or return None with the reason logged."""
|
|
sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
|
sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
|
try:
|
|
sock.bind((HOST, port))
|
|
except OSError as exc:
|
|
# Not fatal on its own. If the platform only routes to one of these,
|
|
# losing the other (already in use, not permitted) should not take the
|
|
# service down - _run() fails only when nothing at all is listening.
|
|
logger.warning("Could not bind %s:%d - %s", HOST, port, exc)
|
|
sock.close()
|
|
return None
|
|
sock.listen(2048)
|
|
sock.set_inheritable(True)
|
|
return sock
|
|
|
|
|
|
def _run() -> int:
|
|
import uvicorn
|
|
|
|
ports = configured_ports()
|
|
sockets = [s for s in (_bind(p) for p in ports) if s is not None]
|
|
|
|
if not sockets:
|
|
logger.error(
|
|
"Could not bind any of %s. The API is not listening; exiting so the "
|
|
"platform restarts or reports the container as failed.",
|
|
", ".join(str(p) for p in ports),
|
|
)
|
|
return 1
|
|
|
|
bound = [s.getsockname()[1] for s in sockets]
|
|
logger.info("Serving on %s port(s): %s", HOST, ", ".join(str(p) for p in bound))
|
|
|
|
config = uvicorn.Config(
|
|
"app.main:app",
|
|
# proxy_headers/forwarded_allow_ips: this container is never reached
|
|
# directly - nginx, and Dokploy's Traefik, sit in front of it. Without
|
|
# them uvicorn ignores X-Forwarded-Proto and reports every request as
|
|
# plain http, so any redirect or generated absolute URL would downgrade
|
|
# an https request.
|
|
proxy_headers=True,
|
|
forwarded_allow_ips="*",
|
|
)
|
|
uvicorn.Server(config).run(sockets=sockets)
|
|
return 0
|
|
|
|
|
|
def _healthcheck() -> int:
|
|
"""
|
|
Probe the API, passing if ANY configured port answers.
|
|
|
|
Shares configured_ports() with the server so the check cannot drift from
|
|
what is actually bound - the reason this lives here rather than being a
|
|
python -c one-liner in the Dockerfile.
|
|
|
|
Probes "/" rather than /api/health, and the distinction matters. This is a
|
|
LIVENESS check: the only question it may ask is "is this process still
|
|
serving HTTP", because the platform's answer to "no" is to stop routing
|
|
traffic to the container.
|
|
|
|
/api/health is a READINESS report - it dials Postgres and Ollama to say
|
|
whether they are reachable. Using it here couples the container's existence
|
|
to its dependencies: an unreachable database made /api/health block for the
|
|
OS TCP timeout, the check timed out, the container was marked unhealthy, and
|
|
a service that was running perfectly well returned Bad Gateway on every
|
|
route - including the ones that never touch the database. "/" is served from
|
|
memory and does no I/O at all, so it can only fail if the app really is gone.
|
|
"""
|
|
ports = configured_ports()
|
|
for port in ports:
|
|
try:
|
|
with urllib.request.urlopen(f"http://127.0.0.1:{port}/", timeout=8) as resp:
|
|
if 200 <= resp.status < 500:
|
|
return 0
|
|
except urllib.error.HTTPError as exc:
|
|
# An HTTP status - even 404 - means something is listening and
|
|
# routing. That is exactly what liveness asks.
|
|
if exc.code < 500:
|
|
return 0
|
|
except (urllib.error.URLError, OSError, ValueError):
|
|
continue
|
|
print(
|
|
f"health: no response from any of {', '.join(str(p) for p in ports)}",
|
|
file=sys.stderr,
|
|
)
|
|
return 1
|
|
|
|
|
|
if __name__ == "__main__":
|
|
logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s")
|
|
if "--healthcheck" in sys.argv:
|
|
sys.exit(_healthcheck())
|
|
sys.exit(_run())
|