Files
catalogue_backend/Dockerfile
2026-09-19 15:39:53 +05:30

191 lines
10 KiB
Docker

# Multi-stage, mirroring catalogue_frontend/Dockerfile: a build stage that
# resolves dependencies, then a clean runtime stage that copies in only the
# result. There it is `npm ci` -> dist/; here it is pip -> a virtualenv.
# ---- Build stage ----
FROM python:3.11-slim AS build
WORKDIR /app
# Dependencies land in a self-contained venv so the runtime stage can take that
# one directory and leave pip, its HTTP cache and the downloaded wheels behind.
#
# requirements.txt is copied on its own, ahead of the source, for the same
# reason the frontend stage copies package.json before the rest of the app:
# this layer is cached on the file's checksum, so editing a router does not
# reinstall torch.
#
# psycopg[binary] avoids needing libpq-dev; sentence-transformers/scikit-learn/
# scipy all ship prebuilt wheels for this image, so no compiler is needed and
# this stage installs no build toolchain. Playwright's Python package installs,
# but its browser binary is NOT installed here - it's only a last-resort
# image-search fallback (see requirements.txt); run `playwright install
# chromium` in the container if you need that specific fallback tier.
COPY requirements.txt .
COPY requirements-ocr.txt .
# torch is installed FIRST, from PyTorch's CPU-only index, and that ordering is
# the point. sentence-transformers pulls torch in transitively, and pip's
# default wheel for Linux bundles the entire CUDA stack - cuBLAS, cuDNN, NVRTC,
# Triton - because it cannot know the target has no GPU. Measured in this image
# it was 2.7GB of nvidia/ plus 691MB of triton/, none of which can ever execute
# on a CPU-only VPS, in a 9.2GB image on a 48GB disk shared with a dozen
# services. A build here has already failed once on "no space left on device".
#
# Installing it up front means the requirements.txt pass below finds torch
# already satisfied and leaves it alone. Keep the two installs in that order.
#
# These three steps are also deliberately SEPARATE `RUN` layers rather than one
# chained command, and that is a deployment fix rather than tidiness. BuildKit
# caches COMPLETED steps, so as a single chained RUN there was no resume point:
# a build killed partway through torch redid the venv, the 191MB download and
# the unpack from zero on every retry. A Dokploy deploy died exactly there -
# "#8 CANCELED / failed to solve: Canceled: context canceled", with no pip
# traceback and no exit code, which is BuildKit reporting that the process
# driving the build went away, not that pip failed. `Installing collected
# packages` unpacks torch to ~1.3GB while the 191MB wheel is still on disk, so
# peak memory and peak disk land in the same second; DEPLOYMENT.md's Sizing
# section already warned that under 4GB "the build itself will fail". Split,
# the torch layer is banked the first time it succeeds and no later deploy
# runs it at all.
RUN python -m venv /opt/venv
# PINNED for the same reason. Unpinned, `pip install torch` took whatever the
# CPU index served that day - 2.13.0+cpu on the run that failed - so the layer
# below could never be trusted as cached and the image's size was set by a
# moving target.
#
# 2.12.1 rather than the newest available, because it is the version this
# project is actually developed and tested against - the working venv here runs
# torch 2.12.1 with sentence-transformers 5.6.0 - rather than whatever shipped
# most recently. The `+cpu` local version is part of the specifier: it is how
# the wheel is named on this index, and a bare `torch==2.12.1` would not match.
RUN /opt/venv/bin/pip install --no-cache-dir \
--index-url https://download.pytorch.org/whl/cpu torch==2.12.1+cpu
RUN /opt/venv/bin/pip install --no-cache-dir -r requirements.txt
# The OCR engine goes in AFTER requirements.txt and WITHOUT its declared
# dependencies. rapidocr's metadata asks for opencv-python (the GUI build);
# letting pip honour that would unpack it over opencv-python-headless and the
# result fails on libGL.so.1 in this image - taking the img_vector embedder
# down with it. Its real dependencies are in requirements.txt already; its
# PP-OCR models are inside the wheel, so nothing is downloaded at runtime.
RUN /opt/venv/bin/pip install --no-cache-dir --no-deps -r requirements-ocr.txt
# Strip payload the running service can never execute. Doing this in the build
# stage is what makes it count: the runtime stage copies /opt/venv as one layer,
# so anything deleted after that COPY would still occupy space in the layer
# below it. Deleting it here means the bytes are never in the final image.
#
# Measured on this image, the venv was 1.65GB, and this removes ~350MB of it:
#
# - Bundled test suites (~237MB, of which torch/test alone is 83MB). Every
# scientific wheel ships its own; pandas/tests is 40MB. Matched on exactly
# `tests`/`test` so numpy.testing and sklearn.utils._testing - which ARE
# imported by library code at runtime - are left alone.
# - torch/include (62MB): C++ headers, needed only to COMPILE an extension
# against libtorch. Nothing here does; torch is used through Python.
# - torch/bin (50MB): C++ gtest binaries (test_api is 15MB, test_jit 13.5MB)
# plus a protoc. torch_shm_manager is the one real program in there - it
# brokers shared-memory tensors between processes - so it is kept.
#
# Verified against this app rather than assumed: sympy IS pulled in by `import
# sentence_transformers` (via torch.fx), so it stays despite being 80MB and
# looking like a pure-math dependency nothing here would want.
RUN set -eux; \
SP=/opt/venv/lib/python3.11/site-packages; \
find "$SP" -type d \( -name tests -o -name test \) -prune -exec rm -rf {} +; \
rm -rf "$SP/torch/include"; \
find "$SP/torch/bin" -type f ! -name torch_shm_manager -delete
# ---- Runtime stage ----
FROM python:3.11-slim AS runtime
WORKDIR /app
# PATH: putting the venv first is what makes a bare `python`/`uvicorn` resolve
# to it - there is no "activate" step in a container.
# PYTHONUNBUFFERED: without it Dokploy's log view stays empty until a buffer
# happens to fill, so startup errors surface minutes after the container died.
# PYTHONDONTWRITEBYTECODE: no .pyc to write into a read-only-ish image layer.
ENV PATH="/opt/venv/bin:$PATH"
ENV PYTHONUNBUFFERED=1
ENV PYTHONDONTWRITEBYTECODE=1
COPY --from=build /opt/venv /opt/venv
COPY app ./app
COPY cli ./cli
COPY scripts ./scripts
COPY data ./data
COPY serve.py .
# The deployment's configuration, landing at /app/.env because settings.py
# resolves it from the backend root - app/infrastructure/settings.py takes
# parents[2], which is /app here - so it must sit next to app/, not inside it.
#
# The source file is named .env.production, NOT .env, and that detail is the
# whole point. Dokploy writes its own .env into the build context from the
# service's Environment tab AFTER cloning the repository. With that tab empty it
# writes an empty file, overwriting the committed one - so `COPY .env .` copied
# a zero-byte file, the container started with no configuration at all, and the
# platform reported only a Bad Gateway. The checkout showed it plainly: every
# file timestamped 08:33, and .env alone at 08:34, 0 bytes.
#
# Dokploy does not manage .env.production, so it survives. Anything set in the
# Environment tab still wins at runtime, because settings.py calls load_dotenv()
# without override=True and the process environment takes precedence.
COPY .env.production .env
# Pristine copies of everything the app also WRITES to, kept at a path that is
# never mounted over.
#
# /app/data/seed_catalogs and /app/app/intelligence/artifacts both need volumes
# (products added through the UI are appended to the first, retrained models
# are written to the second - otherwise a redeploy throws both away). But
# mounting a volume there hides the copies shipped in this image: a *named*
# volume is seeded from the image on first use, a *bind* mount is not, and
# Dokploy offers both. A bind mount would leave the API with zero seed catalogs
# and zero trained models, with nothing in the logs saying why.
#
# So keep a second copy here. On startup restore_bundled_assets()
# (app/infrastructure/persistence.py) copies in whatever the mounted directory
# is missing, and never overwrites what is already there.
RUN mkdir -p /app/.bundled \
&& cp -a /app/data/seed_catalogs /app/.bundled/seed_catalogs \
&& cp -a /app/app/intelligence/artifacts /app/.bundled/artifacts
# Declared so `docker run` without an explicit -v still gets an anonymous
# volume rather than writing into the container layer. Dokploy (and the compose
# file) name them properly; this is the floor, not the recommended setup.
VOLUME ["/app/data", "/app/app/intelligence/artifacts"]
# Answers on BOTH ports, the same way the frontend image does (nginx.conf has
# `listen 80; listen 3000;`). 3000 is what Dokploy routes a domain to; 8000 is
# what this project's README, the vite dev proxy and docker-compose all use.
# Serving both means the container works whichever one the platform is pointed
# at, instead of returning 502 from a perfectly healthy process.
#
# serve.py binds both sockets and hands them to one uvicorn - see the note
# there. To pin a single port, set PORT (PORT=8080 serves only 8080); to change
# the pair, set PORTS.
ENV PORTS=3000,8000
EXPOSE 3000 8000
# Liveness only, and passes if EITHER port answers.
#
# It probes "/", which is served from memory, NOT /api/health, which dials
# Postgres and Ollama. That is the whole point: the platform's response to a
# failed healthcheck is to stop routing traffic, so this may only ask "is the
# process still serving HTTP". Tying it to the database meant an unreachable
# Postgres blocked the handler for the OS TCP timeout, the check timed out, the
# container was marked unhealthy, and a perfectly healthy API returned Bad
# Gateway on every route. Use /api/health to ask whether dependencies are up;
# it reports them in the body and always answers 200.
HEALTHCHECK --interval=30s --timeout=10s --start-period=40s --retries=3 \
CMD ["python", "serve.py", "--healthcheck"]
# Exec form: python is PID 1, so Docker's SIGTERM reaches it directly and a
# redeploy shuts down cleanly instead of waiting out the 10s kill timeout.
CMD ["python", "serve.py"]