From 8f25842281fce5a7a4d8661e4aeac25d77dc819d Mon Sep 17 00:00:00 2001 From: sriram Date: Mon, 31 Aug 2026 11:36:03 +0530 Subject: [PATCH] Dockerfile update --- Dockerfile | 36 +++++++++++++++++++++++++++++++----- 1 file changed, 31 insertions(+), 5 deletions(-) diff --git a/Dockerfile b/Dockerfile index dbacaa1..32de7ea 100644 --- a/Dockerfile +++ b/Dockerfile @@ -31,11 +31,37 @@ COPY requirements.txt . # services. A build here has already failed once on "no space left on device". # # Installing it up front means the requirements.txt pass below finds torch -# already satisfied and leaves it alone. Keep these two steps in this order. -RUN python -m venv /opt/venv \ - && /opt/venv/bin/pip install --no-cache-dir \ - --index-url https://download.pytorch.org/whl/cpu torch \ - && /opt/venv/bin/pip install --no-cache-dir -r requirements.txt +# already satisfied and leaves it alone. Keep the two installs in that order. +# +# These three steps are also deliberately SEPARATE `RUN` layers rather than one +# chained command, and that is a deployment fix rather than tidiness. BuildKit +# caches COMPLETED steps, so as a single chained RUN there was no resume point: +# a build killed partway through torch redid the venv, the 191MB download and +# the unpack from zero on every retry. A Dokploy deploy died exactly there - +# "#8 CANCELED / failed to solve: Canceled: context canceled", with no pip +# traceback and no exit code, which is BuildKit reporting that the process +# driving the build went away, not that pip failed. `Installing collected +# packages` unpacks torch to ~1.3GB while the 191MB wheel is still on disk, so +# peak memory and peak disk land in the same second; DEPLOYMENT.md's Sizing +# section already warned that under 4GB "the build itself will fail". Split, +# the torch layer is banked the first time it succeeds and no later deploy +# runs it at all. +RUN python -m venv /opt/venv + +# PINNED for the same reason. Unpinned, `pip install torch` took whatever the +# CPU index served that day - 2.13.0+cpu on the run that failed - so the layer +# below could never be trusted as cached and the image's size was set by a +# moving target. +# +# 2.12.1 rather than the newest available, because it is the version this +# project is actually developed and tested against - the working venv here runs +# torch 2.12.1 with sentence-transformers 5.6.0 - rather than whatever shipped +# most recently. The `+cpu` local version is part of the specifier: it is how +# the wheel is named on this index, and a bare `torch==2.12.1` would not match. +RUN /opt/venv/bin/pip install --no-cache-dir \ + --index-url https://download.pytorch.org/whl/cpu torch==2.12.1+cpu + +RUN /opt/venv/bin/pip install --no-cache-dir -r requirements.txt # Strip payload the running service can never execute. Doing this in the build # stage is what makes it count: the runtime stage copies /opt/venv as one layer,