Image vector to product details

This commit is contained in:
sriram
2026-09-19 15:39:53 +05:30
parent bc786b1c49
commit f933ea10a1
20 changed files with 2391 additions and 27 deletions

View File

@@ -21,6 +21,7 @@ WORKDIR /app
# image-search fallback (see requirements.txt); run `playwright install
# chromium` in the container if you need that specific fallback tier.
COPY requirements.txt .
COPY requirements-ocr.txt .
# torch is installed FIRST, from PyTorch's CPU-only index, and that ordering is
# the point. sentence-transformers pulls torch in transitively, and pip's
@@ -63,6 +64,14 @@ RUN /opt/venv/bin/pip install --no-cache-dir \
RUN /opt/venv/bin/pip install --no-cache-dir -r requirements.txt
# The OCR engine goes in AFTER requirements.txt and WITHOUT its declared
# dependencies. rapidocr's metadata asks for opencv-python (the GUI build);
# letting pip honour that would unpack it over opencv-python-headless and the
# result fails on libGL.so.1 in this image - taking the img_vector embedder
# down with it. Its real dependencies are in requirements.txt already; its
# PP-OCR models are inside the wheel, so nothing is downloaded at runtime.
RUN /opt/venv/bin/pip install --no-cache-dir --no-deps -r requirements-ocr.txt
# Strip payload the running service can never execute. Doing this in the build
# stage is what makes it count: the runtime stage copies /opt/venv as one layer,
# so anything deleted after that COPY would still occupy space in the layer