Fix nutrition upload crash, persist runtime writes, serve API on its own domain

/api/upload/nutrition called json.dumps() in a module that never imported
json, so every request to it raised NameError, was swallowed by the broad
except, and came back as "500 Database import failed". Import json.

Persist the three directories the app writes to at runtime. Products added
through the UI are appended to data/seed_catalogs/*.json and retrained models
are written to app/intelligence/artifacts/*.joblib; both live inside the image,
so a redeploy silently discarded them. The paths now come from settings
(DATA_DIR / SEED_CATALOG_DIR / MODEL_ARTIFACTS_DIR) so a volume can be mounted
on them, and catalog_engine.save_catalog resolves against DATA_DIR instead of
a working-directory-relative "data/", which landed somewhere different
depending on where the process was started from.

Mounting those volumes would otherwise have made things worse: Docker seeds a
named volume from the image on first use, but a bind mount starts empty and
just hides what the image shipped. A bind mount on /app/data would have left
the API with no seed catalogs, so the next product added would write a JSON
file containing only that product. The image now keeps pristine copies at
/app/.bundled, and restore_bundled_assets() tops up whatever a freshly mounted
directory is missing at startup without overwriting anything already there.

Configure CORS for the split-domain deployment: the React app is served from
catalogue.nearle.ai.in and calls the API on mcp.catalogue.nearle.ai.in, so the
frontend origin has to be in API_CORS_ORIGINS. A wrong list fails only in the
browser while the server logs a healthy 200, so the effective origins are now
logged at startup with a warning when they are localhost-only.

Fix FRONTEND_DIST, which looked for a sibling "frontend/" directory that is
actually named "catalogue_frontend/", so the single-port unified-serving branch
could never activate even with a build sitting next to it.

Rebuild the Dockerfile on the frontend's multi-stage pattern: dependencies
resolve into a venv in a build stage, the runtime stage copies only that.
Adds PYTHONUNBUFFERED so startup errors reach Dokploy's log pane, a liveness
HEALTHCHECK (/api/health answers 200 even when Postgres is down, so a database
blip cannot restart-loop the container), and an overridable PORT. The CMD execs
uvicorn so SIGTERM reaches it rather than the sh wrapper.

Add "from __future__ import annotations" to ollama_service and image_search,
which used PEP 604 unions in runtime-evaluated signatures and so could not be
imported below Python 3.10.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
Suriyakumarvijayanayagam
2026-08-13 12:30:45 +05:30
parent b8d93fbbf2
commit 2493b86ed8
13 changed files with 498 additions and 26 deletions

View File

@@ -1,6 +1,7 @@
from __future__ import annotations
import io
import json
import uuid
import logging
import pandas as pd

View File

@@ -9,6 +9,7 @@ from pydantic import BaseModel, Field
from fastapi import APIRouter, Depends, HTTPException, File, UploadFile
from app.api.deps import require_permission
from app.infrastructure.settings import SEED_CATALOG_DIR
from app.services.vector_store import (
upsert_brand_products,
@@ -22,7 +23,10 @@ from app.services.s3_service import s3_service
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/user/products", tags=["user_products"])
SEED_DIR = Path(__file__).resolve().parents[3] / "data" / "seed_catalogs"
# Written to by /add and /upload-file below. Comes from settings so it can be
# pointed at a mounted volume - inside the image these writes are discarded on
# every redeploy. See app/infrastructure/persistence.py.
SEED_DIR = SEED_CATALOG_DIR
class AddProductRequest(BaseModel):

View File

@@ -19,7 +19,7 @@ sys.path.append(str(Path(__file__).parent.parent))
from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details
from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts
from app.infrastructure.settings import USE_OLLAMA
from app.infrastructure.settings import DATA_DIR, USE_OLLAMA
from app.services.embeddings_service import embed_texts
from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id
from app.services.s3_service import s3_service
@@ -948,20 +948,30 @@ class ProductCatalogEngine:
return catalog
def save_catalog(self, catalog: Dict[str, Any], filename: str = None) -> str:
"""Save catalog to JSON file inside data/ folder"""
"""Save catalog to JSON file inside DATA_DIR.
Resolved against DATA_DIR rather than a bare relative "data/" path: the
latter depends on the process's working directory, so the same call
landed in a different place depending on whether the app was started
from the repo root, from backend/, or by the container's uvicorn - and
only one of those is the directory with a volume mounted on it.
"""
if not filename:
brand = catalog.get('brand', 'unknown')
storage_brand = resolve_parent_brand(brand)
safe_brand = storage_brand.replace(' ', '_')
ts = catalog.get('generation_timestamp', 'latest')
filename = f"data/catalog_{safe_brand}_{ts}.json"
Path("data").mkdir(exist_ok=True)
with open(filename, 'w', encoding='utf-8') as f:
filename = f"catalog_{safe_brand}_{ts}.json"
path = Path(filename)
if not path.is_absolute():
path = DATA_DIR / path
path.parent.mkdir(parents=True, exist_ok=True)
with open(path, 'w', encoding='utf-8') as f:
json.dump(catalog, f, indent=2, ensure_ascii=False)
logger.info(f"💾 Catalog saved to: {filename}")
return filename
logger.info(f"💾 Catalog saved to: {path}")
return str(path)
# Global engine instance
catalog_engine = ProductCatalogEngine()

View File

@@ -0,0 +1,146 @@
"""
First-run restore of the assets bundled into the container image.
The problem this solves
----------------------
Three directories are written to at runtime - the seed catalogs appended to by
``POST /api/user/products/add``, the generated catalogs from the ingestion
pipeline, and the ``*.joblib`` bundles the training endpoints produce. In a
container all three sit inside the image, so a redeploy silently discards
every one of them. They need to be on a volume.
Mounting a volume over them introduces the opposite problem. A *named* Docker
volume is seeded from the image the first time it is used, but a *bind* mount
starts empty and merely hides what the image had underneath. Dokploy offers
both and neither announces which you picked, so a bind mount on ``/app/data``
would leave the API running with no seed catalogs: the next product added would
write a JSON file containing only that product, and every model would report as
untrained.
The fix is to keep a pristine copy inside the image at a path nobody mounts
(``BUNDLED_ASSETS_DIR``, populated by the Dockerfile) and top up the writable
directory from it on startup.
Existing files are never overwritten. That is the whole contract: the bundle
supplies what is missing, and anything the running app has already written wins
over the copy baked into the image. Without that rule every redeploy would
revert user-added products back to the bundled catalog.
"""
from __future__ import annotations
import logging
import shutil
from pathlib import Path
from typing import Tuple
from app.infrastructure.settings import (
BUNDLED_ASSETS_DIR,
DATA_DIR,
MODEL_ARTIFACTS_DIR,
SEED_CATALOG_DIR,
)
logger = logging.getLogger(__name__)
# (subdirectory under BUNDLED_ASSETS_DIR, writable destination)
_BUNDLES: Tuple[Tuple[str, Path], ...] = (
("seed_catalogs", SEED_CATALOG_DIR),
("artifacts", MODEL_ARTIFACTS_DIR),
)
def _restore_one(source: Path, destination: Path) -> int:
"""Copy files missing from ``destination``. Returns how many were copied."""
if not source.is_dir():
return 0
destination.mkdir(parents=True, exist_ok=True)
copied = 0
for item in sorted(source.iterdir()):
if not item.is_file():
continue
target = destination / item.name
if target.exists():
continue
try:
# copy2 rather than copy: it preserves mtime, so "is this artifact
# older than the data it was trained on" stays answerable.
shutil.copy2(item, target)
copied += 1
except OSError as exc:
logger.warning("Could not restore %s -> %s: %s", item, target, exc)
return copied
def restore_bundled_assets() -> None:
"""
Top up the writable directories from the image's read-only bundle.
Safe to call on every boot: it is a no-op once the volume is populated, and
a no-op outside Docker where BUNDLED_ASSETS_DIR does not exist.
"""
# Create these regardless. A volume mounted at DATA_DIR arrives empty, and
# the ingestion pipeline writes into it without creating it first.
for path in (DATA_DIR, SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR):
try:
path.mkdir(parents=True, exist_ok=True)
except OSError as exc:
logger.error(
"Cannot create writable directory %s: %s. Uploads and trained "
"models will fail to save - check the volume's permissions.",
path,
exc,
)
if not BUNDLED_ASSETS_DIR.is_dir():
logger.debug(
"No bundled asset directory at %s - nothing to restore.", BUNDLED_ASSETS_DIR
)
return
for name, destination in _BUNDLES:
copied = _restore_one(BUNDLED_ASSETS_DIR / name, destination)
if copied:
logger.info(
"Restored %d bundled file(s) into %s (first run on this volume).",
copied,
destination,
)
_warn_if_not_persistent()
def _warn_if_not_persistent() -> None:
"""
Point out that the writable directories are still inside the image.
Only reachable when BUNDLED_ASSETS_DIR exists, i.e. in the container. If the
destinations were never mounted, everything written to them is lost on the
next redeploy - which looks exactly like the app quietly ignoring uploads,
hours later and with nothing in the logs to connect it to.
"""
unmounted = [p for p in (SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR) if not _is_mount(p)]
if unmounted:
logger.warning(
"These directories are written at runtime but do not look like "
"mount points: %s. Anything saved there - products added through "
"the UI, retrained models - will be discarded on the next "
"redeploy. Mount a volume on each (see backend/README.md).",
", ".join(str(p) for p in unmounted),
)
def _is_mount(path: Path) -> bool:
"""
Whether ``path`` sits on a different device than its parent.
A mounted volume shows up as a device-number change. Falls back to True on
error so a probe failure produces silence rather than a false alarm telling
somebody their correctly-mounted volume is broken.
"""
try:
if path.is_mount():
return True
return path.stat().st_dev != path.parent.stat().st_dev
except OSError:
return True

View File

@@ -55,6 +55,52 @@ def _require(name: str, *, feature_flag: str) -> str:
return value
# ---------------------------------------------------------------------------
# Writable data directories (persistence)
# ---------------------------------------------------------------------------
# Everything the running app WRITES lives under one of these three paths. They
# are settings rather than hard-coded paths because in a container they must be
# mounted on a volume - otherwise every product added through the UI and every
# retrained model is discarded the next time the image is redeployed.
#
# DATA_DIR generated catalogs (catalog_engine.save_catalog)
# SEED_CATALOG_DIR per-brand JSON catalogs, appended to by
# POST /api/user/products/add and /upload-file
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
#
# See BUNDLED_ASSETS_DIR below for how the read-only copies shipped inside the
# image get into these directories the first time a volume is mounted.
_BACKEND_ROOT = Path(__file__).resolve().parents[2]
def _dir(name: str, default: Path) -> Path:
raw = os.getenv(name, "").strip()
return Path(raw).expanduser() if raw else default
DATA_DIR = _dir("DATA_DIR", _BACKEND_ROOT / "data")
SEED_CATALOG_DIR = _dir("SEED_CATALOG_DIR", DATA_DIR / "seed_catalogs")
MODEL_ARTIFACTS_DIR = _dir(
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
)
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
# here by the Dockerfile at a path that is never itself mounted over.
#
# This exists because the two ways of mounting a volume behave differently, and
# the difference is silent. Docker copies the image's content into a *named*
# volume the first time it is used, but a *bind* mount starts empty and simply
# hides whatever the image had at that path. Mounting a bind mount on /app/data
# would therefore leave the app with no seed catalogs at all: the next product
# added would write a fresh JSON file containing only that one product, and the
# ML endpoints would report no trained models.
#
# So the image keeps a second, unmounted copy, and restore_bundled_assets()
# (app/infrastructure/persistence.py) fills in whatever the writable directory
# is missing at startup. Empty/absent outside Docker, where nothing is mounted
# and the defaults above already point at the real files.
BUNDLED_ASSETS_DIR = _dir("BUNDLED_ASSETS_DIR", Path("/app/.bundled"))
# ---------------------------------------------------------------------------
# Ollama (local LLM)
# ---------------------------------------------------------------------------

View File

@@ -13,9 +13,15 @@ from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional
from app.infrastructure.settings import MODEL_ARTIFACTS_DIR
logger = logging.getLogger(__name__)
ARTIFACTS_DIR = Path(__file__).resolve().parent / "artifacts"
# From settings so it can be pointed at a mounted volume: retraining writes
# *.joblib here, and inside the image those are discarded on the next redeploy,
# silently reverting every model to the version baked in at build time.
# Defaults to this package's own artifacts/ directory.
ARTIFACTS_DIR = MODEL_ARTIFACTS_DIR
ARTIFACTS_DIR.mkdir(parents=True, exist_ok=True)

View File

@@ -9,6 +9,7 @@ Run with:
from __future__ import annotations
import logging
import os
import threading
from contextlib import asynccontextmanager
from pathlib import Path
@@ -18,6 +19,7 @@ from fastapi.middleware.cors import CORSMiddleware
from fastapi.staticfiles import StaticFiles
from fastapi.responses import FileResponse
from app.infrastructure.persistence import restore_bundled_assets
from app.infrastructure.settings import API_CORS_ORIGINS
from app.api.routers import health, brands, search, chat, catalog, system
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
@@ -43,6 +45,15 @@ async def lifespan(_app: FastAPI):
healthcheck pass while that settles.
"""
# Runs before the thread below, and synchronously: the seed catalogs and
# model artifacts have to be in place before the first request can read
# them, and it is a handful of file copies on first boot, nothing on every
# boot after that.
try:
restore_bundled_assets()
except Exception as e:
logger.warning("Could not restore bundled assets: %s", e)
def _async_init():
try:
ensure_store_intelligence_schema()
@@ -93,6 +104,24 @@ app.add_middleware(
allow_headers=["*"],
)
# A wrong origin list fails only in the browser, as an opaque "blocked by CORS"
# with a perfectly healthy 200 in the server log - so state the effective list
# at startup, where it can actually be compared against the frontend's URL.
logger.info(
"CORS allowed origins: %s",
", ".join(API_CORS_ORIGINS) if API_CORS_ORIGINS else "(none - same-origin only)",
)
if API_CORS_ORIGINS and all(
o.startswith(("http://localhost", "http://127.0.0.1")) for o in API_CORS_ORIGINS
):
logger.warning(
"API_CORS_ORIGINS lists only localhost origins (%s). If this API is "
"published on a domain and the frontend is served from a different one, "
"every browser request will be blocked. Set API_CORS_ORIGINS to the "
"frontend's exact origin, e.g. https://catalogue.nearle.ai.in",
", ".join(API_CORS_ORIGINS),
)
app.include_router(health.router, prefix="/api")
app.include_router(auth.router, prefix="/api")
app.include_router(user_products.router, prefix="/api")
@@ -112,9 +141,34 @@ app.include_router(nutrition.router, prefix="/api")
app.include_router(nutrition_admin.router, prefix="/api")
app.include_router(upload.router, prefix="/api")
# Serve built frontend static files if dist exists (single-port unified deployment)
FRONTEND_DIST = Path(__file__).resolve().parents[2] / "frontend" / "dist"
# Serve built frontend static files if dist exists (single-port unified
# deployment). Off in the normal setup: the React app is served by its own
# nginx on catalogue.nearle.ai.in and this API answers on
# mcp.catalogue.nearle.ai.in, so no dist/ is present here and the JSON root
# handler at the bottom of this file is what responds to /.
#
# The candidates cover both repo layouts - the sibling checkout is named
# `catalogue_frontend`, and only `frontend` was checked before, so this branch
# could never activate even when a build was sitting right next to it.
# FRONTEND_DIST_DIR overrides both when the build lands somewhere else.
_dist_override = os.getenv("FRONTEND_DIST_DIR", "").strip()
_repo_root = Path(__file__).resolve().parents[2]
_dist_candidates = (
[Path(_dist_override)]
if _dist_override
else [
_repo_root / "catalogue_frontend" / "dist",
_repo_root / "frontend" / "dist",
Path(__file__).resolve().parents[1] / "frontend" / "dist",
]
)
FRONTEND_DIST = next(
(p for p in _dist_candidates if (p / "assets").exists()),
_dist_candidates[0],
)
if FRONTEND_DIST.exists() and (FRONTEND_DIST / "assets").exists():
logger.info("Serving built frontend from %s", FRONTEND_DIST)
app.mount("/assets", StaticFiles(directory=str(FRONTEND_DIST / "assets")), name="assets")
@app.get("/{full_path:path}")

View File

@@ -42,6 +42,8 @@ actually resolves to real image bytes above a minimum size - this is what
stops garbage/placeholder/expired URLs from reaching the S3 upload step
and failing there silently.
"""
from __future__ import annotations
from typing import Optional, List
import requests
from urllib.parse import urlparse

View File

@@ -1,3 +1,5 @@
from __future__ import annotations
from typing import List, Dict, Any, Optional
import json
import re