Fix nutrition upload crash, persist runtime writes, serve API on its own domain
/api/upload/nutrition called json.dumps() in a module that never imported json, so every request to it raised NameError, was swallowed by the broad except, and came back as "500 Database import failed". Import json. Persist the three directories the app writes to at runtime. Products added through the UI are appended to data/seed_catalogs/*.json and retrained models are written to app/intelligence/artifacts/*.joblib; both live inside the image, so a redeploy silently discarded them. The paths now come from settings (DATA_DIR / SEED_CATALOG_DIR / MODEL_ARTIFACTS_DIR) so a volume can be mounted on them, and catalog_engine.save_catalog resolves against DATA_DIR instead of a working-directory-relative "data/", which landed somewhere different depending on where the process was started from. Mounting those volumes would otherwise have made things worse: Docker seeds a named volume from the image on first use, but a bind mount starts empty and just hides what the image shipped. A bind mount on /app/data would have left the API with no seed catalogs, so the next product added would write a JSON file containing only that product. The image now keeps pristine copies at /app/.bundled, and restore_bundled_assets() tops up whatever a freshly mounted directory is missing at startup without overwriting anything already there. Configure CORS for the split-domain deployment: the React app is served from catalogue.nearle.ai.in and calls the API on mcp.catalogue.nearle.ai.in, so the frontend origin has to be in API_CORS_ORIGINS. A wrong list fails only in the browser while the server logs a healthy 200, so the effective origins are now logged at startup with a warning when they are localhost-only. Fix FRONTEND_DIST, which looked for a sibling "frontend/" directory that is actually named "catalogue_frontend/", so the single-port unified-serving branch could never activate even with a build sitting next to it. Rebuild the Dockerfile on the frontend's multi-stage pattern: dependencies resolve into a venv in a build stage, the runtime stage copies only that. Adds PYTHONUNBUFFERED so startup errors reach Dokploy's log pane, a liveness HEALTHCHECK (/api/health answers 200 even when Postgres is down, so a database blip cannot restart-loop the container), and an overridable PORT. The CMD execs uvicorn so SIGTERM reaches it rather than the sh wrapper. Add "from __future__ import annotations" to ollama_service and image_search, which used PEP 604 unions in runtime-evaluated signatures and so could not be imported below Python 3.10. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -1,6 +1,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import uuid
|
||||
import logging
|
||||
import pandas as pd
|
||||
|
||||
@@ -9,6 +9,7 @@ from pydantic import BaseModel, Field
|
||||
from fastapi import APIRouter, Depends, HTTPException, File, UploadFile
|
||||
|
||||
from app.api.deps import require_permission
|
||||
from app.infrastructure.settings import SEED_CATALOG_DIR
|
||||
|
||||
from app.services.vector_store import (
|
||||
upsert_brand_products,
|
||||
@@ -22,7 +23,10 @@ from app.services.s3_service import s3_service
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/user/products", tags=["user_products"])
|
||||
|
||||
SEED_DIR = Path(__file__).resolve().parents[3] / "data" / "seed_catalogs"
|
||||
# Written to by /add and /upload-file below. Comes from settings so it can be
|
||||
# pointed at a mounted volume - inside the image these writes are discarded on
|
||||
# every redeploy. See app/infrastructure/persistence.py.
|
||||
SEED_DIR = SEED_CATALOG_DIR
|
||||
|
||||
|
||||
class AddProductRequest(BaseModel):
|
||||
|
||||
@@ -19,7 +19,7 @@ sys.path.append(str(Path(__file__).parent.parent))
|
||||
|
||||
from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details
|
||||
from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts
|
||||
from app.infrastructure.settings import USE_OLLAMA
|
||||
from app.infrastructure.settings import DATA_DIR, USE_OLLAMA
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id
|
||||
from app.services.s3_service import s3_service
|
||||
@@ -948,20 +948,30 @@ class ProductCatalogEngine:
|
||||
return catalog
|
||||
|
||||
def save_catalog(self, catalog: Dict[str, Any], filename: str = None) -> str:
|
||||
"""Save catalog to JSON file inside data/ folder"""
|
||||
"""Save catalog to JSON file inside DATA_DIR.
|
||||
|
||||
Resolved against DATA_DIR rather than a bare relative "data/" path: the
|
||||
latter depends on the process's working directory, so the same call
|
||||
landed in a different place depending on whether the app was started
|
||||
from the repo root, from backend/, or by the container's uvicorn - and
|
||||
only one of those is the directory with a volume mounted on it.
|
||||
"""
|
||||
if not filename:
|
||||
brand = catalog.get('brand', 'unknown')
|
||||
storage_brand = resolve_parent_brand(brand)
|
||||
safe_brand = storage_brand.replace(' ', '_')
|
||||
ts = catalog.get('generation_timestamp', 'latest')
|
||||
filename = f"data/catalog_{safe_brand}_{ts}.json"
|
||||
|
||||
Path("data").mkdir(exist_ok=True)
|
||||
with open(filename, 'w', encoding='utf-8') as f:
|
||||
filename = f"catalog_{safe_brand}_{ts}.json"
|
||||
|
||||
path = Path(filename)
|
||||
if not path.is_absolute():
|
||||
path = DATA_DIR / path
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(path, 'w', encoding='utf-8') as f:
|
||||
json.dump(catalog, f, indent=2, ensure_ascii=False)
|
||||
|
||||
logger.info(f"💾 Catalog saved to: {filename}")
|
||||
return filename
|
||||
|
||||
logger.info(f"💾 Catalog saved to: {path}")
|
||||
return str(path)
|
||||
|
||||
# Global engine instance
|
||||
catalog_engine = ProductCatalogEngine()
|
||||
|
||||
146
app/infrastructure/persistence.py
Normal file
146
app/infrastructure/persistence.py
Normal file
@@ -0,0 +1,146 @@
|
||||
"""
|
||||
First-run restore of the assets bundled into the container image.
|
||||
|
||||
The problem this solves
|
||||
----------------------
|
||||
Three directories are written to at runtime - the seed catalogs appended to by
|
||||
``POST /api/user/products/add``, the generated catalogs from the ingestion
|
||||
pipeline, and the ``*.joblib`` bundles the training endpoints produce. In a
|
||||
container all three sit inside the image, so a redeploy silently discards
|
||||
every one of them. They need to be on a volume.
|
||||
|
||||
Mounting a volume over them introduces the opposite problem. A *named* Docker
|
||||
volume is seeded from the image the first time it is used, but a *bind* mount
|
||||
starts empty and merely hides what the image had underneath. Dokploy offers
|
||||
both and neither announces which you picked, so a bind mount on ``/app/data``
|
||||
would leave the API running with no seed catalogs: the next product added would
|
||||
write a JSON file containing only that product, and every model would report as
|
||||
untrained.
|
||||
|
||||
The fix is to keep a pristine copy inside the image at a path nobody mounts
|
||||
(``BUNDLED_ASSETS_DIR``, populated by the Dockerfile) and top up the writable
|
||||
directory from it on startup.
|
||||
|
||||
Existing files are never overwritten. That is the whole contract: the bundle
|
||||
supplies what is missing, and anything the running app has already written wins
|
||||
over the copy baked into the image. Without that rule every redeploy would
|
||||
revert user-added products back to the bundled catalog.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
from typing import Tuple
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
BUNDLED_ASSETS_DIR,
|
||||
DATA_DIR,
|
||||
MODEL_ARTIFACTS_DIR,
|
||||
SEED_CATALOG_DIR,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# (subdirectory under BUNDLED_ASSETS_DIR, writable destination)
|
||||
_BUNDLES: Tuple[Tuple[str, Path], ...] = (
|
||||
("seed_catalogs", SEED_CATALOG_DIR),
|
||||
("artifacts", MODEL_ARTIFACTS_DIR),
|
||||
)
|
||||
|
||||
|
||||
def _restore_one(source: Path, destination: Path) -> int:
|
||||
"""Copy files missing from ``destination``. Returns how many were copied."""
|
||||
if not source.is_dir():
|
||||
return 0
|
||||
|
||||
destination.mkdir(parents=True, exist_ok=True)
|
||||
copied = 0
|
||||
for item in sorted(source.iterdir()):
|
||||
if not item.is_file():
|
||||
continue
|
||||
target = destination / item.name
|
||||
if target.exists():
|
||||
continue
|
||||
try:
|
||||
# copy2 rather than copy: it preserves mtime, so "is this artifact
|
||||
# older than the data it was trained on" stays answerable.
|
||||
shutil.copy2(item, target)
|
||||
copied += 1
|
||||
except OSError as exc:
|
||||
logger.warning("Could not restore %s -> %s: %s", item, target, exc)
|
||||
return copied
|
||||
|
||||
|
||||
def restore_bundled_assets() -> None:
|
||||
"""
|
||||
Top up the writable directories from the image's read-only bundle.
|
||||
|
||||
Safe to call on every boot: it is a no-op once the volume is populated, and
|
||||
a no-op outside Docker where BUNDLED_ASSETS_DIR does not exist.
|
||||
"""
|
||||
# Create these regardless. A volume mounted at DATA_DIR arrives empty, and
|
||||
# the ingestion pipeline writes into it without creating it first.
|
||||
for path in (DATA_DIR, SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR):
|
||||
try:
|
||||
path.mkdir(parents=True, exist_ok=True)
|
||||
except OSError as exc:
|
||||
logger.error(
|
||||
"Cannot create writable directory %s: %s. Uploads and trained "
|
||||
"models will fail to save - check the volume's permissions.",
|
||||
path,
|
||||
exc,
|
||||
)
|
||||
|
||||
if not BUNDLED_ASSETS_DIR.is_dir():
|
||||
logger.debug(
|
||||
"No bundled asset directory at %s - nothing to restore.", BUNDLED_ASSETS_DIR
|
||||
)
|
||||
return
|
||||
|
||||
for name, destination in _BUNDLES:
|
||||
copied = _restore_one(BUNDLED_ASSETS_DIR / name, destination)
|
||||
if copied:
|
||||
logger.info(
|
||||
"Restored %d bundled file(s) into %s (first run on this volume).",
|
||||
copied,
|
||||
destination,
|
||||
)
|
||||
|
||||
_warn_if_not_persistent()
|
||||
|
||||
|
||||
def _warn_if_not_persistent() -> None:
|
||||
"""
|
||||
Point out that the writable directories are still inside the image.
|
||||
|
||||
Only reachable when BUNDLED_ASSETS_DIR exists, i.e. in the container. If the
|
||||
destinations were never mounted, everything written to them is lost on the
|
||||
next redeploy - which looks exactly like the app quietly ignoring uploads,
|
||||
hours later and with nothing in the logs to connect it to.
|
||||
"""
|
||||
unmounted = [p for p in (SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR) if not _is_mount(p)]
|
||||
if unmounted:
|
||||
logger.warning(
|
||||
"These directories are written at runtime but do not look like "
|
||||
"mount points: %s. Anything saved there - products added through "
|
||||
"the UI, retrained models - will be discarded on the next "
|
||||
"redeploy. Mount a volume on each (see backend/README.md).",
|
||||
", ".join(str(p) for p in unmounted),
|
||||
)
|
||||
|
||||
|
||||
def _is_mount(path: Path) -> bool:
|
||||
"""
|
||||
Whether ``path`` sits on a different device than its parent.
|
||||
|
||||
A mounted volume shows up as a device-number change. Falls back to True on
|
||||
error so a probe failure produces silence rather than a false alarm telling
|
||||
somebody their correctly-mounted volume is broken.
|
||||
"""
|
||||
try:
|
||||
if path.is_mount():
|
||||
return True
|
||||
return path.stat().st_dev != path.parent.stat().st_dev
|
||||
except OSError:
|
||||
return True
|
||||
@@ -55,6 +55,52 @@ def _require(name: str, *, feature_flag: str) -> str:
|
||||
return value
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Writable data directories (persistence)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Everything the running app WRITES lives under one of these three paths. They
|
||||
# are settings rather than hard-coded paths because in a container they must be
|
||||
# mounted on a volume - otherwise every product added through the UI and every
|
||||
# retrained model is discarded the next time the image is redeployed.
|
||||
#
|
||||
# DATA_DIR generated catalogs (catalog_engine.save_catalog)
|
||||
# SEED_CATALOG_DIR per-brand JSON catalogs, appended to by
|
||||
# POST /api/user/products/add and /upload-file
|
||||
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
|
||||
#
|
||||
# See BUNDLED_ASSETS_DIR below for how the read-only copies shipped inside the
|
||||
# image get into these directories the first time a volume is mounted.
|
||||
_BACKEND_ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def _dir(name: str, default: Path) -> Path:
|
||||
raw = os.getenv(name, "").strip()
|
||||
return Path(raw).expanduser() if raw else default
|
||||
|
||||
|
||||
DATA_DIR = _dir("DATA_DIR", _BACKEND_ROOT / "data")
|
||||
SEED_CATALOG_DIR = _dir("SEED_CATALOG_DIR", DATA_DIR / "seed_catalogs")
|
||||
MODEL_ARTIFACTS_DIR = _dir(
|
||||
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
|
||||
)
|
||||
|
||||
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
|
||||
# here by the Dockerfile at a path that is never itself mounted over.
|
||||
#
|
||||
# This exists because the two ways of mounting a volume behave differently, and
|
||||
# the difference is silent. Docker copies the image's content into a *named*
|
||||
# volume the first time it is used, but a *bind* mount starts empty and simply
|
||||
# hides whatever the image had at that path. Mounting a bind mount on /app/data
|
||||
# would therefore leave the app with no seed catalogs at all: the next product
|
||||
# added would write a fresh JSON file containing only that one product, and the
|
||||
# ML endpoints would report no trained models.
|
||||
#
|
||||
# So the image keeps a second, unmounted copy, and restore_bundled_assets()
|
||||
# (app/infrastructure/persistence.py) fills in whatever the writable directory
|
||||
# is missing at startup. Empty/absent outside Docker, where nothing is mounted
|
||||
# and the defaults above already point at the real files.
|
||||
BUNDLED_ASSETS_DIR = _dir("BUNDLED_ASSETS_DIR", Path("/app/.bundled"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ollama (local LLM)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -13,9 +13,15 @@ from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from app.infrastructure.settings import MODEL_ARTIFACTS_DIR
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
ARTIFACTS_DIR = Path(__file__).resolve().parent / "artifacts"
|
||||
# From settings so it can be pointed at a mounted volume: retraining writes
|
||||
# *.joblib here, and inside the image those are discarded on the next redeploy,
|
||||
# silently reverting every model to the version baked in at build time.
|
||||
# Defaults to this package's own artifacts/ directory.
|
||||
ARTIFACTS_DIR = MODEL_ARTIFACTS_DIR
|
||||
ARTIFACTS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
|
||||
|
||||
58
app/main.py
58
app/main.py
@@ -9,6 +9,7 @@ Run with:
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import threading
|
||||
from contextlib import asynccontextmanager
|
||||
from pathlib import Path
|
||||
@@ -18,6 +19,7 @@ from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from fastapi.responses import FileResponse
|
||||
|
||||
from app.infrastructure.persistence import restore_bundled_assets
|
||||
from app.infrastructure.settings import API_CORS_ORIGINS
|
||||
from app.api.routers import health, brands, search, chat, catalog, system
|
||||
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
|
||||
@@ -43,6 +45,15 @@ async def lifespan(_app: FastAPI):
|
||||
healthcheck pass while that settles.
|
||||
"""
|
||||
|
||||
# Runs before the thread below, and synchronously: the seed catalogs and
|
||||
# model artifacts have to be in place before the first request can read
|
||||
# them, and it is a handful of file copies on first boot, nothing on every
|
||||
# boot after that.
|
||||
try:
|
||||
restore_bundled_assets()
|
||||
except Exception as e:
|
||||
logger.warning("Could not restore bundled assets: %s", e)
|
||||
|
||||
def _async_init():
|
||||
try:
|
||||
ensure_store_intelligence_schema()
|
||||
@@ -93,6 +104,24 @@ app.add_middleware(
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
# A wrong origin list fails only in the browser, as an opaque "blocked by CORS"
|
||||
# with a perfectly healthy 200 in the server log - so state the effective list
|
||||
# at startup, where it can actually be compared against the frontend's URL.
|
||||
logger.info(
|
||||
"CORS allowed origins: %s",
|
||||
", ".join(API_CORS_ORIGINS) if API_CORS_ORIGINS else "(none - same-origin only)",
|
||||
)
|
||||
if API_CORS_ORIGINS and all(
|
||||
o.startswith(("http://localhost", "http://127.0.0.1")) for o in API_CORS_ORIGINS
|
||||
):
|
||||
logger.warning(
|
||||
"API_CORS_ORIGINS lists only localhost origins (%s). If this API is "
|
||||
"published on a domain and the frontend is served from a different one, "
|
||||
"every browser request will be blocked. Set API_CORS_ORIGINS to the "
|
||||
"frontend's exact origin, e.g. https://catalogue.nearle.ai.in",
|
||||
", ".join(API_CORS_ORIGINS),
|
||||
)
|
||||
|
||||
app.include_router(health.router, prefix="/api")
|
||||
app.include_router(auth.router, prefix="/api")
|
||||
app.include_router(user_products.router, prefix="/api")
|
||||
@@ -112,9 +141,34 @@ app.include_router(nutrition.router, prefix="/api")
|
||||
app.include_router(nutrition_admin.router, prefix="/api")
|
||||
app.include_router(upload.router, prefix="/api")
|
||||
|
||||
# Serve built frontend static files if dist exists (single-port unified deployment)
|
||||
FRONTEND_DIST = Path(__file__).resolve().parents[2] / "frontend" / "dist"
|
||||
# Serve built frontend static files if dist exists (single-port unified
|
||||
# deployment). Off in the normal setup: the React app is served by its own
|
||||
# nginx on catalogue.nearle.ai.in and this API answers on
|
||||
# mcp.catalogue.nearle.ai.in, so no dist/ is present here and the JSON root
|
||||
# handler at the bottom of this file is what responds to /.
|
||||
#
|
||||
# The candidates cover both repo layouts - the sibling checkout is named
|
||||
# `catalogue_frontend`, and only `frontend` was checked before, so this branch
|
||||
# could never activate even when a build was sitting right next to it.
|
||||
# FRONTEND_DIST_DIR overrides both when the build lands somewhere else.
|
||||
_dist_override = os.getenv("FRONTEND_DIST_DIR", "").strip()
|
||||
_repo_root = Path(__file__).resolve().parents[2]
|
||||
_dist_candidates = (
|
||||
[Path(_dist_override)]
|
||||
if _dist_override
|
||||
else [
|
||||
_repo_root / "catalogue_frontend" / "dist",
|
||||
_repo_root / "frontend" / "dist",
|
||||
Path(__file__).resolve().parents[1] / "frontend" / "dist",
|
||||
]
|
||||
)
|
||||
FRONTEND_DIST = next(
|
||||
(p for p in _dist_candidates if (p / "assets").exists()),
|
||||
_dist_candidates[0],
|
||||
)
|
||||
|
||||
if FRONTEND_DIST.exists() and (FRONTEND_DIST / "assets").exists():
|
||||
logger.info("Serving built frontend from %s", FRONTEND_DIST)
|
||||
app.mount("/assets", StaticFiles(directory=str(FRONTEND_DIST / "assets")), name="assets")
|
||||
|
||||
@app.get("/{full_path:path}")
|
||||
|
||||
@@ -42,6 +42,8 @@ actually resolves to real image bytes above a minimum size - this is what
|
||||
stops garbage/placeholder/expired URLs from reaching the S3 upload step
|
||||
and failing there silently.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional, List
|
||||
import requests
|
||||
from urllib.parse import urlparse
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List, Dict, Any, Optional
|
||||
import json
|
||||
import re
|
||||
|
||||
Reference in New Issue
Block a user