This commit is contained in:
sriram
2026-08-13 16:11:24 +05:30
22 changed files with 1742 additions and 34 deletions

View File

@@ -3,6 +3,15 @@ __pycache__
*.pyc
.pytest_cache
*.log
.env
.git
tests
# .env.production carries this deployment's configuration and the Dockerfile
# copies it to /app/.env inside the image, so it must NOT be ignored here.
# Dokploy's own .env (written from the Environment tab, empty when that tab is
# blank) is ignored instead - it is what silently overwrote the committed one.
#
# The generated sign-in passwords must never be in the image.
.env
.env.local
SIGNIN_PASSWORDS.txt

View File

@@ -16,6 +16,64 @@
# localhost URL points the backend at its own empty ports. Containers reach
# each other by service name over the compose network instead.
# --- Ports (container only) -----------------------------------------------
# The image serves 3000 and 8000 at once - 3000 because that is what Dokploy
# routes a domain to, 8000 because the README, the vite dev proxy and
# docker-compose all use it. Serving both means the container works whichever
# one the platform points at.
#
# PORTS the comma-separated pair to bind. PORT pins a single port instead and
# takes precedence, e.g. PORT=8080 serves only 8080.
#
# Neither affects running uvicorn directly for local development.
# PORTS=3000,8000
# PORT=8080
# --- Persistence (IMPORTANT in Docker) ------------------------------------
# The three directories the app WRITES to at runtime. The defaults point inside
# the repo/image and are right for local development; in a container each one
# needs a volume mounted on it, or a redeploy throws away everything written
# since the last build:
#
# SEED_CATALOG_DIR products added via POST /api/user/products/add and
# /upload-file are appended to the JSON files here
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
# DATA_DIR catalogs saved by the ingestion pipeline
#
# Mount these two paths in Dokploy (SEED_CATALOG_DIR sits inside DATA_DIR, so
# one mount covers both):
#
# /app/data
# /app/app/intelligence/artifacts
#
# Either a named volume or a bind mount works. The image keeps read-only copies
# of the bundled seed catalogs and pre-trained models at /app/.bundled, and the
# app restores whatever a freshly-mounted directory is missing on startup
# without overwriting anything already there - so a bind mount, which starts
# empty and would otherwise hide them, is safe.
#
# Leave all three unset unless the writable data belongs somewhere else.
# DATA_DIR=/app/data
# SEED_CATALOG_DIR=/app/data/seed_catalogs
# MODEL_ARTIFACTS_DIR=/app/app/intelligence/artifacts
# --- CORS (REQUIRED when the API is on its own domain) --------------------
# Comma-separated list of the exact browser origins allowed to call this API.
# Scheme and host both matter; no trailing slash, and no wildcard - the
# frontend sends an Authorization header, and browsers refuse a credentialed
# cross-origin request whose Allow-Origin is "*" (app/main.py logs and turns
# credentials off if it sees one, so a wildcard silently breaks every call).
#
# Production - the React app is served from catalogue.nearle.ai.in and calls
# the API at mcp.nearle.ai.in, so that frontend origin must be listed:
#
# API_CORS_ORIGINS=https://catalogue.nearle.ai.in
#
# Add http://localhost:5173 alongside it if you point a local Vite dev server
# at the deployed API. Server-to-server callers (MCP clients, scripts) are not
# affected by any of this - CORS is a browser rule; they use X-API-Key.
API_CORS_ORIGINS=http://localhost:5173,http://127.0.0.1:5173
# --- Authentication (REQUIRED) -------------------------------------------
# The backend will not start without these while AUTH_ENABLED=true. Generate
# all four lines, plus sign-in passwords, with:

113
.env.production Normal file
View File

@@ -0,0 +1,113 @@
# Deployment configuration for mcp.nearle.ai.in.
#
# Committed at the repo owner's instruction so the deploy does not depend on
# re-entering config in the Dokploy UI. Everything needed to boot is here; no
# environment variables are required in Dokploy any more.
#
# A real environment variable still overrides anything set here - settings.py
# calls load_dotenv() without override=True, so the process environment wins.
# That is the escape hatch for changing a value without a commit.
#
# WHAT IS IN THIS FILE: live database, S3 and Google credentials, and the key
# that signs every access token. Anyone with read access to this repository has
# all of it, and git history keeps it after any rotation.
# --- Ports -----------------------------------------------------------------
# Dokploy routes the domain to 3000; 8000 is kept for the vite dev proxy and
# docker-compose. serve.py binds both.
PORTS=3000,8000
# --- CORS ------------------------------------------------------------------
# The FRONTEND's origin, not this API's. Wrong value = the browser blocks every
# response while the server logs healthy 200s.
API_CORS_ORIGINS=https://catalogue.nearle.ai.in
# --- Authentication --------------------------------------------------------
AUTH_ENABLED=true
# Freshly generated for this deployment - deliberately NOT the values from the
# development .env. Those hashes are for the passwords DevAdmin!2026 and
# DevUser!2026, which are sitting in plaintext in test_login_fix.py in this very
# repository: shipping them would publish working admin credentials.
# Sign-in passwords for the hashes below are in SIGNIN_PASSWORDS.txt (ignored).
AUTH_SECRET_KEY=4Kmyr4Cjf_kdUIq_4EGxo5vFHfCT5_uKVR3eouszB8Le6F0n45m7eDY94_KJoqSz
AUTH_ADMIN_USERNAME=admin
AUTH_ADMIN_PASSWORD_HASH=pbkdf2_sha256$600000$Xa07unPO4LeTU4bz04eh7Q==$KmZ2ZBrclJ0z0sDiCoKOrfTO2UK8e7hjsZZ6HB2PY9o=
AUTH_USER_USERNAME=user
AUTH_USER_PASSWORD_HASH=pbkdf2_sha256$600000$65VMpqwUSyFzCqnhlwBqgQ==$sMVnar+Hnp5ZmXcObFNI3jJMxqnVrW8naLZNFFh0KCw=
AUTH_TOKEN_TTL_MINUTES=720
AUTH_MAX_LOGIN_ATTEMPTS=10
AUTH_LOCKOUT_SECONDS=300
# MUST stay false here. The development .env has this true, where it is a
# convenience: it skips the password check entirely, so any username signs in
# and `admin` gets the admin pages. On a host published to the internet it means
# anyone who finds mcp.nearle.ai.in signs in as admin by typing anything at all.
AUTH_ALLOW_ANY_LOGIN=false
# Machine consumers. Empty: MCP clients authenticate with a login token instead.
API_KEYS=
# --- Postgres / pgvector ---------------------------------------------------
# DB_NAME is not set in the development .env, so it falls back to settings.py's
# default. Stated explicitly here so the deployment does not depend on that
# default staying the same.
USE_PGVECTOR=true
DB_HOST=31.97.228.132
DB_PORT=6054
DB_NAME=pgvector
DB_USER=admin
DB_PASSWORD=Package@321#
# --- Embeddings ------------------------------------------------------------
USE_EMBEDDINGS=true
EMBEDDINGS_MODEL=sentence-transformers/all-MiniLM-L6-v2
EMBEDDINGS_DIM=384
# --- Ollama (local LLM, powers /api/chat) ----------------------------------
# Off, because the development value (http://localhost:11434) cannot work from
# inside a container: there, localhost is the container itself, not the VPS
# host. Left on with nothing listening, /api/chat fails AND every healthcheck
# takes ~3s longer, because the health handler probes Ollama with a 3s timeout.
#
# To enable: set USE_OLLAMA=true and point OLLAMA_BASE_URL at something the
# container can actually reach - http://host.docker.internal:11434 with a
# host-gateway mapping, the VPS's LAN IP, or an ollama service name.
USE_OLLAMA=false
OLLAMA_BASE_URL=http://host.docker.internal:11434
OLLAMA_MODEL_NAME=qwen2.5:1.5b
OLLAMA_TIMEOUT_SECONDS=120
# --- DigitalOcean Spaces (product image storage) ---------------------------
USE_S3=true
S3_ACCESS_KEY=DO801G8Q8JAZKF49U3WJ
S3_SECRET_KEY=lBQExYfkVqH+ybmGVmQH5MkThBbrIohA/VQLgcPUvug
S3_ENDPOINT=https://nearle.sgp1.digitaloceanspaces.com
S3_BUCKET=nearle
S3_REGION=sgp1
# --- Google Custom Search (optional image source) --------------------------
USE_GOOGLE_CSE=true
GOOGLE_API_KEY=AIzaSyBY4pIO_Fp5FCMqeVxDNcfalzdWNHJWVn0
GOOGLE_CSE_ID=9745cbd96dd164562
# --- Open-source image sources (no key needed) -----------------------------
USE_DDG_IMAGES=true
USE_OPEN_FACTS=true
USE_WIKIMEDIA=true
# The Playwright browser binary is NOT installed in the image (see Dockerfile),
# so this tier is skipped at runtime regardless. false stops it being attempted.
USE_PLAYWRIGHT_FALLBACK=false
MIN_IMAGE_BYTES=3000
# --- Product validation ----------------------------------------------------
ENABLE_PRODUCT_VALIDATION=true
VALIDATION_REJECT_THRESHOLD=0.35
VALIDATION_REVIEW_THRESHOLD=0.70
# --- RAG -------------------------------------------------------------------
RAG_DEFAULT_TOP_K=5
RAG_MAX_TOP_K=15
RAG_MAX_CONTEXT_CHARS=4000

22
.gitignore vendored
View File

@@ -1,6 +1,26 @@
# Secrets - never commit
# .env.production is committed deliberately, at the repo owner's instruction, so
# the deployment does not depend on re-entering config in the Dokploy UI.
#
# It is NOT named .env, and that matters: Dokploy writes its own .env into the
# build context from the service's Environment tab after cloning, so a committed
# .env is silently replaced (with an empty file when that tab is blank).
# The Dockerfile copies .env.production to /app/.env inside the image.
#
# The two database secrets are NOT in it - they are set as Dokploy environment
# variables, which override the file (settings.py calls load_dotenv() without
# override=True, so the process environment wins).
#
# The auth secrets ARE in it. AUTH_SECRET_KEY signs every access token, so
# anyone with read access to this repository can mint an admin token, and git
# history keeps it after any rotation. Regenerate with
# `python scripts/make_auth_secrets.py` if that stops being acceptable.
!.env.production
.env
# Local overrides and the generated sign-in passwords stay out of git.
.env.local
SIGNIN_PASSWORDS.txt
# Python
__pycache__/
*.pyc

View File

@@ -1,25 +1,116 @@
FROM python:3.11-slim
# Multi-stage, mirroring catalogue_frontend/Dockerfile: a build stage that
# resolves dependencies, then a clean runtime stage that copies in only the
# result. There it is `npm ci` -> dist/; here it is pip -> a virtualenv.
# ---- Build stage ----
FROM python:3.11-slim AS build
WORKDIR /app
# Dependencies land in a self-contained venv so the runtime stage can take that
# one directory and leave pip, its HTTP cache and the downloaded wheels behind.
#
# requirements.txt is copied on its own, ahead of the source, for the same
# reason the frontend stage copies package.json before the rest of the app:
# this layer is cached on the file's checksum, so editing a router does not
# reinstall torch.
#
# psycopg[binary] avoids needing libpq-dev; sentence-transformers/scikit-learn/
# scipy all ship prebuilt wheels for this image, so no extra build toolchain
# is needed. Playwright's Python package installs, but its browser binary is
# NOT installed here - it's only a last-resort image-search fallback
# (see requirements.txt); run `playwright install chromium` in the container
# if you need that specific fallback tier.
# scipy all ship prebuilt wheels for this image, so no compiler is needed and
# this stage installs no build toolchain. Playwright's Python package installs,
# but its browser binary is NOT installed here - it's only a last-resort
# image-search fallback (see requirements.txt); run `playwright install
# chromium` in the container if you need that specific fallback tier.
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
RUN python -m venv /opt/venv \
&& /opt/venv/bin/pip install --no-cache-dir -r requirements.txt
# ---- Runtime stage ----
FROM python:3.11-slim AS runtime
WORKDIR /app
# PATH: putting the venv first is what makes a bare `python`/`uvicorn` resolve
# to it - there is no "activate" step in a container.
# PYTHONUNBUFFERED: without it Dokploy's log view stays empty until a buffer
# happens to fill, so startup errors surface minutes after the container died.
# PYTHONDONTWRITEBYTECODE: no .pyc to write into a read-only-ish image layer.
ENV PATH="/opt/venv/bin:$PATH"
ENV PYTHONUNBUFFERED=1
ENV PYTHONDONTWRITEBYTECODE=1
COPY --from=build /opt/venv /opt/venv
COPY app ./app
COPY cli ./cli
COPY scripts ./scripts
COPY data ./data
COPY serve.py .
EXPOSE 8000
# --proxy-headers/--forwarded-allow-ips: this container is never reached
# directly - nginx (and, in prod, Caddy) sit in front of it. Without these,
# uvicorn ignores X-Forwarded-Proto and reports every request as plain http,
# so any redirect or generated absolute URL would downgrade an https request.
CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000", \
"--proxy-headers", "--forwarded-allow-ips", "*"]
# The deployment's configuration, landing at /app/.env because settings.py
# resolves it from the backend root - app/infrastructure/settings.py takes
# parents[2], which is /app here - so it must sit next to app/, not inside it.
#
# The source file is named .env.production, NOT .env, and that detail is the
# whole point. Dokploy writes its own .env into the build context from the
# service's Environment tab AFTER cloning the repository. With that tab empty it
# writes an empty file, overwriting the committed one - so `COPY .env .` copied
# a zero-byte file, the container started with no configuration at all, and the
# platform reported only a Bad Gateway. The checkout showed it plainly: every
# file timestamped 08:33, and .env alone at 08:34, 0 bytes.
#
# Dokploy does not manage .env.production, so it survives. Anything set in the
# Environment tab still wins at runtime, because settings.py calls load_dotenv()
# without override=True and the process environment takes precedence.
COPY .env.production .env
# Pristine copies of everything the app also WRITES to, kept at a path that is
# never mounted over.
#
# /app/data/seed_catalogs and /app/app/intelligence/artifacts both need volumes
# (products added through the UI are appended to the first, retrained models
# are written to the second - otherwise a redeploy throws both away). But
# mounting a volume there hides the copies shipped in this image: a *named*
# volume is seeded from the image on first use, a *bind* mount is not, and
# Dokploy offers both. A bind mount would leave the API with zero seed catalogs
# and zero trained models, with nothing in the logs saying why.
#
# So keep a second copy here. On startup restore_bundled_assets()
# (app/infrastructure/persistence.py) copies in whatever the mounted directory
# is missing, and never overwrites what is already there.
RUN mkdir -p /app/.bundled \
&& cp -a /app/data/seed_catalogs /app/.bundled/seed_catalogs \
&& cp -a /app/app/intelligence/artifacts /app/.bundled/artifacts
# Declared so `docker run` without an explicit -v still gets an anonymous
# volume rather than writing into the container layer. Dokploy (and the compose
# file) name them properly; this is the floor, not the recommended setup.
VOLUME ["/app/data", "/app/app/intelligence/artifacts"]
# Answers on BOTH ports, the same way the frontend image does (nginx.conf has
# `listen 80; listen 3000;`). 3000 is what Dokploy routes a domain to; 8000 is
# what this project's README, the vite dev proxy and docker-compose all use.
# Serving both means the container works whichever one the platform is pointed
# at, instead of returning 502 from a perfectly healthy process.
#
# serve.py binds both sockets and hands them to one uvicorn - see the note
# there. To pin a single port, set PORT (PORT=8080 serves only 8080); to change
# the pair, set PORTS.
ENV PORTS=3000,8000
EXPOSE 3000 8000
# Liveness only, and passes if EITHER port answers.
#
# It probes "/", which is served from memory, NOT /api/health, which dials
# Postgres and Ollama. That is the whole point: the platform's response to a
# failed healthcheck is to stop routing traffic, so this may only ask "is the
# process still serving HTTP". Tying it to the database meant an unreachable
# Postgres blocked the handler for the OS TCP timeout, the check timed out, the
# container was marked unhealthy, and a perfectly healthy API returned Bad
# Gateway on every route. Use /api/health to ask whether dependencies are up;
# it reports them in the body and always answers 200.
HEALTHCHECK --interval=30s --timeout=10s --start-period=40s --retries=3 \
CMD ["python", "serve.py", "--healthcheck"]
# Exec form: python is PID 1, so Docker's SIGTERM reaches it directly and a
# redeploy shuts down cleanly instead of waiting out the 10s kill timeout.
CMD ["python", "serve.py"]

View File

@@ -69,6 +69,105 @@ permission check; `user` holds the product/store/inventory permissions.
`AUTH_ENABLED=false` disables all of it for local work — never in a deployment.
See the Authentication section of `../DEPLOYMENT.md` for the full endpoint map.
## MCP server
The catalog is exposed to AI clients over the Model Context Protocol at `/mcp`,
mounted onto this same FastAPI app (`app/mcp_server.py`) - no separate process
or container, so it deploys with the API and answers on the same host.
**15 tools, all read-only.** Catalog search and browse, nutrition facts and
health scores, healthier alternatives, per-store inventory and pricing,
discounts, trending and sales analytics. None of the write or compute endpoints
are reachable through MCP: a tool list is chosen from by a model rather than by
a person, and "retrain the models" is not something to leave one tool call away.
**Auth is the app's own access token** - there is no separate MCP credential.
Clients send `Authorization: Bearer <token>` from `POST /api/auth/login`, and
the same `Principal` and expiry apply as on the REST side. Note the practical
consequence: tokens expire after `AUTH_TOKEN_TTL_MINUTES` (12h by default), so a
long-running client has to refresh. Raise the TTL if that is a problem.
```jsonc
{
"mcpServers": {
"nearle-catalogue": {
"type": "http",
"url": "https://mcp.nearle.ai.in/mcp",
"headers": { "Authorization": "Bearer <token>" }
}
}
}
```
The admin UI has an inspector at `/mcp` (React route, admin-only) listing every
tool with its arguments and a console to run one against live data. It is backed
by `GET /api/mcp/info` and `POST /api/mcp/tools/{name}` rather than by the MCP
endpoint itself - the browser would otherwise need a full MCP client to render a
tool list.
`fastmcp` requires Python 3.10+. The image is 3.11; on an older interpreter the
import fails, `app/main.py` logs a warning, and the REST API starts without the
MCP endpoint rather than failing outright.
## Ports
The container answers on **3000 and 8000 at the same time**, the same way the
frontend image answers on 80 and 3000. 3000 is what Dokploy routes a domain to;
8000 is what this README, the vite dev proxy and `docker-compose.yml` use. Both
being live means the deployment works whichever one it is pointed at, instead of
returning 502 from a healthy container.
`serve.py` is what makes that possible - uvicorn's CLI binds a single `--port`,
but `Server.run()` accepts a list of pre-bound sockets, so it is still one
process. If one port is unavailable it logs and carries on with the other; it
exits non-zero only when nothing is listening.
```bash
python serve.py # 3000 and 8000
PORT=8080 python serve.py # only 8080 (PORT pins a single port)
PORTS=80,3000 python serve.py # a different pair
```
For local development `uvicorn app.main:app --reload --port 8000` is still the
normal thing to run - one port is all you need, and it gives you autoreload.
## Persistence: the two volumes a deployment needs
Most state lives in Postgres, but three things are written to the filesystem,
and in a container those live inside the image - so a redeploy rebuilds the
image and silently discards them:
| Path | Written by |
|---|---|
| `/app/data/seed_catalogs` | `POST /api/user/products/add`, `/upload-file` - every product added through the UI is appended to the brand's JSON |
| `/app/app/intelligence/artifacts` | the training endpoints - every retrained `*.joblib` model |
| `/app/data` | catalogs saved by the ingestion pipeline |
Mount a volume on each (the first is inside the third, so two mounts cover all
three):
```
/app/data
/app/app/intelligence/artifacts
```
In Dokploy, add both under the service's **Volumes**. Named volume or bind
mount, either is fine - `docker compose --profile full up -d` shows the same
two mounts as named volumes.
Bind mounts normally break this pattern, because they start empty and hide the
seed catalogs and pre-trained models the image ships with. They are safe here:
the image keeps read-only copies at `/app/.bundled`, and on startup
`app/infrastructure/persistence.py` copies in whatever the mounted directory is
missing. It never overwrites an existing file, so a user-added product always
survives the next redeploy rather than being reverted to the bundled catalog.
If you leave the volumes off, the API still runs and logs a warning naming the
directories that will be lost.
Override the locations with `DATA_DIR`, `SEED_CATALOG_DIR` and
`MODEL_ARTIFACTS_DIR` if the writable data belongs somewhere else.
## Pulling the local LLM (one-time)
```bash

167
app/api/routers/mcp_info.py Normal file
View File

@@ -0,0 +1,167 @@
"""
REST inspector for the MCP server.
The MCP endpoint itself speaks streamable HTTP with session handling, which a
browser cannot usefully talk to without a full MCP client implementation. Rather
than ship one in React, these two endpoints expose what the admin page needs -
what tools exist, and what one returns - as ordinary JSON over the REST API the
frontend already authenticates against.
Admin-only. The tool inventory describes the whole catalog API surface, and the
invoke endpoint executes a tool; neither is something to leave open just because
the tools happen to be read-only.
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List, Optional
from fastapi import APIRouter, Body, Depends, HTTPException, Request
from app.api.deps import require_admin
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/mcp", tags=["mcp"])
def _server():
"""The FastMCP instance, or None when the optional dependency is absent."""
try:
from app.mcp_server import mcp
return mcp
except Exception: # pragma: no cover - depends on fastmcp being installed
return None
def _public_url(request: Request, path: str) -> str:
"""
The URL an external MCP client should connect to.
Built from the forwarded headers rather than a configured constant so it is
right on whichever host the request arrived on - the API's own domain, the
frontend's nginx, or localhost in development. uvicorn runs with
--proxy-headers, so request.url already reflects the external scheme.
"""
base = str(request.base_url).rstrip("/")
return f"{base}{path}"
@router.get("/info", dependencies=[Depends(require_admin)])
async def mcp_info(request: Request) -> Dict[str, Any]:
"""Describe the MCP server and list its tools, for the admin page."""
mcp = _server()
if mcp is None:
return {
"enabled": False,
"reason": (
"The fastmcp package is not installed, or failed to import. It "
"requires Python 3.10 or newer."
),
"tools": [],
}
from app.mcp_server import MCP_PATH
try:
# run_middleware=False: the auth middleware reads an Authorization header
# from an MCP request context that does not exist on this REST call. The
# caller is already established as an admin by the route dependency.
tools = await mcp.list_tools(run_middleware=False)
except Exception as exc: # noqa: BLE001
logger.exception("Could not list MCP tools")
raise HTTPException(status_code=500, detail=f"Could not list MCP tools: {exc}")
described: List[Dict[str, Any]] = []
for tool in sorted(tools, key=lambda t: t.name):
# Server-side FunctionTool exposes its JSON Schema as `parameters`.
# `inputSchema` is the wire-format name, present on the client-side Tool
# an MCP client receives - checked second so this keeps working if a
# future version converges on one name.
schema = getattr(tool, "parameters", None) or getattr(tool, "inputSchema", None) or {}
properties = schema.get("properties", {}) or {}
required = set(schema.get("required", []) or [])
described.append(
{
"name": tool.name,
"description": (tool.description or "").strip(),
"tags": sorted(getattr(tool, "tags", set()) or []),
"parameters": [
{
"name": pname,
"type": pinfo.get("type") or "any",
"description": (pinfo.get("description") or "").strip(),
"required": pname in required,
"default": pinfo.get("default"),
}
for pname, pinfo in properties.items()
],
}
)
return {
"enabled": True,
"name": mcp.name,
"version": getattr(mcp, "version", None),
"url": _public_url(request, MCP_PATH),
"transport": "http",
"auth": "Bearer token from POST /api/auth/login",
"tool_count": len(described),
"tools": described,
}
@router.post("/tools/{tool_name}", dependencies=[Depends(require_admin)])
async def invoke_tool(
tool_name: str,
arguments: Optional[Dict[str, Any]] = Body(default=None),
) -> Dict[str, Any]:
"""
Run one MCP tool and return its result - the admin page's "try it" console.
Exists so the server can be checked from the browser without installing an
MCP client. Every tool is read-only, so this executes the same call an AI
client would make, with the same code path.
"""
mcp = _server()
if mcp is None:
raise HTTPException(status_code=503, detail="The MCP server is not available.")
try:
# Same reasoning as above: this request authenticated through the REST
# guard, so the MCP auth middleware would have no headers to inspect.
tools = {t.name: t for t in await mcp.list_tools(run_middleware=False)}
except Exception as exc: # noqa: BLE001
raise HTTPException(status_code=500, detail=f"Could not list MCP tools: {exc}")
if tool_name not in tools:
raise HTTPException(
status_code=404,
detail=f"No MCP tool named '{tool_name}'. Known tools: {', '.join(sorted(tools))}.",
)
try:
result = await mcp.call_tool(tool_name, arguments or {}, run_middleware=False)
except Exception as exc: # noqa: BLE001
# A tool raising ToolError is a normal outcome worth showing verbatim -
# "no nutrition data for this product" is an answer, not a server fault.
return {"tool": tool_name, "ok": False, "error": str(exc)}
return {"tool": tool_name, "ok": True, "result": _jsonable(result)}
def _jsonable(result: Any) -> Any:
"""Reduce a ToolResult to something FastAPI can serialise."""
for attr in ("structured_content", "data"):
value = getattr(result, attr, None)
if value is not None:
return value
content = getattr(result, "content", None)
if content is None:
return result
out = []
for block in content:
text = getattr(block, "text", None)
out.append(text if text is not None else str(block))
return out

View File

@@ -1,6 +1,7 @@
from __future__ import annotations
import io
import json
import uuid
import logging
import pandas as pd

View File

@@ -7,6 +7,7 @@ from pydantic import BaseModel, Field
from fastapi import APIRouter, Depends, HTTPException, File, UploadFile
from app.api.deps import require_permission
from app.infrastructure.settings import SEED_CATALOG_DIR
from app.services.vector_store import (
upsert_brand_products,

View File

@@ -19,7 +19,7 @@ sys.path.append(str(Path(__file__).parent.parent))
from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details
from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts
from app.infrastructure.settings import USE_OLLAMA
from app.infrastructure.settings import DATA_DIR, USE_OLLAMA
from app.services.embeddings_service import embed_texts
from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id
from app.services.s3_service import s3_service
@@ -948,20 +948,30 @@ class ProductCatalogEngine:
return catalog
def save_catalog(self, catalog: Dict[str, Any], filename: str = None) -> str:
"""Save catalog to JSON file inside data/ folder"""
"""Save catalog to JSON file inside DATA_DIR.
Resolved against DATA_DIR rather than a bare relative "data/" path: the
latter depends on the process's working directory, so the same call
landed in a different place depending on whether the app was started
from the repo root, from backend/, or by the container's uvicorn - and
only one of those is the directory with a volume mounted on it.
"""
if not filename:
brand = catalog.get('brand', 'unknown')
storage_brand = resolve_parent_brand(brand)
safe_brand = storage_brand.replace(' ', '_')
ts = catalog.get('generation_timestamp', 'latest')
filename = f"data/catalog_{safe_brand}_{ts}.json"
Path("data").mkdir(exist_ok=True)
with open(filename, 'w', encoding='utf-8') as f:
filename = f"catalog_{safe_brand}_{ts}.json"
path = Path(filename)
if not path.is_absolute():
path = DATA_DIR / path
path.parent.mkdir(parents=True, exist_ok=True)
with open(path, 'w', encoding='utf-8') as f:
json.dump(catalog, f, indent=2, ensure_ascii=False)
logger.info(f"💾 Catalog saved to: {filename}")
return filename
logger.info(f"💾 Catalog saved to: {path}")
return str(path)
# Global engine instance
catalog_engine = ProductCatalogEngine()

View File

@@ -0,0 +1,146 @@
"""
First-run restore of the assets bundled into the container image.
The problem this solves
----------------------
Three directories are written to at runtime - the seed catalogs appended to by
``POST /api/user/products/add``, the generated catalogs from the ingestion
pipeline, and the ``*.joblib`` bundles the training endpoints produce. In a
container all three sit inside the image, so a redeploy silently discards
every one of them. They need to be on a volume.
Mounting a volume over them introduces the opposite problem. A *named* Docker
volume is seeded from the image the first time it is used, but a *bind* mount
starts empty and merely hides what the image had underneath. Dokploy offers
both and neither announces which you picked, so a bind mount on ``/app/data``
would leave the API running with no seed catalogs: the next product added would
write a JSON file containing only that product, and every model would report as
untrained.
The fix is to keep a pristine copy inside the image at a path nobody mounts
(``BUNDLED_ASSETS_DIR``, populated by the Dockerfile) and top up the writable
directory from it on startup.
Existing files are never overwritten. That is the whole contract: the bundle
supplies what is missing, and anything the running app has already written wins
over the copy baked into the image. Without that rule every redeploy would
revert user-added products back to the bundled catalog.
"""
from __future__ import annotations
import logging
import shutil
from pathlib import Path
from typing import Tuple
from app.infrastructure.settings import (
BUNDLED_ASSETS_DIR,
DATA_DIR,
MODEL_ARTIFACTS_DIR,
SEED_CATALOG_DIR,
)
logger = logging.getLogger(__name__)
# (subdirectory under BUNDLED_ASSETS_DIR, writable destination)
_BUNDLES: Tuple[Tuple[str, Path], ...] = (
("seed_catalogs", SEED_CATALOG_DIR),
("artifacts", MODEL_ARTIFACTS_DIR),
)
def _restore_one(source: Path, destination: Path) -> int:
"""Copy files missing from ``destination``. Returns how many were copied."""
if not source.is_dir():
return 0
destination.mkdir(parents=True, exist_ok=True)
copied = 0
for item in sorted(source.iterdir()):
if not item.is_file():
continue
target = destination / item.name
if target.exists():
continue
try:
# copy2 rather than copy: it preserves mtime, so "is this artifact
# older than the data it was trained on" stays answerable.
shutil.copy2(item, target)
copied += 1
except OSError as exc:
logger.warning("Could not restore %s -> %s: %s", item, target, exc)
return copied
def restore_bundled_assets() -> None:
"""
Top up the writable directories from the image's read-only bundle.
Safe to call on every boot: it is a no-op once the volume is populated, and
a no-op outside Docker where BUNDLED_ASSETS_DIR does not exist.
"""
# Create these regardless. A volume mounted at DATA_DIR arrives empty, and
# the ingestion pipeline writes into it without creating it first.
for path in (DATA_DIR, SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR):
try:
path.mkdir(parents=True, exist_ok=True)
except OSError as exc:
logger.error(
"Cannot create writable directory %s: %s. Uploads and trained "
"models will fail to save - check the volume's permissions.",
path,
exc,
)
if not BUNDLED_ASSETS_DIR.is_dir():
logger.debug(
"No bundled asset directory at %s - nothing to restore.", BUNDLED_ASSETS_DIR
)
return
for name, destination in _BUNDLES:
copied = _restore_one(BUNDLED_ASSETS_DIR / name, destination)
if copied:
logger.info(
"Restored %d bundled file(s) into %s (first run on this volume).",
copied,
destination,
)
_warn_if_not_persistent()
def _warn_if_not_persistent() -> None:
"""
Point out that the writable directories are still inside the image.
Only reachable when BUNDLED_ASSETS_DIR exists, i.e. in the container. If the
destinations were never mounted, everything written to them is lost on the
next redeploy - which looks exactly like the app quietly ignoring uploads,
hours later and with nothing in the logs to connect it to.
"""
unmounted = [p for p in (SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR) if not _is_mount(p)]
if unmounted:
logger.warning(
"These directories are written at runtime but do not look like "
"mount points: %s. Anything saved there - products added through "
"the UI, retrained models - will be discarded on the next "
"redeploy. Mount a volume on each (see backend/README.md).",
", ".join(str(p) for p in unmounted),
)
def _is_mount(path: Path) -> bool:
"""
Whether ``path`` sits on a different device than its parent.
A mounted volume shows up as a device-number change. Falls back to True on
error so a probe failure produces silence rather than a false alarm telling
somebody their correctly-mounted volume is broken.
"""
try:
if path.is_mount():
return True
return path.stat().st_dev != path.parent.stat().st_dev
except OSError:
return True

View File

@@ -55,6 +55,52 @@ def _require(name: str, *, feature_flag: str) -> str:
return value
# ---------------------------------------------------------------------------
# Writable data directories (persistence)
# ---------------------------------------------------------------------------
# Everything the running app WRITES lives under one of these three paths. They
# are settings rather than hard-coded paths because in a container they must be
# mounted on a volume - otherwise every product added through the UI and every
# retrained model is discarded the next time the image is redeployed.
#
# DATA_DIR generated catalogs (catalog_engine.save_catalog)
# SEED_CATALOG_DIR per-brand JSON catalogs, appended to by
# POST /api/user/products/add and /upload-file
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
#
# See BUNDLED_ASSETS_DIR below for how the read-only copies shipped inside the
# image get into these directories the first time a volume is mounted.
_BACKEND_ROOT = Path(__file__).resolve().parents[2]
def _dir(name: str, default: Path) -> Path:
raw = os.getenv(name, "").strip()
return Path(raw).expanduser() if raw else default
DATA_DIR = _dir("DATA_DIR", _BACKEND_ROOT / "data")
SEED_CATALOG_DIR = _dir("SEED_CATALOG_DIR", DATA_DIR / "seed_catalogs")
MODEL_ARTIFACTS_DIR = _dir(
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
)
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
# here by the Dockerfile at a path that is never itself mounted over.
#
# This exists because the two ways of mounting a volume behave differently, and
# the difference is silent. Docker copies the image's content into a *named*
# volume the first time it is used, but a *bind* mount starts empty and simply
# hides whatever the image had at that path. Mounting a bind mount on /app/data
# would therefore leave the app with no seed catalogs at all: the next product
# added would write a fresh JSON file containing only that one product, and the
# ML endpoints would report no trained models.
#
# So the image keeps a second, unmounted copy, and restore_bundled_assets()
# (app/infrastructure/persistence.py) fills in whatever the writable directory
# is missing at startup. Empty/absent outside Docker, where nothing is mounted
# and the defaults above already point at the real files.
BUNDLED_ASSETS_DIR = _dir("BUNDLED_ASSETS_DIR", Path("/app/.bundled"))
# ---------------------------------------------------------------------------
# Ollama (local LLM)
# ---------------------------------------------------------------------------
@@ -81,6 +127,14 @@ DB_PORT = os.getenv("DB_PORT", "5432")
DB_NAME = os.getenv("DB_NAME", "pgvector")
DB_USER = os.getenv("DB_USER", "postgres")
DB_PASSWORD = _require("DB_PASSWORD", feature_flag="USE_PGVECTOR") if USE_PGVECTOR else os.getenv("DB_PASSWORD", "")
# How long to wait for the TCP connect before giving up. Matters more than it
# looks: a host that DROPS packets (a firewall, a typo'd DB_HOST) otherwise
# blocks until the OS timeout - about 130 seconds on Linux - and every request
# that touches the database inherits that wait, including /api/health. A short
# ceiling turns "the database is unreachable" into a fast, honest error instead
# of a hung worker and a container the platform decides is unhealthy.
DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5"))
DATABASE_URL = os.getenv(
"DATABASE_URL",
f"postgresql://{DB_USER}:{DB_PASSWORD}@{DB_HOST}:{DB_PORT}/{DB_NAME}",

View File

@@ -13,9 +13,15 @@ from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional
from app.infrastructure.settings import MODEL_ARTIFACTS_DIR
logger = logging.getLogger(__name__)
ARTIFACTS_DIR = Path(__file__).resolve().parent / "artifacts"
# From settings so it can be pointed at a mounted volume: retraining writes
# *.joblib here, and inside the image those are discarded on the next redeploy,
# silently reverting every model to the version baked in at build time.
# Defaults to this package's own artifacts/ directory.
ARTIFACTS_DIR = MODEL_ARTIFACTS_DIR
ARTIFACTS_DIR.mkdir(parents=True, exist_ok=True)

View File

@@ -9,6 +9,7 @@ Run with:
from __future__ import annotations
import logging
import os
import threading
from contextlib import asynccontextmanager
from pathlib import Path
@@ -18,11 +19,12 @@ from fastapi.middleware.cors import CORSMiddleware
from fastapi.staticfiles import StaticFiles
from fastapi.responses import FileResponse
from app.infrastructure.persistence import restore_bundled_assets
from app.infrastructure.settings import API_CORS_ORIGINS
from app.api.routers import health, brands, search, chat, catalog, system
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
from app.api.routers import nutrition, nutrition_admin, upload
from app.api.routers import auth, user_products, admin_train
from app.api.routers import auth, user_products, admin_train, mcp_info
from app.services.store_db import ensure_store_intelligence_schema
from app.services.nutrition_db import ensure_nutrition_schema
@@ -32,6 +34,18 @@ logging.basicConfig(
)
logger = logging.getLogger(__name__)
try:
from app.mcp_server import MCP_PATH, build_http_app as build_mcp_app
_mcp_app = build_mcp_app()
except Exception as e: # pragma: no cover - depends on an optional dependency
# fastmcp needs Python >=3.10. The image is 3.11, but an older local
# interpreter should lose the MCP endpoint, not the whole API.
logging.getLogger(__name__).warning("MCP server unavailable: %s", e)
_mcp_app = None
MCP_PATH = "/mcp"
@asynccontextmanager
async def lifespan(_app: FastAPI):
"""
@@ -43,6 +57,15 @@ async def lifespan(_app: FastAPI):
healthcheck pass while that settles.
"""
# Runs before the thread below, and synchronously: the seed catalogs and
# model artifacts have to be in place before the first request can read
# them, and it is a handful of file copies on first boot, nothing on every
# boot after that.
try:
restore_bundled_assets()
except Exception as e:
logger.warning("Could not restore bundled assets: %s", e)
def _async_init():
try:
ensure_store_intelligence_schema()
@@ -58,7 +81,17 @@ async def lifespan(_app: FastAPI):
logger.warning("Startup background init warning: %s", e)
threading.Thread(target=_async_init, daemon=True).start()
yield
# The mounted MCP app carries its own lifespan, which starts the session
# manager its request handler depends on. Mounting a sub-app does NOT run
# that lifespan - only the outermost app's is executed - so without this
# the endpoint exists, accepts a connection, and then fails on the first
# message with a session manager that was never started.
if _mcp_app is not None:
async with _mcp_app.lifespan(_mcp_app):
yield
else:
yield
app = FastAPI(
@@ -98,6 +131,24 @@ app.add_middleware(
allow_headers=["*"],
)
# A wrong origin list fails only in the browser, as an opaque "blocked by CORS"
# with a perfectly healthy 200 in the server log - so state the effective list
# at startup, where it can actually be compared against the frontend's URL.
logger.info(
"CORS allowed origins: %s",
", ".join(API_CORS_ORIGINS) if API_CORS_ORIGINS else "(none - same-origin only)",
)
if API_CORS_ORIGINS and all(
o.startswith(("http://localhost", "http://127.0.0.1")) for o in API_CORS_ORIGINS
):
logger.warning(
"API_CORS_ORIGINS lists only localhost origins (%s). If this API is "
"published on a domain and the frontend is served from a different one, "
"every browser request will be blocked. Set API_CORS_ORIGINS to the "
"frontend's exact origin, e.g. https://catalogue.nearle.ai.in",
", ".join(API_CORS_ORIGINS),
)
app.include_router(health.router, prefix="/api")
app.include_router(auth.router, prefix="/api")
app.include_router(user_products.router, prefix="/api")
@@ -116,10 +167,44 @@ app.include_router(store_admin.router, prefix="/api")
app.include_router(nutrition.router, prefix="/api")
app.include_router(nutrition_admin.router, prefix="/api")
app.include_router(upload.router, prefix="/api")
app.include_router(mcp_info.router, prefix="/api")
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,
# not part of the REST surface, and the catch-all SPA route below only skips
# paths it recognises. Mounted rather than routed because it is a whole ASGI
# app with its own request handling.
if _mcp_app is not None:
app.mount(MCP_PATH, _mcp_app)
logger.info("MCP server mounted at %s", MCP_PATH)
# Serve built frontend static files if dist exists (single-port unified
# deployment). Off in the normal setup: the React app is served by its own
# nginx on catalogue.nearle.ai.in and this API answers on
# mcp.nearle.ai.in, so no dist/ is present here and the JSON root
# handler at the bottom of this file is what responds to /.
#
# The candidates cover both repo layouts - the sibling checkout is named
# `catalogue_frontend`, and only `frontend` was checked before, so this branch
# could never activate even when a build was sitting right next to it.
# FRONTEND_DIST_DIR overrides both when the build lands somewhere else.
_dist_override = os.getenv("FRONTEND_DIST_DIR", "").strip()
_repo_root = Path(__file__).resolve().parents[2]
_dist_candidates = (
[Path(_dist_override)]
if _dist_override
else [
_repo_root / "catalogue_frontend" / "dist",
_repo_root / "frontend" / "dist",
Path(__file__).resolve().parents[1] / "frontend" / "dist",
]
)
FRONTEND_DIST = next(
(p for p in _dist_candidates if (p / "assets").exists()),
_dist_candidates[0],
)
# Serve built frontend static files if dist exists (single-port unified deployment)
FRONTEND_DIST = Path(__file__).resolve().parents[2] / "frontend" / "dist"
if FRONTEND_DIST.exists() and (FRONTEND_DIST / "assets").exists():
logger.info("Serving built frontend from %s", FRONTEND_DIST)
app.mount("/assets", StaticFiles(directory=str(FRONTEND_DIST / "assets")), name="assets")
@app.get("/{full_path:path}")
@@ -130,7 +215,10 @@ if FRONTEND_DIST.exists() and (FRONTEND_DIST / "assets").exists():
# endpoint would look like a successful call to any client. 404 is the
# honest answer, and it matters now that the API is also reachable
# directly at api.{$DOMAIN} rather than only behind the frontend.
if full_path.startswith(("api", "docs", "redoc", "openapi.json")):
# "mcp" is in this list for the case where the MCP app failed to load:
# the mount would be absent, and without this the SPA fallback would
# answer an MCP client with index.html and a 200.
if full_path.startswith(("api", "docs", "redoc", "openapi.json", "mcp")):
raise HTTPException(status_code=404, detail="Not found")
file_path = FRONTEND_DIST / full_path
if file_path.exists() and file_path.is_file():

522
app/mcp_server.py Normal file
View File

@@ -0,0 +1,522 @@
"""
MCP server exposing the catalog as tools for AI clients.
Mounted onto the FastAPI app at /mcp (see app/main.py), so it ships in the same
container and answers on the same host - mcp.nearle.ai.in/mcp.
Scope: reads only. Every tool here maps to a service function the REST API
already exposes through a GET. None of the write or compute endpoints - catalog
generation, ML training, the upload endpoints, product creation - are reachable
through MCP, because a tool list is chosen from by a model rather than by a
person, and "retrain the models" is not a thing to leave one tool call away.
Adding a write tool later is deliberate work, not an oversight to correct.
Authentication reuses the access token from POST /api/auth/login. There is no
separate MCP credential: the same Principal, the same expiry, the same
revocation story as the REST API, and nothing new to keep secret. Clients send
it as an Authorization: Bearer header, which _AuthMiddleware checks once for
every tool call and tool listing.
The tools are hand-written rather than generated from the OpenAPI schema. All
63 routes would technically work, but a model picks a tool by reading its
description, and 63 near-identical auto-generated entries is a worse starting
point than a dozen written to be chosen between.
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List, Optional
from fastmcp import FastMCP
from fastmcp.exceptions import ToolError
from fastmcp.server.dependencies import get_http_headers
from fastmcp.server.middleware import CallNext, Middleware, MiddlewareContext
from app.infrastructure.security import AuthError, Principal, anonymous_principal, decode_access_token
from app.infrastructure.settings import AUTH_ENABLED
logger = logging.getLogger(__name__)
MCP_PATH = "/mcp"
INSTRUCTIONS = """\
Read-only access to an Indian FMCG product catalog: products and brands, \
nutrition facts and health scores, per-store inventory and pricing, discounts, \
and sales analytics.
Products are identified by a (brand, image_id) pair - image_id is the SKU-like \
product key, not an image URL. Get one from search_products or \
list_brand_products before calling any tool that takes an image_id.
Nutrition figures are per 100g unless the product states otherwise, and health \
scores run 0-100, higher being better. Prices are in INR.
"""
# ---------------------------------------------------------------------------
# Authentication
# ---------------------------------------------------------------------------
def _principal_from_headers() -> Principal:
"""
Resolve the caller from the Authorization header of the current request.
Raises ToolError rather than AuthError: ToolError is what reaches the client
as a readable message instead of an opaque internal failure.
"""
if not AUTH_ENABLED:
return anonymous_principal()
# include={"authorization"} is required, not optional: get_http_headers()
# strips the Authorization header by default, so that it is not forwarded to
# downstream services by accident. Without this the header is invisible here
# and every request looks unauthenticated - including valid ones.
headers = get_http_headers(include={"authorization"})
raw = headers.get("authorization", "") # keys are lowercased by the helper
scheme, _, token = raw.partition(" ")
if scheme.lower() != "bearer" or not token.strip():
raise ToolError(
"Not authenticated. Send an Authorization: Bearer <token> header. "
"Get a token from POST /api/auth/login."
)
try:
return decode_access_token(token.strip())
except AuthError as exc:
raise ToolError(str(exc)) from exc
class _AuthMiddleware(Middleware):
"""
One authentication check covering every tool.
Sitting here rather than at the top of each tool means a tool added later
cannot be left unguarded by forgetting a line - the same reason the REST
side puts its guard in a route dependency instead of the handler body.
Listing is checked as well as calling. An unauthenticated client learning
the exact shape of the catalog API is a smaller problem than an
unauthenticated call, but it is not nothing, and there is no reason to
publish it.
"""
async def on_call_tool(self, context: MiddlewareContext, call_next: CallNext):
principal = _principal_from_headers()
name = getattr(context.message, "name", "?")
logger.info("MCP tool call %r by %s (%s)", name, principal.username, principal.kind)
return await call_next(context)
async def on_list_tools(self, context: MiddlewareContext, call_next: CallNext):
_principal_from_headers()
return await call_next(context)
mcp: FastMCP = FastMCP(
name="nearle-catalogue",
instructions=INSTRUCTIONS,
version="1.0.0",
middleware=[_AuthMiddleware()],
)
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def _guard(what: str, fn, *args, **kwargs):
"""
Run a service call, turning failures into a ToolError the client can read.
Nearly every tool here reads Postgres, so the common failure is the database
being unreachable. Left to propagate, that surfaces to the model as an
unexplained error; named, the model can say what went wrong instead of
retrying or inventing an answer.
"""
try:
return fn(*args, **kwargs)
except Exception as exc: # noqa: BLE001 - deliberately broad, see docstring
logger.exception("MCP tool %s failed", what)
raise ToolError(f"{what} failed: {exc}") from exc
def _clip(n: Optional[int], default: int, maximum: int) -> int:
"""Keep result counts sane - a model will happily ask for 10000 rows."""
if n is None:
return default
return max(1, min(int(n), maximum))
def _slim(product: Dict[str, Any]) -> Dict[str, Any]:
"""
Drop the fields a model has no use for.
Embeddings especially: a 384-float vector per product would swamp the
context window and says nothing a model can reason about.
"""
return {
k: v
for k, v in product.items()
if k not in {"embedding", "embedding_text", "distance"} and v not in (None, "", [], {})
}
# ---------------------------------------------------------------------------
# Catalog
# ---------------------------------------------------------------------------
@mcp.tool(
annotations={"readOnlyHint": True},
tags={"catalog"},
)
def search_products(
query: str,
brand: Optional[str] = None,
category: Optional[str] = None,
limit: Optional[int] = 10,
) -> List[Dict[str, Any]]:
"""Search the catalog by meaning, not keywords.
The main entry point: use this to turn a description ("sugar-free biscuits",
"something to replace butter") into concrete products. Returns each match
with its brand and image_id, which the nutrition, store and recommendation
tools take as input.
Args:
query: What to look for, in natural language.
brand: Restrict to one brand. Omit to search everything.
category: Restrict to one category, e.g. "Dairy".
limit: How many products to return (1-50).
"""
from app.services.rag_service import retrieve
top_k = _clip(limit, 10, 50)
found = _guard("search_products", retrieve, query, brand=brand, top_k=top_k, category=category)
return [
_slim(
{
"brand": p.brand,
"image_id": p.image_id,
"title": p.title,
"category": p.category,
"description": p.description,
"price_range": p.price_range,
"selling_price": p.final_selling_price or p.selling_price,
"size_variants": p.size_variants,
"barcode": p.barcode,
}
)
for p in found
]
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def list_brands() -> List[str]:
"""List every brand in the catalog.
Useful for grounding a brand name before passing it to another tool, since
brand names must match the catalog's spelling.
"""
from app.services.vector_store import list_available_brands
return _guard("list_brands", list_available_brands)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def list_categories(brand: str) -> List[str]:
"""List the product categories a brand sells.
Args:
brand: Brand name, as returned by list_brands.
"""
from app.services.vector_store import list_categories_for_brand
return _guard("list_categories", list_categories_for_brand, brand)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def list_brand_products(
brand: str,
category: Optional[str] = None,
limit: Optional[int] = 25,
) -> List[Dict[str, Any]]:
"""List a brand's products, newest catalog entries first.
Browsing, as opposed to search_products' semantic matching. Prefer
search_products when the user described what they want rather than naming a
brand outright.
Args:
brand: Brand name, as returned by list_brands.
category: Restrict to one category.
limit: How many products to return (1-100).
"""
from app.services.vector_store import get_products_by_brand
rows = _guard(
"list_brand_products",
get_products_by_brand,
brand,
limit=_clip(limit, 25, 100),
category=category,
)
return [_slim(r) for r in rows]
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def get_product(brand: str, image_id: str) -> Dict[str, Any]:
"""Get the full catalog record for one product.
Args:
brand: Brand name.
image_id: The product key from search_products or list_brand_products.
"""
from app.services.vector_store import get_product_by_image_id
row = _guard("get_product", get_product_by_image_id, brand, image_id)
if not row:
raise ToolError(
f"No product {image_id!r} for brand {brand!r}. "
"Check the pair with search_products."
)
return _slim(row)
# ---------------------------------------------------------------------------
# Nutrition
# ---------------------------------------------------------------------------
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
def get_nutrition(brand: str, image_id: str) -> Dict[str, Any]:
"""Get nutrition facts, health score and dietary flags for one product.
Covers macros and key micronutrients per 100g, a 0-100 health score, diet
tags (vegetarian, high-protein, ...) and declared allergens.
Args:
brand: Brand name.
image_id: The product key from search_products.
"""
from app.services.nutrition_db import get_full_nutrition
row = _guard("get_nutrition", get_full_nutrition, brand, image_id)
if not row:
raise ToolError(
f"No nutrition data for {brand}/{image_id}. Not every catalog "
"product has been enriched yet."
)
return _slim(row)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
def find_healthier_alternatives(
brand: str,
image_id: str,
limit: Optional[int] = 5,
) -> List[Dict[str, Any]]:
"""Find comparable products with a better health score.
Answers "what should I buy instead of this?" - matches are drawn from the
same category, so the suggestion is a real substitute rather than merely
something healthier.
Args:
brand: Brand name of the product to improve on.
image_id: Product key of the product to improve on.
limit: How many alternatives to return (1-20).
"""
from app.services.nutrition_alternatives_service import find_alternatives
return _guard(
"find_healthier_alternatives",
find_alternatives,
brand,
image_id,
top_k=_clip(limit, 5, 20),
)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
def filter_products_by_nutrition(
sort_by: str = "health_score",
order: str = "desc",
category: Optional[str] = None,
diet_tag: Optional[str] = None,
exclude_allergen: Optional[str] = None,
limit: Optional[int] = 20,
) -> List[Dict[str, Any]]:
"""Rank and filter products by a nutritional measure.
The tool for questions shaped like "highest protein snacks", "lowest sugar
dairy", or "gluten-free products without nuts".
Args:
sort_by: Field to rank by - health_score, protein_g, total_sugar_g,
dietary_fiber_g, sodium_mg, calories_kcal, total_fat_g.
order: "desc" for highest first, "asc" for lowest first.
category: Restrict to one category.
diet_tag: Keep only products carrying this tag, e.g. "High Protein".
exclude_allergen: Drop products declaring this allergen, e.g. "Dairy".
limit: How many products to return (1-100).
"""
from app.services.nutrition_db import query_products
return _guard(
"filter_products_by_nutrition",
query_products,
sort_by=sort_by,
order=order,
category=category,
diet_tag=diet_tag,
exclude_allergen=exclude_allergen,
limit=_clip(limit, 20, 100),
)
# ---------------------------------------------------------------------------
# Stores
# ---------------------------------------------------------------------------
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
def list_stores() -> List[Dict[str, Any]]:
"""List the retail stores, with city and tier.
Call this first for anything store-specific - the other store tools need a
store_id from here.
"""
from app.services.store_db import list_stores as _list
return _guard("list_stores", _list)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
def get_store_inventory(
store_id: str,
category: Optional[str] = None,
in_stock_only: bool = False,
limit: Optional[int] = 25,
) -> List[Dict[str, Any]]:
"""List what one store carries, with stock levels and its own pricing.
Prices vary per store, so this is the tool for "what does it cost at X",
not the catalog price_range.
Args:
store_id: Store identifier from list_stores.
category: Restrict to one category.
in_stock_only: Drop products currently out of stock.
limit: How many products to return (1-100).
"""
from app.services.store_db import get_store_products
return _guard(
"get_store_inventory",
get_store_products,
store_id,
category=category,
in_stock_only=in_stock_only,
limit=_clip(limit, 25, 100),
)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
def get_store_discounts(store_id: str, limit: Optional[int] = 25) -> List[Dict[str, Any]]:
"""List the discounts currently allocated at one store.
Args:
store_id: Store identifier from list_stores.
limit: How many discounts to return (1-100).
"""
from app.services.store_db import get_latest_discounts
return _guard(
"get_store_discounts", get_latest_discounts, store_id, limit=_clip(limit, 25, 100)
)
# ---------------------------------------------------------------------------
# Analytics
# ---------------------------------------------------------------------------
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
def get_trending(
window: str = "weekly",
scope: str = "overall",
scope_value: Optional[str] = None,
limit: Optional[int] = 10,
) -> List[Dict[str, Any]]:
"""List what is selling fastest right now.
Args:
window: Period to measure over - "daily", "weekly" or "monthly".
scope: "overall", or "store"/"category" to narrow it.
scope_value: The store_id or category name, when scope is not "overall".
limit: How many products to return (1-50).
"""
from app.services.store_db import get_trending as _trending
return _guard(
"get_trending",
_trending,
window_label=window,
scope=scope,
scope_value=scope_value,
top_k=_clip(limit, 10, 50),
)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
def get_top_products(
by: str = "revenue",
limit: Optional[int] = 10,
ascending: bool = False,
) -> List[Dict[str, Any]]:
"""Rank products across every store by a sales measure.
Args:
by: What to rank on - "revenue", "units" or "orders".
limit: How many products to return (1-50).
ascending: True ranks worst-performing first.
"""
from app.services.analytics_service import top_products
return _guard(
"get_top_products", top_products, by=by, limit=_clip(limit, 10, 50), ascending=ascending
)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
def get_store_analytics(store_id: str) -> Dict[str, Any]:
"""Get one store's sales dashboard - revenue, orders, top sellers, stock health.
Args:
store_id: Store identifier from list_stores.
"""
from app.services.analytics_service import store_dashboard
return _guard("get_store_analytics", store_dashboard, store_id)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def get_recommendations(
brand: str,
image_id: str,
limit: Optional[int] = 5,
) -> List[Dict[str, Any]]:
"""List products frequently bought together with this one.
Co-purchase based, so this answers "what else goes in the basket", which is
a different question from find_healthier_alternatives' "what instead".
Args:
brand: Brand name.
image_id: Product key from search_products.
limit: How many recommendations to return (1-20).
"""
from app.services.recommendation_service import recommend_for_product
return _guard(
"get_recommendations",
recommend_for_product,
brand,
image_id,
top_k=_clip(limit, 5, 20),
)
def build_http_app(path: str = MCP_PATH):
"""The ASGI app to mount on FastAPI. See app/main.py for the lifespan wiring."""
return mcp.http_app(path="/", transport="http", stateless_http=True)

View File

@@ -42,6 +42,8 @@ actually resolves to real image bytes above a minimum size - this is what
stops garbage/placeholder/expired URLs from reaching the S3 upload step
and failing there silently.
"""
from __future__ import annotations
from typing import Optional, List
import requests
from urllib.parse import urlparse

View File

@@ -1,3 +1,5 @@
from __future__ import annotations
from typing import List, Dict, Any, Optional
import json
import re

View File

@@ -9,7 +9,10 @@ import re
import time
import psycopg
from app.infrastructure.settings import DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD
from app.infrastructure.settings import (
DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD,
DB_CONNECT_TIMEOUT_SECONDS,
)
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
@@ -93,7 +96,15 @@ def _connect() -> Optional[psycopg.Connection]:
dbname=DB_NAME,
user=DB_USER,
password=DB_PASSWORD,
autocommit=True
autocommit=True,
# Without this, a host that DROPS packets rather than refusing them
# - a firewall, a wrong DB_HOST - blocks here until the OS gives up,
# which is around 130 seconds on Linux. Every caller of _connect()
# inherits that: /api/health stops answering, the container's
# healthcheck times out, and the platform pulls the service out of
# its load balancer. "Database unreachable" then presents as a Bad
# Gateway on every route, including ones that never touch the DB.
connect_timeout=DB_CONNECT_TIMEOUT_SECONDS,
)
except Exception as e:
logger.error(f"Vector DB connection failed: {e}")

View File

@@ -9,7 +9,8 @@
# container layer, and so `ollama pull` model files persist outside Docker.
#
# Usage:
# docker compose up -d
# docker compose up -d # Postgres only (the default)
# docker compose --profile full up -d # Postgres + the API, with volumes
# # then in backend/.env: DB_HOST=localhost, DB_PORT=5432, DB_NAME=pgvector,
# # DB_USER=postgres, DB_PASSWORD=<the value you set below>
services:
@@ -34,5 +35,53 @@ services:
timeout: 5s
retries: 10
# The API. Started only with `--profile full`, so the default
# `docker compose up -d` still brings up Postgres alone as it always did.
#
# docker compose --profile full up -d --build
#
# Present mainly as the reference for the two volume mounts below - the paths
# are the same ones to configure in Dokploy.
backend:
profiles: ["full"]
build: .
container_name: catalog_rag_backend
restart: unless-stopped
depends_on:
postgres:
condition: service_healthy
env_file: .env
environment:
# Inside a container `localhost` is the container itself, so the DB is
# reached by service name over the compose network.
DB_HOST: postgres
DB_PORT: "5432"
DB_PASSWORD: ${POSTGRES_PASSWORD:-changeme}
# Ollama runs natively on the host, not in compose (see the note above).
OLLAMA_BASE_URL: http://host.docker.internal:11434
extra_hosts:
- "host.docker.internal:host-gateway"
# The container listens on 3000 and 8000 at once (see serve.py), so either
# side of this mapping can change without touching the image. 8000 is
# published because vite.config.js proxies /api to 127.0.0.1:8000.
ports:
- "8000:8000"
volumes:
# WITHOUT THESE TWO MOUNTS, a redeploy silently discards:
# - every product added through the UI (POST /api/user/products/add and
# /upload-file append to data/seed_catalogs/*.json), and
# - every retrained model (the training endpoints write *.joblib).
# Both directories live inside the image, so rebuilding it resets them to
# whatever was committed to the repo.
#
# The image also carries a pristine copy at /app/.bundled, and the app
# tops up anything missing on startup without overwriting what is already
# there - so these work whether they are named volumes or bind mounts.
# See app/infrastructure/persistence.py.
- catalog_rag_data:/app/data
- catalog_rag_artifacts:/app/app/intelligence/artifacts
volumes:
catalog_rag_pgdata:
catalog_rag_data:
catalog_rag_artifacts:

View File

@@ -14,6 +14,13 @@ python-dotenv>=1.0.1
# deliberately no bcrypt/argon2/passlib dependency here.
PyJWT>=2.9.0
# --- MCP (Model Context Protocol) ---
# Serves the catalog as tools for AI clients at /mcp (see app/mcp_server.py),
# mounted onto the FastAPI app so it needs no separate process or container.
# Requires Python >=3.10; the Dockerfile's python:3.11-slim satisfies that, and
# app/main.py degrades to serving the REST API alone if the import fails.
fastmcp>=3.4.7
# --- Database / pgvector ---
psycopg[binary]>=3.2.3
pgvector>=0.2.5

82
scripts/check_deploy.sh Executable file
View File

@@ -0,0 +1,82 @@
#!/usr/bin/env bash
# Post-deploy smoke test.
#
# ./scripts/check_deploy.sh # https://mcp.nearle.ai.in
# ./scripts/check_deploy.sh http://localhost:3000 # a local container
#
# Separates the three failures that all look alike from a browser:
# 502 everywhere - the container is not running (it exited at startup)
# 200 + database:false - the API is fine, Postgres is not
# 200 + database:true - working
set -uo pipefail
BASE="${1:-https://mcp.nearle.ai.in}"
pass=0; fail=0
hit() { curl -sS -m 20 -o /tmp/_body -w '%{http_code}' "$BASE$1" 2>/dev/null || echo 000; }
check() { # path expected label
local code; code=$(hit "$1")
if [ "$code" = "$2" ]; then printf ' \033[32mPASS\033[0m %-16s %s\n' "$1" "$3"; pass=$((pass+1))
else printf ' \033[31mFAIL\033[0m %-16s expected %s, got %s\n' "$1" "$2" "$code"; fail=$((fail+1)); fi
}
echo "Checking $BASE"
echo
echo "Is the container running at all?"
code=$(hit /)
if [ "$code" = "502" ] || [ "$code" = "000" ]; then
echo " FAIL / -> $code"
echo
echo " The container is not serving. It almost certainly exited at startup."
echo " Read the Dokploy logs; settings.py names the missing value explicitly."
echo " Confirm the image actually contains /app/.env - the build log must show"
echo " a 'COPY .env .' step, and .dockerignore must not list .env."
exit 1
fi
printf ' \033[32mPASS\033[0m %-16s container is up\n' "/"
pass=$((pass+1))
echo
echo "Routes that must not depend on the database:"
check /docs 200 "interactive API docs"
check /openapi.json 200 "OpenAPI schema"
check /api/health 200 "health endpoint answers"
echo
echo "Dependencies (reported in the body; 'false' does not mean the API is broken):"
health=$(curl -sS -m 20 "$BASE/api/health" 2>/dev/null)
db=$(printf '%s' "$health" | sed -n 's/.*"database":\([a-z]*\).*/\1/p')
ol=$(printf '%s' "$health" | sed -n 's/.*"ollama":\([a-z]*\).*/\1/p')
if [ "$db" = "true" ]; then printf ' \033[32mPASS\033[0m database connected\n'; pass=$((pass+1))
else
printf ' \033[33mWARN\033[0m database NOT connected\n'
echo " The API works; catalog pages will be empty. Check DB_HOST/DB_USER/"
echo " DB_PASSWORD/DB_NAME. A wrong password logs 'password authentication"
echo " failed' rather than a timeout."
fi
[ "$ol" = "true" ] \
&& printf ' \033[32mPASS\033[0m ollama connected\n' \
|| printf ' \033[33mWARN\033[0m ollama not connected (/api/chat unavailable; expected if USE_OLLAMA=false)\n'
echo
echo "Auth:"
code=$(curl -sS -m 20 -o /dev/null -w '%{http_code}' -X POST "$BASE/api/auth/login" \
-H 'Content-Type: application/json' -d '{"username":"admin","password":"definitely-not-the-password"}' 2>/dev/null)
if [ "$code" = "401" ]; then printf ' \033[32mPASS\033[0m wrong password rejected (401)\n'; pass=$((pass+1))
elif [ "$code" = "200" ]; then
printf ' \033[31mFAIL\033[0m a WRONG PASSWORD WAS ACCEPTED - AUTH_ALLOW_ANY_LOGIN is true.\n'
echo " Anyone who finds this host can sign in as admin. Set it to false."
fail=$((fail+1))
else printf ' \033[31mFAIL\033[0m login endpoint returned %s\n' "$code"; fail=$((fail+1)); fi
echo
echo "MCP:"
code=$(curl -sS -m 20 -o /dev/null -w '%{http_code}' "$BASE/api/mcp/info" 2>/dev/null)
[ "$code" = "401" ] \
&& { printf ' \033[32mPASS\033[0m /api/mcp/info guarded (401 without a token)\n'; pass=$((pass+1)); } \
|| printf ' \033[33mWARN\033[0m /api/mcp/info returned %s (expected 401)\n' "$code"
echo
echo "-------- $pass passed, $fail failed --------"
[ "$fail" -eq 0 ]

170
serve.py Normal file
View File

@@ -0,0 +1,170 @@
"""
Container entry point: serve the API on every port the platform might route to.
The frontend image answers on both 80 and 3000 (``listen 80; listen 3000;`` in
nginx.conf) so it works whatever port the deployment is configured to hit. This
does the same for the API, which otherwise has to guess: Dokploy routes a domain
to one container port, this project's own README, vite.config.js proxy and
docker-compose mapping all say 8000, and picking wrong produces a 502 with a
perfectly healthy container behind it.
uvicorn's CLI binds a single ``--port``, but ``Server.run()`` accepts a list of
already-bound sockets, so one process can listen on several. That is what this
does - no extra worker, no second process to supervise.
Usage::
python serve.py # binds PORTS (default "3000,8000")
PORT=8080 python serve.py # binds only 8080
python serve.py --healthcheck # probe mode, used by HEALTHCHECK
Run it directly rather than through ``uvicorn app.main:app`` when you need the
multi-port behaviour; the plain uvicorn command still works for local
development where one port is all anyone wants.
"""
from __future__ import annotations
import logging
import os
import socket
import sys
import urllib.error
import urllib.request
logger = logging.getLogger("serve")
# Both of the ports this project actually uses anywhere: 3000 because that is
# what the platform routes a domain to by default, 8000 because the README,
# the vite dev proxy and docker-compose all target it.
DEFAULT_PORTS = "3000,8000"
HOST = os.getenv("HOST", "0.0.0.0")
def configured_ports() -> list[int]:
"""
The ports to bind, in order.
``PORT`` wins over ``PORTS`` and is treated as an explicit single choice:
setting it means "serve here", not "serve here as well". ``PORTS`` takes a
comma-separated list for the both-at-once case.
"""
raw = os.getenv("PORT") or os.getenv("PORTS") or DEFAULT_PORTS
ports: list[int] = []
for chunk in raw.split(","):
chunk = chunk.strip()
if not chunk:
continue
try:
port = int(chunk)
except ValueError:
logger.warning("Ignoring non-numeric port %r in %r", chunk, raw)
continue
if not 1 <= port <= 65535:
logger.warning("Ignoring out-of-range port %d", port)
continue
if port not in ports: # binding the same port twice would fail
ports.append(port)
if not ports:
logger.error("No usable port in %r - falling back to %s", raw, DEFAULT_PORTS)
return [int(p) for p in DEFAULT_PORTS.split(",")]
return ports
def _bind(port: int) -> socket.socket | None:
"""Bind one listening socket, or return None with the reason logged."""
sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
try:
sock.bind((HOST, port))
except OSError as exc:
# Not fatal on its own. If the platform only routes to one of these,
# losing the other (already in use, not permitted) should not take the
# service down - _run() fails only when nothing at all is listening.
logger.warning("Could not bind %s:%d - %s", HOST, port, exc)
sock.close()
return None
sock.listen(2048)
sock.set_inheritable(True)
return sock
def _run() -> int:
import uvicorn
ports = configured_ports()
sockets = [s for s in (_bind(p) for p in ports) if s is not None]
if not sockets:
logger.error(
"Could not bind any of %s. The API is not listening; exiting so the "
"platform restarts or reports the container as failed.",
", ".join(str(p) for p in ports),
)
return 1
bound = [s.getsockname()[1] for s in sockets]
logger.info("Serving on %s port(s): %s", HOST, ", ".join(str(p) for p in bound))
config = uvicorn.Config(
"app.main:app",
# proxy_headers/forwarded_allow_ips: this container is never reached
# directly - nginx, and Dokploy's Traefik, sit in front of it. Without
# them uvicorn ignores X-Forwarded-Proto and reports every request as
# plain http, so any redirect or generated absolute URL would downgrade
# an https request.
proxy_headers=True,
forwarded_allow_ips="*",
)
uvicorn.Server(config).run(sockets=sockets)
return 0
def _healthcheck() -> int:
"""
Probe the API, passing if ANY configured port answers.
Shares configured_ports() with the server so the check cannot drift from
what is actually bound - the reason this lives here rather than being a
python -c one-liner in the Dockerfile.
Probes "/" rather than /api/health, and the distinction matters. This is a
LIVENESS check: the only question it may ask is "is this process still
serving HTTP", because the platform's answer to "no" is to stop routing
traffic to the container.
/api/health is a READINESS report - it dials Postgres and Ollama to say
whether they are reachable. Using it here couples the container's existence
to its dependencies: an unreachable database made /api/health block for the
OS TCP timeout, the check timed out, the container was marked unhealthy, and
a service that was running perfectly well returned Bad Gateway on every
route - including the ones that never touch the database. "/" is served from
memory and does no I/O at all, so it can only fail if the app really is gone.
"""
ports = configured_ports()
for port in ports:
try:
with urllib.request.urlopen(f"http://127.0.0.1:{port}/", timeout=8) as resp:
if 200 <= resp.status < 500:
return 0
except urllib.error.HTTPError as exc:
# An HTTP status - even 404 - means something is listening and
# routing. That is exactly what liveness asks.
if exc.code < 500:
return 0
except (urllib.error.URLError, OSError, ValueError):
continue
print(
f"health: no response from any of {', '.join(str(p) for p in ports)}",
file=sys.stderr,
)
return 1
if __name__ == "__main__":
logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s")
if "--healthcheck" in sys.argv:
sys.exit(_healthcheck())
sys.exit(_run())