Merge branch 'master' of https://gitapp.workolik.com/nearle_daily/catalogue_backend
This commit is contained in:
@@ -3,6 +3,15 @@ __pycache__
|
||||
*.pyc
|
||||
.pytest_cache
|
||||
*.log
|
||||
.env
|
||||
.git
|
||||
tests
|
||||
|
||||
# .env.production carries this deployment's configuration and the Dockerfile
|
||||
# copies it to /app/.env inside the image, so it must NOT be ignored here.
|
||||
# Dokploy's own .env (written from the Environment tab, empty when that tab is
|
||||
# blank) is ignored instead - it is what silently overwrote the committed one.
|
||||
#
|
||||
# The generated sign-in passwords must never be in the image.
|
||||
.env
|
||||
.env.local
|
||||
SIGNIN_PASSWORDS.txt
|
||||
|
||||
58
.env.example
58
.env.example
@@ -16,6 +16,64 @@
|
||||
# localhost URL points the backend at its own empty ports. Containers reach
|
||||
# each other by service name over the compose network instead.
|
||||
|
||||
# --- Ports (container only) -----------------------------------------------
|
||||
# The image serves 3000 and 8000 at once - 3000 because that is what Dokploy
|
||||
# routes a domain to, 8000 because the README, the vite dev proxy and
|
||||
# docker-compose all use it. Serving both means the container works whichever
|
||||
# one the platform points at.
|
||||
#
|
||||
# PORTS the comma-separated pair to bind. PORT pins a single port instead and
|
||||
# takes precedence, e.g. PORT=8080 serves only 8080.
|
||||
#
|
||||
# Neither affects running uvicorn directly for local development.
|
||||
# PORTS=3000,8000
|
||||
# PORT=8080
|
||||
|
||||
# --- Persistence (IMPORTANT in Docker) ------------------------------------
|
||||
# The three directories the app WRITES to at runtime. The defaults point inside
|
||||
# the repo/image and are right for local development; in a container each one
|
||||
# needs a volume mounted on it, or a redeploy throws away everything written
|
||||
# since the last build:
|
||||
#
|
||||
# SEED_CATALOG_DIR products added via POST /api/user/products/add and
|
||||
# /upload-file are appended to the JSON files here
|
||||
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
|
||||
# DATA_DIR catalogs saved by the ingestion pipeline
|
||||
#
|
||||
# Mount these two paths in Dokploy (SEED_CATALOG_DIR sits inside DATA_DIR, so
|
||||
# one mount covers both):
|
||||
#
|
||||
# /app/data
|
||||
# /app/app/intelligence/artifacts
|
||||
#
|
||||
# Either a named volume or a bind mount works. The image keeps read-only copies
|
||||
# of the bundled seed catalogs and pre-trained models at /app/.bundled, and the
|
||||
# app restores whatever a freshly-mounted directory is missing on startup
|
||||
# without overwriting anything already there - so a bind mount, which starts
|
||||
# empty and would otherwise hide them, is safe.
|
||||
#
|
||||
# Leave all three unset unless the writable data belongs somewhere else.
|
||||
# DATA_DIR=/app/data
|
||||
# SEED_CATALOG_DIR=/app/data/seed_catalogs
|
||||
# MODEL_ARTIFACTS_DIR=/app/app/intelligence/artifacts
|
||||
|
||||
# --- CORS (REQUIRED when the API is on its own domain) --------------------
|
||||
# Comma-separated list of the exact browser origins allowed to call this API.
|
||||
# Scheme and host both matter; no trailing slash, and no wildcard - the
|
||||
# frontend sends an Authorization header, and browsers refuse a credentialed
|
||||
# cross-origin request whose Allow-Origin is "*" (app/main.py logs and turns
|
||||
# credentials off if it sees one, so a wildcard silently breaks every call).
|
||||
#
|
||||
# Production - the React app is served from catalogue.nearle.ai.in and calls
|
||||
# the API at mcp.nearle.ai.in, so that frontend origin must be listed:
|
||||
#
|
||||
# API_CORS_ORIGINS=https://catalogue.nearle.ai.in
|
||||
#
|
||||
# Add http://localhost:5173 alongside it if you point a local Vite dev server
|
||||
# at the deployed API. Server-to-server callers (MCP clients, scripts) are not
|
||||
# affected by any of this - CORS is a browser rule; they use X-API-Key.
|
||||
API_CORS_ORIGINS=http://localhost:5173,http://127.0.0.1:5173
|
||||
|
||||
# --- Authentication (REQUIRED) -------------------------------------------
|
||||
# The backend will not start without these while AUTH_ENABLED=true. Generate
|
||||
# all four lines, plus sign-in passwords, with:
|
||||
|
||||
113
.env.production
Normal file
113
.env.production
Normal file
@@ -0,0 +1,113 @@
|
||||
# Deployment configuration for mcp.nearle.ai.in.
|
||||
#
|
||||
# Committed at the repo owner's instruction so the deploy does not depend on
|
||||
# re-entering config in the Dokploy UI. Everything needed to boot is here; no
|
||||
# environment variables are required in Dokploy any more.
|
||||
#
|
||||
# A real environment variable still overrides anything set here - settings.py
|
||||
# calls load_dotenv() without override=True, so the process environment wins.
|
||||
# That is the escape hatch for changing a value without a commit.
|
||||
#
|
||||
# WHAT IS IN THIS FILE: live database, S3 and Google credentials, and the key
|
||||
# that signs every access token. Anyone with read access to this repository has
|
||||
# all of it, and git history keeps it after any rotation.
|
||||
|
||||
# --- Ports -----------------------------------------------------------------
|
||||
# Dokploy routes the domain to 3000; 8000 is kept for the vite dev proxy and
|
||||
# docker-compose. serve.py binds both.
|
||||
PORTS=3000,8000
|
||||
|
||||
# --- CORS ------------------------------------------------------------------
|
||||
# The FRONTEND's origin, not this API's. Wrong value = the browser blocks every
|
||||
# response while the server logs healthy 200s.
|
||||
API_CORS_ORIGINS=https://catalogue.nearle.ai.in
|
||||
|
||||
# --- Authentication --------------------------------------------------------
|
||||
AUTH_ENABLED=true
|
||||
|
||||
# Freshly generated for this deployment - deliberately NOT the values from the
|
||||
# development .env. Those hashes are for the passwords DevAdmin!2026 and
|
||||
# DevUser!2026, which are sitting in plaintext in test_login_fix.py in this very
|
||||
# repository: shipping them would publish working admin credentials.
|
||||
# Sign-in passwords for the hashes below are in SIGNIN_PASSWORDS.txt (ignored).
|
||||
AUTH_SECRET_KEY=4Kmyr4Cjf_kdUIq_4EGxo5vFHfCT5_uKVR3eouszB8Le6F0n45m7eDY94_KJoqSz
|
||||
AUTH_ADMIN_USERNAME=admin
|
||||
AUTH_ADMIN_PASSWORD_HASH=pbkdf2_sha256$600000$Xa07unPO4LeTU4bz04eh7Q==$KmZ2ZBrclJ0z0sDiCoKOrfTO2UK8e7hjsZZ6HB2PY9o=
|
||||
AUTH_USER_USERNAME=user
|
||||
AUTH_USER_PASSWORD_HASH=pbkdf2_sha256$600000$65VMpqwUSyFzCqnhlwBqgQ==$sMVnar+Hnp5ZmXcObFNI3jJMxqnVrW8naLZNFFh0KCw=
|
||||
|
||||
AUTH_TOKEN_TTL_MINUTES=720
|
||||
AUTH_MAX_LOGIN_ATTEMPTS=10
|
||||
AUTH_LOCKOUT_SECONDS=300
|
||||
|
||||
# MUST stay false here. The development .env has this true, where it is a
|
||||
# convenience: it skips the password check entirely, so any username signs in
|
||||
# and `admin` gets the admin pages. On a host published to the internet it means
|
||||
# anyone who finds mcp.nearle.ai.in signs in as admin by typing anything at all.
|
||||
AUTH_ALLOW_ANY_LOGIN=false
|
||||
|
||||
# Machine consumers. Empty: MCP clients authenticate with a login token instead.
|
||||
API_KEYS=
|
||||
|
||||
# --- Postgres / pgvector ---------------------------------------------------
|
||||
# DB_NAME is not set in the development .env, so it falls back to settings.py's
|
||||
# default. Stated explicitly here so the deployment does not depend on that
|
||||
# default staying the same.
|
||||
USE_PGVECTOR=true
|
||||
DB_HOST=31.97.228.132
|
||||
DB_PORT=6054
|
||||
DB_NAME=pgvector
|
||||
DB_USER=admin
|
||||
DB_PASSWORD=Package@321#
|
||||
|
||||
# --- Embeddings ------------------------------------------------------------
|
||||
USE_EMBEDDINGS=true
|
||||
EMBEDDINGS_MODEL=sentence-transformers/all-MiniLM-L6-v2
|
||||
EMBEDDINGS_DIM=384
|
||||
|
||||
# --- Ollama (local LLM, powers /api/chat) ----------------------------------
|
||||
# Off, because the development value (http://localhost:11434) cannot work from
|
||||
# inside a container: there, localhost is the container itself, not the VPS
|
||||
# host. Left on with nothing listening, /api/chat fails AND every healthcheck
|
||||
# takes ~3s longer, because the health handler probes Ollama with a 3s timeout.
|
||||
#
|
||||
# To enable: set USE_OLLAMA=true and point OLLAMA_BASE_URL at something the
|
||||
# container can actually reach - http://host.docker.internal:11434 with a
|
||||
# host-gateway mapping, the VPS's LAN IP, or an ollama service name.
|
||||
USE_OLLAMA=false
|
||||
OLLAMA_BASE_URL=http://host.docker.internal:11434
|
||||
OLLAMA_MODEL_NAME=qwen2.5:1.5b
|
||||
OLLAMA_TIMEOUT_SECONDS=120
|
||||
|
||||
# --- DigitalOcean Spaces (product image storage) ---------------------------
|
||||
USE_S3=true
|
||||
S3_ACCESS_KEY=DO801G8Q8JAZKF49U3WJ
|
||||
S3_SECRET_KEY=lBQExYfkVqH+ybmGVmQH5MkThBbrIohA/VQLgcPUvug
|
||||
S3_ENDPOINT=https://nearle.sgp1.digitaloceanspaces.com
|
||||
S3_BUCKET=nearle
|
||||
S3_REGION=sgp1
|
||||
|
||||
# --- Google Custom Search (optional image source) --------------------------
|
||||
USE_GOOGLE_CSE=true
|
||||
GOOGLE_API_KEY=AIzaSyBY4pIO_Fp5FCMqeVxDNcfalzdWNHJWVn0
|
||||
GOOGLE_CSE_ID=9745cbd96dd164562
|
||||
|
||||
# --- Open-source image sources (no key needed) -----------------------------
|
||||
USE_DDG_IMAGES=true
|
||||
USE_OPEN_FACTS=true
|
||||
USE_WIKIMEDIA=true
|
||||
# The Playwright browser binary is NOT installed in the image (see Dockerfile),
|
||||
# so this tier is skipped at runtime regardless. false stops it being attempted.
|
||||
USE_PLAYWRIGHT_FALLBACK=false
|
||||
|
||||
MIN_IMAGE_BYTES=3000
|
||||
|
||||
# --- Product validation ----------------------------------------------------
|
||||
ENABLE_PRODUCT_VALIDATION=true
|
||||
VALIDATION_REJECT_THRESHOLD=0.35
|
||||
VALIDATION_REVIEW_THRESHOLD=0.70
|
||||
|
||||
# --- RAG -------------------------------------------------------------------
|
||||
RAG_DEFAULT_TOP_K=5
|
||||
RAG_MAX_TOP_K=15
|
||||
RAG_MAX_CONTEXT_CHARS=4000
|
||||
22
.gitignore
vendored
22
.gitignore
vendored
@@ -1,6 +1,26 @@
|
||||
# Secrets - never commit
|
||||
# .env.production is committed deliberately, at the repo owner's instruction, so
|
||||
# the deployment does not depend on re-entering config in the Dokploy UI.
|
||||
#
|
||||
# It is NOT named .env, and that matters: Dokploy writes its own .env into the
|
||||
# build context from the service's Environment tab after cloning, so a committed
|
||||
# .env is silently replaced (with an empty file when that tab is blank).
|
||||
# The Dockerfile copies .env.production to /app/.env inside the image.
|
||||
#
|
||||
# The two database secrets are NOT in it - they are set as Dokploy environment
|
||||
# variables, which override the file (settings.py calls load_dotenv() without
|
||||
# override=True, so the process environment wins).
|
||||
#
|
||||
# The auth secrets ARE in it. AUTH_SECRET_KEY signs every access token, so
|
||||
# anyone with read access to this repository can mint an admin token, and git
|
||||
# history keeps it after any rotation. Regenerate with
|
||||
# `python scripts/make_auth_secrets.py` if that stops being acceptable.
|
||||
!.env.production
|
||||
.env
|
||||
|
||||
# Local overrides and the generated sign-in passwords stay out of git.
|
||||
.env.local
|
||||
SIGNIN_PASSWORDS.txt
|
||||
|
||||
# Python
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
119
Dockerfile
119
Dockerfile
@@ -1,25 +1,116 @@
|
||||
FROM python:3.11-slim
|
||||
# Multi-stage, mirroring catalogue_frontend/Dockerfile: a build stage that
|
||||
# resolves dependencies, then a clean runtime stage that copies in only the
|
||||
# result. There it is `npm ci` -> dist/; here it is pip -> a virtualenv.
|
||||
|
||||
# ---- Build stage ----
|
||||
FROM python:3.11-slim AS build
|
||||
WORKDIR /app
|
||||
|
||||
# Dependencies land in a self-contained venv so the runtime stage can take that
|
||||
# one directory and leave pip, its HTTP cache and the downloaded wheels behind.
|
||||
#
|
||||
# requirements.txt is copied on its own, ahead of the source, for the same
|
||||
# reason the frontend stage copies package.json before the rest of the app:
|
||||
# this layer is cached on the file's checksum, so editing a router does not
|
||||
# reinstall torch.
|
||||
#
|
||||
# psycopg[binary] avoids needing libpq-dev; sentence-transformers/scikit-learn/
|
||||
# scipy all ship prebuilt wheels for this image, so no extra build toolchain
|
||||
# is needed. Playwright's Python package installs, but its browser binary is
|
||||
# NOT installed here - it's only a last-resort image-search fallback
|
||||
# (see requirements.txt); run `playwright install chromium` in the container
|
||||
# if you need that specific fallback tier.
|
||||
# scipy all ship prebuilt wheels for this image, so no compiler is needed and
|
||||
# this stage installs no build toolchain. Playwright's Python package installs,
|
||||
# but its browser binary is NOT installed here - it's only a last-resort
|
||||
# image-search fallback (see requirements.txt); run `playwright install
|
||||
# chromium` in the container if you need that specific fallback tier.
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
RUN python -m venv /opt/venv \
|
||||
&& /opt/venv/bin/pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
|
||||
# ---- Runtime stage ----
|
||||
FROM python:3.11-slim AS runtime
|
||||
WORKDIR /app
|
||||
|
||||
# PATH: putting the venv first is what makes a bare `python`/`uvicorn` resolve
|
||||
# to it - there is no "activate" step in a container.
|
||||
# PYTHONUNBUFFERED: without it Dokploy's log view stays empty until a buffer
|
||||
# happens to fill, so startup errors surface minutes after the container died.
|
||||
# PYTHONDONTWRITEBYTECODE: no .pyc to write into a read-only-ish image layer.
|
||||
ENV PATH="/opt/venv/bin:$PATH"
|
||||
ENV PYTHONUNBUFFERED=1
|
||||
ENV PYTHONDONTWRITEBYTECODE=1
|
||||
|
||||
COPY --from=build /opt/venv /opt/venv
|
||||
|
||||
COPY app ./app
|
||||
COPY cli ./cli
|
||||
COPY scripts ./scripts
|
||||
COPY data ./data
|
||||
COPY serve.py .
|
||||
|
||||
EXPOSE 8000
|
||||
# --proxy-headers/--forwarded-allow-ips: this container is never reached
|
||||
# directly - nginx (and, in prod, Caddy) sit in front of it. Without these,
|
||||
# uvicorn ignores X-Forwarded-Proto and reports every request as plain http,
|
||||
# so any redirect or generated absolute URL would downgrade an https request.
|
||||
CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000", \
|
||||
"--proxy-headers", "--forwarded-allow-ips", "*"]
|
||||
# The deployment's configuration, landing at /app/.env because settings.py
|
||||
# resolves it from the backend root - app/infrastructure/settings.py takes
|
||||
# parents[2], which is /app here - so it must sit next to app/, not inside it.
|
||||
#
|
||||
# The source file is named .env.production, NOT .env, and that detail is the
|
||||
# whole point. Dokploy writes its own .env into the build context from the
|
||||
# service's Environment tab AFTER cloning the repository. With that tab empty it
|
||||
# writes an empty file, overwriting the committed one - so `COPY .env .` copied
|
||||
# a zero-byte file, the container started with no configuration at all, and the
|
||||
# platform reported only a Bad Gateway. The checkout showed it plainly: every
|
||||
# file timestamped 08:33, and .env alone at 08:34, 0 bytes.
|
||||
#
|
||||
# Dokploy does not manage .env.production, so it survives. Anything set in the
|
||||
# Environment tab still wins at runtime, because settings.py calls load_dotenv()
|
||||
# without override=True and the process environment takes precedence.
|
||||
COPY .env.production .env
|
||||
|
||||
# Pristine copies of everything the app also WRITES to, kept at a path that is
|
||||
# never mounted over.
|
||||
#
|
||||
# /app/data/seed_catalogs and /app/app/intelligence/artifacts both need volumes
|
||||
# (products added through the UI are appended to the first, retrained models
|
||||
# are written to the second - otherwise a redeploy throws both away). But
|
||||
# mounting a volume there hides the copies shipped in this image: a *named*
|
||||
# volume is seeded from the image on first use, a *bind* mount is not, and
|
||||
# Dokploy offers both. A bind mount would leave the API with zero seed catalogs
|
||||
# and zero trained models, with nothing in the logs saying why.
|
||||
#
|
||||
# So keep a second copy here. On startup restore_bundled_assets()
|
||||
# (app/infrastructure/persistence.py) copies in whatever the mounted directory
|
||||
# is missing, and never overwrites what is already there.
|
||||
RUN mkdir -p /app/.bundled \
|
||||
&& cp -a /app/data/seed_catalogs /app/.bundled/seed_catalogs \
|
||||
&& cp -a /app/app/intelligence/artifacts /app/.bundled/artifacts
|
||||
|
||||
# Declared so `docker run` without an explicit -v still gets an anonymous
|
||||
# volume rather than writing into the container layer. Dokploy (and the compose
|
||||
# file) name them properly; this is the floor, not the recommended setup.
|
||||
VOLUME ["/app/data", "/app/app/intelligence/artifacts"]
|
||||
|
||||
# Answers on BOTH ports, the same way the frontend image does (nginx.conf has
|
||||
# `listen 80; listen 3000;`). 3000 is what Dokploy routes a domain to; 8000 is
|
||||
# what this project's README, the vite dev proxy and docker-compose all use.
|
||||
# Serving both means the container works whichever one the platform is pointed
|
||||
# at, instead of returning 502 from a perfectly healthy process.
|
||||
#
|
||||
# serve.py binds both sockets and hands them to one uvicorn - see the note
|
||||
# there. To pin a single port, set PORT (PORT=8080 serves only 8080); to change
|
||||
# the pair, set PORTS.
|
||||
ENV PORTS=3000,8000
|
||||
EXPOSE 3000 8000
|
||||
|
||||
# Liveness only, and passes if EITHER port answers.
|
||||
#
|
||||
# It probes "/", which is served from memory, NOT /api/health, which dials
|
||||
# Postgres and Ollama. That is the whole point: the platform's response to a
|
||||
# failed healthcheck is to stop routing traffic, so this may only ask "is the
|
||||
# process still serving HTTP". Tying it to the database meant an unreachable
|
||||
# Postgres blocked the handler for the OS TCP timeout, the check timed out, the
|
||||
# container was marked unhealthy, and a perfectly healthy API returned Bad
|
||||
# Gateway on every route. Use /api/health to ask whether dependencies are up;
|
||||
# it reports them in the body and always answers 200.
|
||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=40s --retries=3 \
|
||||
CMD ["python", "serve.py", "--healthcheck"]
|
||||
|
||||
# Exec form: python is PID 1, so Docker's SIGTERM reaches it directly and a
|
||||
# redeploy shuts down cleanly instead of waiting out the 10s kill timeout.
|
||||
CMD ["python", "serve.py"]
|
||||
|
||||
99
README.md
99
README.md
@@ -69,6 +69,105 @@ permission check; `user` holds the product/store/inventory permissions.
|
||||
`AUTH_ENABLED=false` disables all of it for local work — never in a deployment.
|
||||
See the Authentication section of `../DEPLOYMENT.md` for the full endpoint map.
|
||||
|
||||
## MCP server
|
||||
|
||||
The catalog is exposed to AI clients over the Model Context Protocol at `/mcp`,
|
||||
mounted onto this same FastAPI app (`app/mcp_server.py`) - no separate process
|
||||
or container, so it deploys with the API and answers on the same host.
|
||||
|
||||
**15 tools, all read-only.** Catalog search and browse, nutrition facts and
|
||||
health scores, healthier alternatives, per-store inventory and pricing,
|
||||
discounts, trending and sales analytics. None of the write or compute endpoints
|
||||
are reachable through MCP: a tool list is chosen from by a model rather than by
|
||||
a person, and "retrain the models" is not something to leave one tool call away.
|
||||
|
||||
**Auth is the app's own access token** - there is no separate MCP credential.
|
||||
Clients send `Authorization: Bearer <token>` from `POST /api/auth/login`, and
|
||||
the same `Principal` and expiry apply as on the REST side. Note the practical
|
||||
consequence: tokens expire after `AUTH_TOKEN_TTL_MINUTES` (12h by default), so a
|
||||
long-running client has to refresh. Raise the TTL if that is a problem.
|
||||
|
||||
```jsonc
|
||||
{
|
||||
"mcpServers": {
|
||||
"nearle-catalogue": {
|
||||
"type": "http",
|
||||
"url": "https://mcp.nearle.ai.in/mcp",
|
||||
"headers": { "Authorization": "Bearer <token>" }
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The admin UI has an inspector at `/mcp` (React route, admin-only) listing every
|
||||
tool with its arguments and a console to run one against live data. It is backed
|
||||
by `GET /api/mcp/info` and `POST /api/mcp/tools/{name}` rather than by the MCP
|
||||
endpoint itself - the browser would otherwise need a full MCP client to render a
|
||||
tool list.
|
||||
|
||||
`fastmcp` requires Python 3.10+. The image is 3.11; on an older interpreter the
|
||||
import fails, `app/main.py` logs a warning, and the REST API starts without the
|
||||
MCP endpoint rather than failing outright.
|
||||
|
||||
## Ports
|
||||
|
||||
The container answers on **3000 and 8000 at the same time**, the same way the
|
||||
frontend image answers on 80 and 3000. 3000 is what Dokploy routes a domain to;
|
||||
8000 is what this README, the vite dev proxy and `docker-compose.yml` use. Both
|
||||
being live means the deployment works whichever one it is pointed at, instead of
|
||||
returning 502 from a healthy container.
|
||||
|
||||
`serve.py` is what makes that possible - uvicorn's CLI binds a single `--port`,
|
||||
but `Server.run()` accepts a list of pre-bound sockets, so it is still one
|
||||
process. If one port is unavailable it logs and carries on with the other; it
|
||||
exits non-zero only when nothing is listening.
|
||||
|
||||
```bash
|
||||
python serve.py # 3000 and 8000
|
||||
PORT=8080 python serve.py # only 8080 (PORT pins a single port)
|
||||
PORTS=80,3000 python serve.py # a different pair
|
||||
```
|
||||
|
||||
For local development `uvicorn app.main:app --reload --port 8000` is still the
|
||||
normal thing to run - one port is all you need, and it gives you autoreload.
|
||||
|
||||
## Persistence: the two volumes a deployment needs
|
||||
|
||||
Most state lives in Postgres, but three things are written to the filesystem,
|
||||
and in a container those live inside the image - so a redeploy rebuilds the
|
||||
image and silently discards them:
|
||||
|
||||
| Path | Written by |
|
||||
|---|---|
|
||||
| `/app/data/seed_catalogs` | `POST /api/user/products/add`, `/upload-file` - every product added through the UI is appended to the brand's JSON |
|
||||
| `/app/app/intelligence/artifacts` | the training endpoints - every retrained `*.joblib` model |
|
||||
| `/app/data` | catalogs saved by the ingestion pipeline |
|
||||
|
||||
Mount a volume on each (the first is inside the third, so two mounts cover all
|
||||
three):
|
||||
|
||||
```
|
||||
/app/data
|
||||
/app/app/intelligence/artifacts
|
||||
```
|
||||
|
||||
In Dokploy, add both under the service's **Volumes**. Named volume or bind
|
||||
mount, either is fine - `docker compose --profile full up -d` shows the same
|
||||
two mounts as named volumes.
|
||||
|
||||
Bind mounts normally break this pattern, because they start empty and hide the
|
||||
seed catalogs and pre-trained models the image ships with. They are safe here:
|
||||
the image keeps read-only copies at `/app/.bundled`, and on startup
|
||||
`app/infrastructure/persistence.py` copies in whatever the mounted directory is
|
||||
missing. It never overwrites an existing file, so a user-added product always
|
||||
survives the next redeploy rather than being reverted to the bundled catalog.
|
||||
|
||||
If you leave the volumes off, the API still runs and logs a warning naming the
|
||||
directories that will be lost.
|
||||
|
||||
Override the locations with `DATA_DIR`, `SEED_CATALOG_DIR` and
|
||||
`MODEL_ARTIFACTS_DIR` if the writable data belongs somewhere else.
|
||||
|
||||
## Pulling the local LLM (one-time)
|
||||
|
||||
```bash
|
||||
|
||||
167
app/api/routers/mcp_info.py
Normal file
167
app/api/routers/mcp_info.py
Normal file
@@ -0,0 +1,167 @@
|
||||
"""
|
||||
REST inspector for the MCP server.
|
||||
|
||||
The MCP endpoint itself speaks streamable HTTP with session handling, which a
|
||||
browser cannot usefully talk to without a full MCP client implementation. Rather
|
||||
than ship one in React, these two endpoints expose what the admin page needs -
|
||||
what tools exist, and what one returns - as ordinary JSON over the REST API the
|
||||
frontend already authenticates against.
|
||||
|
||||
Admin-only. The tool inventory describes the whole catalog API surface, and the
|
||||
invoke endpoint executes a tool; neither is something to leave open just because
|
||||
the tools happen to be read-only.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from fastapi import APIRouter, Body, Depends, HTTPException, Request
|
||||
|
||||
from app.api.deps import require_admin
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
router = APIRouter(prefix="/mcp", tags=["mcp"])
|
||||
|
||||
|
||||
def _server():
|
||||
"""The FastMCP instance, or None when the optional dependency is absent."""
|
||||
try:
|
||||
from app.mcp_server import mcp
|
||||
|
||||
return mcp
|
||||
except Exception: # pragma: no cover - depends on fastmcp being installed
|
||||
return None
|
||||
|
||||
|
||||
def _public_url(request: Request, path: str) -> str:
|
||||
"""
|
||||
The URL an external MCP client should connect to.
|
||||
|
||||
Built from the forwarded headers rather than a configured constant so it is
|
||||
right on whichever host the request arrived on - the API's own domain, the
|
||||
frontend's nginx, or localhost in development. uvicorn runs with
|
||||
--proxy-headers, so request.url already reflects the external scheme.
|
||||
"""
|
||||
base = str(request.base_url).rstrip("/")
|
||||
return f"{base}{path}"
|
||||
|
||||
|
||||
@router.get("/info", dependencies=[Depends(require_admin)])
|
||||
async def mcp_info(request: Request) -> Dict[str, Any]:
|
||||
"""Describe the MCP server and list its tools, for the admin page."""
|
||||
mcp = _server()
|
||||
if mcp is None:
|
||||
return {
|
||||
"enabled": False,
|
||||
"reason": (
|
||||
"The fastmcp package is not installed, or failed to import. It "
|
||||
"requires Python 3.10 or newer."
|
||||
),
|
||||
"tools": [],
|
||||
}
|
||||
|
||||
from app.mcp_server import MCP_PATH
|
||||
|
||||
try:
|
||||
# run_middleware=False: the auth middleware reads an Authorization header
|
||||
# from an MCP request context that does not exist on this REST call. The
|
||||
# caller is already established as an admin by the route dependency.
|
||||
tools = await mcp.list_tools(run_middleware=False)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.exception("Could not list MCP tools")
|
||||
raise HTTPException(status_code=500, detail=f"Could not list MCP tools: {exc}")
|
||||
|
||||
described: List[Dict[str, Any]] = []
|
||||
for tool in sorted(tools, key=lambda t: t.name):
|
||||
# Server-side FunctionTool exposes its JSON Schema as `parameters`.
|
||||
# `inputSchema` is the wire-format name, present on the client-side Tool
|
||||
# an MCP client receives - checked second so this keeps working if a
|
||||
# future version converges on one name.
|
||||
schema = getattr(tool, "parameters", None) or getattr(tool, "inputSchema", None) or {}
|
||||
properties = schema.get("properties", {}) or {}
|
||||
required = set(schema.get("required", []) or [])
|
||||
described.append(
|
||||
{
|
||||
"name": tool.name,
|
||||
"description": (tool.description or "").strip(),
|
||||
"tags": sorted(getattr(tool, "tags", set()) or []),
|
||||
"parameters": [
|
||||
{
|
||||
"name": pname,
|
||||
"type": pinfo.get("type") or "any",
|
||||
"description": (pinfo.get("description") or "").strip(),
|
||||
"required": pname in required,
|
||||
"default": pinfo.get("default"),
|
||||
}
|
||||
for pname, pinfo in properties.items()
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
return {
|
||||
"enabled": True,
|
||||
"name": mcp.name,
|
||||
"version": getattr(mcp, "version", None),
|
||||
"url": _public_url(request, MCP_PATH),
|
||||
"transport": "http",
|
||||
"auth": "Bearer token from POST /api/auth/login",
|
||||
"tool_count": len(described),
|
||||
"tools": described,
|
||||
}
|
||||
|
||||
|
||||
@router.post("/tools/{tool_name}", dependencies=[Depends(require_admin)])
|
||||
async def invoke_tool(
|
||||
tool_name: str,
|
||||
arguments: Optional[Dict[str, Any]] = Body(default=None),
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Run one MCP tool and return its result - the admin page's "try it" console.
|
||||
|
||||
Exists so the server can be checked from the browser without installing an
|
||||
MCP client. Every tool is read-only, so this executes the same call an AI
|
||||
client would make, with the same code path.
|
||||
"""
|
||||
mcp = _server()
|
||||
if mcp is None:
|
||||
raise HTTPException(status_code=503, detail="The MCP server is not available.")
|
||||
|
||||
try:
|
||||
# Same reasoning as above: this request authenticated through the REST
|
||||
# guard, so the MCP auth middleware would have no headers to inspect.
|
||||
tools = {t.name: t for t in await mcp.list_tools(run_middleware=False)}
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise HTTPException(status_code=500, detail=f"Could not list MCP tools: {exc}")
|
||||
|
||||
if tool_name not in tools:
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail=f"No MCP tool named '{tool_name}'. Known tools: {', '.join(sorted(tools))}.",
|
||||
)
|
||||
|
||||
try:
|
||||
result = await mcp.call_tool(tool_name, arguments or {}, run_middleware=False)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
# A tool raising ToolError is a normal outcome worth showing verbatim -
|
||||
# "no nutrition data for this product" is an answer, not a server fault.
|
||||
return {"tool": tool_name, "ok": False, "error": str(exc)}
|
||||
|
||||
return {"tool": tool_name, "ok": True, "result": _jsonable(result)}
|
||||
|
||||
|
||||
def _jsonable(result: Any) -> Any:
|
||||
"""Reduce a ToolResult to something FastAPI can serialise."""
|
||||
for attr in ("structured_content", "data"):
|
||||
value = getattr(result, attr, None)
|
||||
if value is not None:
|
||||
return value
|
||||
|
||||
content = getattr(result, "content", None)
|
||||
if content is None:
|
||||
return result
|
||||
out = []
|
||||
for block in content:
|
||||
text = getattr(block, "text", None)
|
||||
out.append(text if text is not None else str(block))
|
||||
return out
|
||||
@@ -1,6 +1,7 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import uuid
|
||||
import logging
|
||||
import pandas as pd
|
||||
|
||||
@@ -7,6 +7,7 @@ from pydantic import BaseModel, Field
|
||||
from fastapi import APIRouter, Depends, HTTPException, File, UploadFile
|
||||
|
||||
from app.api.deps import require_permission
|
||||
from app.infrastructure.settings import SEED_CATALOG_DIR
|
||||
|
||||
from app.services.vector_store import (
|
||||
upsert_brand_products,
|
||||
|
||||
@@ -19,7 +19,7 @@ sys.path.append(str(Path(__file__).parent.parent))
|
||||
|
||||
from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details
|
||||
from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts
|
||||
from app.infrastructure.settings import USE_OLLAMA
|
||||
from app.infrastructure.settings import DATA_DIR, USE_OLLAMA
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id
|
||||
from app.services.s3_service import s3_service
|
||||
@@ -948,20 +948,30 @@ class ProductCatalogEngine:
|
||||
return catalog
|
||||
|
||||
def save_catalog(self, catalog: Dict[str, Any], filename: str = None) -> str:
|
||||
"""Save catalog to JSON file inside data/ folder"""
|
||||
"""Save catalog to JSON file inside DATA_DIR.
|
||||
|
||||
Resolved against DATA_DIR rather than a bare relative "data/" path: the
|
||||
latter depends on the process's working directory, so the same call
|
||||
landed in a different place depending on whether the app was started
|
||||
from the repo root, from backend/, or by the container's uvicorn - and
|
||||
only one of those is the directory with a volume mounted on it.
|
||||
"""
|
||||
if not filename:
|
||||
brand = catalog.get('brand', 'unknown')
|
||||
storage_brand = resolve_parent_brand(brand)
|
||||
safe_brand = storage_brand.replace(' ', '_')
|
||||
ts = catalog.get('generation_timestamp', 'latest')
|
||||
filename = f"data/catalog_{safe_brand}_{ts}.json"
|
||||
|
||||
Path("data").mkdir(exist_ok=True)
|
||||
with open(filename, 'w', encoding='utf-8') as f:
|
||||
filename = f"catalog_{safe_brand}_{ts}.json"
|
||||
|
||||
path = Path(filename)
|
||||
if not path.is_absolute():
|
||||
path = DATA_DIR / path
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(path, 'w', encoding='utf-8') as f:
|
||||
json.dump(catalog, f, indent=2, ensure_ascii=False)
|
||||
|
||||
logger.info(f"💾 Catalog saved to: {filename}")
|
||||
return filename
|
||||
|
||||
logger.info(f"💾 Catalog saved to: {path}")
|
||||
return str(path)
|
||||
|
||||
# Global engine instance
|
||||
catalog_engine = ProductCatalogEngine()
|
||||
|
||||
146
app/infrastructure/persistence.py
Normal file
146
app/infrastructure/persistence.py
Normal file
@@ -0,0 +1,146 @@
|
||||
"""
|
||||
First-run restore of the assets bundled into the container image.
|
||||
|
||||
The problem this solves
|
||||
----------------------
|
||||
Three directories are written to at runtime - the seed catalogs appended to by
|
||||
``POST /api/user/products/add``, the generated catalogs from the ingestion
|
||||
pipeline, and the ``*.joblib`` bundles the training endpoints produce. In a
|
||||
container all three sit inside the image, so a redeploy silently discards
|
||||
every one of them. They need to be on a volume.
|
||||
|
||||
Mounting a volume over them introduces the opposite problem. A *named* Docker
|
||||
volume is seeded from the image the first time it is used, but a *bind* mount
|
||||
starts empty and merely hides what the image had underneath. Dokploy offers
|
||||
both and neither announces which you picked, so a bind mount on ``/app/data``
|
||||
would leave the API running with no seed catalogs: the next product added would
|
||||
write a JSON file containing only that product, and every model would report as
|
||||
untrained.
|
||||
|
||||
The fix is to keep a pristine copy inside the image at a path nobody mounts
|
||||
(``BUNDLED_ASSETS_DIR``, populated by the Dockerfile) and top up the writable
|
||||
directory from it on startup.
|
||||
|
||||
Existing files are never overwritten. That is the whole contract: the bundle
|
||||
supplies what is missing, and anything the running app has already written wins
|
||||
over the copy baked into the image. Without that rule every redeploy would
|
||||
revert user-added products back to the bundled catalog.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
from typing import Tuple
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
BUNDLED_ASSETS_DIR,
|
||||
DATA_DIR,
|
||||
MODEL_ARTIFACTS_DIR,
|
||||
SEED_CATALOG_DIR,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# (subdirectory under BUNDLED_ASSETS_DIR, writable destination)
|
||||
_BUNDLES: Tuple[Tuple[str, Path], ...] = (
|
||||
("seed_catalogs", SEED_CATALOG_DIR),
|
||||
("artifacts", MODEL_ARTIFACTS_DIR),
|
||||
)
|
||||
|
||||
|
||||
def _restore_one(source: Path, destination: Path) -> int:
|
||||
"""Copy files missing from ``destination``. Returns how many were copied."""
|
||||
if not source.is_dir():
|
||||
return 0
|
||||
|
||||
destination.mkdir(parents=True, exist_ok=True)
|
||||
copied = 0
|
||||
for item in sorted(source.iterdir()):
|
||||
if not item.is_file():
|
||||
continue
|
||||
target = destination / item.name
|
||||
if target.exists():
|
||||
continue
|
||||
try:
|
||||
# copy2 rather than copy: it preserves mtime, so "is this artifact
|
||||
# older than the data it was trained on" stays answerable.
|
||||
shutil.copy2(item, target)
|
||||
copied += 1
|
||||
except OSError as exc:
|
||||
logger.warning("Could not restore %s -> %s: %s", item, target, exc)
|
||||
return copied
|
||||
|
||||
|
||||
def restore_bundled_assets() -> None:
|
||||
"""
|
||||
Top up the writable directories from the image's read-only bundle.
|
||||
|
||||
Safe to call on every boot: it is a no-op once the volume is populated, and
|
||||
a no-op outside Docker where BUNDLED_ASSETS_DIR does not exist.
|
||||
"""
|
||||
# Create these regardless. A volume mounted at DATA_DIR arrives empty, and
|
||||
# the ingestion pipeline writes into it without creating it first.
|
||||
for path in (DATA_DIR, SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR):
|
||||
try:
|
||||
path.mkdir(parents=True, exist_ok=True)
|
||||
except OSError as exc:
|
||||
logger.error(
|
||||
"Cannot create writable directory %s: %s. Uploads and trained "
|
||||
"models will fail to save - check the volume's permissions.",
|
||||
path,
|
||||
exc,
|
||||
)
|
||||
|
||||
if not BUNDLED_ASSETS_DIR.is_dir():
|
||||
logger.debug(
|
||||
"No bundled asset directory at %s - nothing to restore.", BUNDLED_ASSETS_DIR
|
||||
)
|
||||
return
|
||||
|
||||
for name, destination in _BUNDLES:
|
||||
copied = _restore_one(BUNDLED_ASSETS_DIR / name, destination)
|
||||
if copied:
|
||||
logger.info(
|
||||
"Restored %d bundled file(s) into %s (first run on this volume).",
|
||||
copied,
|
||||
destination,
|
||||
)
|
||||
|
||||
_warn_if_not_persistent()
|
||||
|
||||
|
||||
def _warn_if_not_persistent() -> None:
|
||||
"""
|
||||
Point out that the writable directories are still inside the image.
|
||||
|
||||
Only reachable when BUNDLED_ASSETS_DIR exists, i.e. in the container. If the
|
||||
destinations were never mounted, everything written to them is lost on the
|
||||
next redeploy - which looks exactly like the app quietly ignoring uploads,
|
||||
hours later and with nothing in the logs to connect it to.
|
||||
"""
|
||||
unmounted = [p for p in (SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR) if not _is_mount(p)]
|
||||
if unmounted:
|
||||
logger.warning(
|
||||
"These directories are written at runtime but do not look like "
|
||||
"mount points: %s. Anything saved there - products added through "
|
||||
"the UI, retrained models - will be discarded on the next "
|
||||
"redeploy. Mount a volume on each (see backend/README.md).",
|
||||
", ".join(str(p) for p in unmounted),
|
||||
)
|
||||
|
||||
|
||||
def _is_mount(path: Path) -> bool:
|
||||
"""
|
||||
Whether ``path`` sits on a different device than its parent.
|
||||
|
||||
A mounted volume shows up as a device-number change. Falls back to True on
|
||||
error so a probe failure produces silence rather than a false alarm telling
|
||||
somebody their correctly-mounted volume is broken.
|
||||
"""
|
||||
try:
|
||||
if path.is_mount():
|
||||
return True
|
||||
return path.stat().st_dev != path.parent.stat().st_dev
|
||||
except OSError:
|
||||
return True
|
||||
@@ -55,6 +55,52 @@ def _require(name: str, *, feature_flag: str) -> str:
|
||||
return value
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Writable data directories (persistence)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Everything the running app WRITES lives under one of these three paths. They
|
||||
# are settings rather than hard-coded paths because in a container they must be
|
||||
# mounted on a volume - otherwise every product added through the UI and every
|
||||
# retrained model is discarded the next time the image is redeployed.
|
||||
#
|
||||
# DATA_DIR generated catalogs (catalog_engine.save_catalog)
|
||||
# SEED_CATALOG_DIR per-brand JSON catalogs, appended to by
|
||||
# POST /api/user/products/add and /upload-file
|
||||
# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints
|
||||
#
|
||||
# See BUNDLED_ASSETS_DIR below for how the read-only copies shipped inside the
|
||||
# image get into these directories the first time a volume is mounted.
|
||||
_BACKEND_ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def _dir(name: str, default: Path) -> Path:
|
||||
raw = os.getenv(name, "").strip()
|
||||
return Path(raw).expanduser() if raw else default
|
||||
|
||||
|
||||
DATA_DIR = _dir("DATA_DIR", _BACKEND_ROOT / "data")
|
||||
SEED_CATALOG_DIR = _dir("SEED_CATALOG_DIR", DATA_DIR / "seed_catalogs")
|
||||
MODEL_ARTIFACTS_DIR = _dir(
|
||||
"MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts"
|
||||
)
|
||||
|
||||
# Pristine copies of the bundled seed catalogs and pre-trained models, placed
|
||||
# here by the Dockerfile at a path that is never itself mounted over.
|
||||
#
|
||||
# This exists because the two ways of mounting a volume behave differently, and
|
||||
# the difference is silent. Docker copies the image's content into a *named*
|
||||
# volume the first time it is used, but a *bind* mount starts empty and simply
|
||||
# hides whatever the image had at that path. Mounting a bind mount on /app/data
|
||||
# would therefore leave the app with no seed catalogs at all: the next product
|
||||
# added would write a fresh JSON file containing only that one product, and the
|
||||
# ML endpoints would report no trained models.
|
||||
#
|
||||
# So the image keeps a second, unmounted copy, and restore_bundled_assets()
|
||||
# (app/infrastructure/persistence.py) fills in whatever the writable directory
|
||||
# is missing at startup. Empty/absent outside Docker, where nothing is mounted
|
||||
# and the defaults above already point at the real files.
|
||||
BUNDLED_ASSETS_DIR = _dir("BUNDLED_ASSETS_DIR", Path("/app/.bundled"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ollama (local LLM)
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -81,6 +127,14 @@ DB_PORT = os.getenv("DB_PORT", "5432")
|
||||
DB_NAME = os.getenv("DB_NAME", "pgvector")
|
||||
DB_USER = os.getenv("DB_USER", "postgres")
|
||||
DB_PASSWORD = _require("DB_PASSWORD", feature_flag="USE_PGVECTOR") if USE_PGVECTOR else os.getenv("DB_PASSWORD", "")
|
||||
# How long to wait for the TCP connect before giving up. Matters more than it
|
||||
# looks: a host that DROPS packets (a firewall, a typo'd DB_HOST) otherwise
|
||||
# blocks until the OS timeout - about 130 seconds on Linux - and every request
|
||||
# that touches the database inherits that wait, including /api/health. A short
|
||||
# ceiling turns "the database is unreachable" into a fast, honest error instead
|
||||
# of a hung worker and a container the platform decides is unhealthy.
|
||||
DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5"))
|
||||
|
||||
DATABASE_URL = os.getenv(
|
||||
"DATABASE_URL",
|
||||
f"postgresql://{DB_USER}:{DB_PASSWORD}@{DB_HOST}:{DB_PORT}/{DB_NAME}",
|
||||
|
||||
@@ -13,9 +13,15 @@ from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from app.infrastructure.settings import MODEL_ARTIFACTS_DIR
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
ARTIFACTS_DIR = Path(__file__).resolve().parent / "artifacts"
|
||||
# From settings so it can be pointed at a mounted volume: retraining writes
|
||||
# *.joblib here, and inside the image those are discarded on the next redeploy,
|
||||
# silently reverting every model to the version baked in at build time.
|
||||
# Defaults to this package's own artifacts/ directory.
|
||||
ARTIFACTS_DIR = MODEL_ARTIFACTS_DIR
|
||||
ARTIFACTS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
|
||||
|
||||
98
app/main.py
98
app/main.py
@@ -9,6 +9,7 @@ Run with:
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import threading
|
||||
from contextlib import asynccontextmanager
|
||||
from pathlib import Path
|
||||
@@ -18,11 +19,12 @@ from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from fastapi.responses import FileResponse
|
||||
|
||||
from app.infrastructure.persistence import restore_bundled_assets
|
||||
from app.infrastructure.settings import API_CORS_ORIGINS
|
||||
from app.api.routers import health, brands, search, chat, catalog, system
|
||||
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
|
||||
from app.api.routers import nutrition, nutrition_admin, upload
|
||||
from app.api.routers import auth, user_products, admin_train
|
||||
from app.api.routers import auth, user_products, admin_train, mcp_info
|
||||
from app.services.store_db import ensure_store_intelligence_schema
|
||||
from app.services.nutrition_db import ensure_nutrition_schema
|
||||
|
||||
@@ -32,6 +34,18 @@ logging.basicConfig(
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
from app.mcp_server import MCP_PATH, build_http_app as build_mcp_app
|
||||
|
||||
_mcp_app = build_mcp_app()
|
||||
except Exception as e: # pragma: no cover - depends on an optional dependency
|
||||
# fastmcp needs Python >=3.10. The image is 3.11, but an older local
|
||||
# interpreter should lose the MCP endpoint, not the whole API.
|
||||
logging.getLogger(__name__).warning("MCP server unavailable: %s", e)
|
||||
_mcp_app = None
|
||||
MCP_PATH = "/mcp"
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(_app: FastAPI):
|
||||
"""
|
||||
@@ -43,6 +57,15 @@ async def lifespan(_app: FastAPI):
|
||||
healthcheck pass while that settles.
|
||||
"""
|
||||
|
||||
# Runs before the thread below, and synchronously: the seed catalogs and
|
||||
# model artifacts have to be in place before the first request can read
|
||||
# them, and it is a handful of file copies on first boot, nothing on every
|
||||
# boot after that.
|
||||
try:
|
||||
restore_bundled_assets()
|
||||
except Exception as e:
|
||||
logger.warning("Could not restore bundled assets: %s", e)
|
||||
|
||||
def _async_init():
|
||||
try:
|
||||
ensure_store_intelligence_schema()
|
||||
@@ -58,7 +81,17 @@ async def lifespan(_app: FastAPI):
|
||||
logger.warning("Startup background init warning: %s", e)
|
||||
|
||||
threading.Thread(target=_async_init, daemon=True).start()
|
||||
yield
|
||||
|
||||
# The mounted MCP app carries its own lifespan, which starts the session
|
||||
# manager its request handler depends on. Mounting a sub-app does NOT run
|
||||
# that lifespan - only the outermost app's is executed - so without this
|
||||
# the endpoint exists, accepts a connection, and then fails on the first
|
||||
# message with a session manager that was never started.
|
||||
if _mcp_app is not None:
|
||||
async with _mcp_app.lifespan(_mcp_app):
|
||||
yield
|
||||
else:
|
||||
yield
|
||||
|
||||
|
||||
app = FastAPI(
|
||||
@@ -98,6 +131,24 @@ app.add_middleware(
|
||||
allow_headers=["*"],
|
||||
)
|
||||
|
||||
# A wrong origin list fails only in the browser, as an opaque "blocked by CORS"
|
||||
# with a perfectly healthy 200 in the server log - so state the effective list
|
||||
# at startup, where it can actually be compared against the frontend's URL.
|
||||
logger.info(
|
||||
"CORS allowed origins: %s",
|
||||
", ".join(API_CORS_ORIGINS) if API_CORS_ORIGINS else "(none - same-origin only)",
|
||||
)
|
||||
if API_CORS_ORIGINS and all(
|
||||
o.startswith(("http://localhost", "http://127.0.0.1")) for o in API_CORS_ORIGINS
|
||||
):
|
||||
logger.warning(
|
||||
"API_CORS_ORIGINS lists only localhost origins (%s). If this API is "
|
||||
"published on a domain and the frontend is served from a different one, "
|
||||
"every browser request will be blocked. Set API_CORS_ORIGINS to the "
|
||||
"frontend's exact origin, e.g. https://catalogue.nearle.ai.in",
|
||||
", ".join(API_CORS_ORIGINS),
|
||||
)
|
||||
|
||||
app.include_router(health.router, prefix="/api")
|
||||
app.include_router(auth.router, prefix="/api")
|
||||
app.include_router(user_products.router, prefix="/api")
|
||||
@@ -116,10 +167,44 @@ app.include_router(store_admin.router, prefix="/api")
|
||||
app.include_router(nutrition.router, prefix="/api")
|
||||
app.include_router(nutrition_admin.router, prefix="/api")
|
||||
app.include_router(upload.router, prefix="/api")
|
||||
app.include_router(mcp_info.router, prefix="/api")
|
||||
|
||||
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,
|
||||
# not part of the REST surface, and the catch-all SPA route below only skips
|
||||
# paths it recognises. Mounted rather than routed because it is a whole ASGI
|
||||
# app with its own request handling.
|
||||
if _mcp_app is not None:
|
||||
app.mount(MCP_PATH, _mcp_app)
|
||||
logger.info("MCP server mounted at %s", MCP_PATH)
|
||||
|
||||
# Serve built frontend static files if dist exists (single-port unified
|
||||
# deployment). Off in the normal setup: the React app is served by its own
|
||||
# nginx on catalogue.nearle.ai.in and this API answers on
|
||||
# mcp.nearle.ai.in, so no dist/ is present here and the JSON root
|
||||
# handler at the bottom of this file is what responds to /.
|
||||
#
|
||||
# The candidates cover both repo layouts - the sibling checkout is named
|
||||
# `catalogue_frontend`, and only `frontend` was checked before, so this branch
|
||||
# could never activate even when a build was sitting right next to it.
|
||||
# FRONTEND_DIST_DIR overrides both when the build lands somewhere else.
|
||||
_dist_override = os.getenv("FRONTEND_DIST_DIR", "").strip()
|
||||
_repo_root = Path(__file__).resolve().parents[2]
|
||||
_dist_candidates = (
|
||||
[Path(_dist_override)]
|
||||
if _dist_override
|
||||
else [
|
||||
_repo_root / "catalogue_frontend" / "dist",
|
||||
_repo_root / "frontend" / "dist",
|
||||
Path(__file__).resolve().parents[1] / "frontend" / "dist",
|
||||
]
|
||||
)
|
||||
FRONTEND_DIST = next(
|
||||
(p for p in _dist_candidates if (p / "assets").exists()),
|
||||
_dist_candidates[0],
|
||||
)
|
||||
|
||||
# Serve built frontend static files if dist exists (single-port unified deployment)
|
||||
FRONTEND_DIST = Path(__file__).resolve().parents[2] / "frontend" / "dist"
|
||||
if FRONTEND_DIST.exists() and (FRONTEND_DIST / "assets").exists():
|
||||
logger.info("Serving built frontend from %s", FRONTEND_DIST)
|
||||
app.mount("/assets", StaticFiles(directory=str(FRONTEND_DIST / "assets")), name="assets")
|
||||
|
||||
@app.get("/{full_path:path}")
|
||||
@@ -130,7 +215,10 @@ if FRONTEND_DIST.exists() and (FRONTEND_DIST / "assets").exists():
|
||||
# endpoint would look like a successful call to any client. 404 is the
|
||||
# honest answer, and it matters now that the API is also reachable
|
||||
# directly at api.{$DOMAIN} rather than only behind the frontend.
|
||||
if full_path.startswith(("api", "docs", "redoc", "openapi.json")):
|
||||
# "mcp" is in this list for the case where the MCP app failed to load:
|
||||
# the mount would be absent, and without this the SPA fallback would
|
||||
# answer an MCP client with index.html and a 200.
|
||||
if full_path.startswith(("api", "docs", "redoc", "openapi.json", "mcp")):
|
||||
raise HTTPException(status_code=404, detail="Not found")
|
||||
file_path = FRONTEND_DIST / full_path
|
||||
if file_path.exists() and file_path.is_file():
|
||||
|
||||
522
app/mcp_server.py
Normal file
522
app/mcp_server.py
Normal file
@@ -0,0 +1,522 @@
|
||||
"""
|
||||
MCP server exposing the catalog as tools for AI clients.
|
||||
|
||||
Mounted onto the FastAPI app at /mcp (see app/main.py), so it ships in the same
|
||||
container and answers on the same host - mcp.nearle.ai.in/mcp.
|
||||
|
||||
Scope: reads only. Every tool here maps to a service function the REST API
|
||||
already exposes through a GET. None of the write or compute endpoints - catalog
|
||||
generation, ML training, the upload endpoints, product creation - are reachable
|
||||
through MCP, because a tool list is chosen from by a model rather than by a
|
||||
person, and "retrain the models" is not a thing to leave one tool call away.
|
||||
Adding a write tool later is deliberate work, not an oversight to correct.
|
||||
|
||||
Authentication reuses the access token from POST /api/auth/login. There is no
|
||||
separate MCP credential: the same Principal, the same expiry, the same
|
||||
revocation story as the REST API, and nothing new to keep secret. Clients send
|
||||
it as an Authorization: Bearer header, which _AuthMiddleware checks once for
|
||||
every tool call and tool listing.
|
||||
|
||||
The tools are hand-written rather than generated from the OpenAPI schema. All
|
||||
63 routes would technically work, but a model picks a tool by reading its
|
||||
description, and 63 near-identical auto-generated entries is a worse starting
|
||||
point than a dozen written to be chosen between.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from fastmcp import FastMCP
|
||||
from fastmcp.exceptions import ToolError
|
||||
from fastmcp.server.dependencies import get_http_headers
|
||||
from fastmcp.server.middleware import CallNext, Middleware, MiddlewareContext
|
||||
|
||||
from app.infrastructure.security import AuthError, Principal, anonymous_principal, decode_access_token
|
||||
from app.infrastructure.settings import AUTH_ENABLED
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MCP_PATH = "/mcp"
|
||||
|
||||
INSTRUCTIONS = """\
|
||||
Read-only access to an Indian FMCG product catalog: products and brands, \
|
||||
nutrition facts and health scores, per-store inventory and pricing, discounts, \
|
||||
and sales analytics.
|
||||
|
||||
Products are identified by a (brand, image_id) pair - image_id is the SKU-like \
|
||||
product key, not an image URL. Get one from search_products or \
|
||||
list_brand_products before calling any tool that takes an image_id.
|
||||
|
||||
Nutrition figures are per 100g unless the product states otherwise, and health \
|
||||
scores run 0-100, higher being better. Prices are in INR.
|
||||
"""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Authentication
|
||||
# ---------------------------------------------------------------------------
|
||||
def _principal_from_headers() -> Principal:
|
||||
"""
|
||||
Resolve the caller from the Authorization header of the current request.
|
||||
|
||||
Raises ToolError rather than AuthError: ToolError is what reaches the client
|
||||
as a readable message instead of an opaque internal failure.
|
||||
"""
|
||||
if not AUTH_ENABLED:
|
||||
return anonymous_principal()
|
||||
|
||||
# include={"authorization"} is required, not optional: get_http_headers()
|
||||
# strips the Authorization header by default, so that it is not forwarded to
|
||||
# downstream services by accident. Without this the header is invisible here
|
||||
# and every request looks unauthenticated - including valid ones.
|
||||
headers = get_http_headers(include={"authorization"})
|
||||
raw = headers.get("authorization", "") # keys are lowercased by the helper
|
||||
scheme, _, token = raw.partition(" ")
|
||||
|
||||
if scheme.lower() != "bearer" or not token.strip():
|
||||
raise ToolError(
|
||||
"Not authenticated. Send an Authorization: Bearer <token> header. "
|
||||
"Get a token from POST /api/auth/login."
|
||||
)
|
||||
try:
|
||||
return decode_access_token(token.strip())
|
||||
except AuthError as exc:
|
||||
raise ToolError(str(exc)) from exc
|
||||
|
||||
|
||||
class _AuthMiddleware(Middleware):
|
||||
"""
|
||||
One authentication check covering every tool.
|
||||
|
||||
Sitting here rather than at the top of each tool means a tool added later
|
||||
cannot be left unguarded by forgetting a line - the same reason the REST
|
||||
side puts its guard in a route dependency instead of the handler body.
|
||||
|
||||
Listing is checked as well as calling. An unauthenticated client learning
|
||||
the exact shape of the catalog API is a smaller problem than an
|
||||
unauthenticated call, but it is not nothing, and there is no reason to
|
||||
publish it.
|
||||
"""
|
||||
|
||||
async def on_call_tool(self, context: MiddlewareContext, call_next: CallNext):
|
||||
principal = _principal_from_headers()
|
||||
name = getattr(context.message, "name", "?")
|
||||
logger.info("MCP tool call %r by %s (%s)", name, principal.username, principal.kind)
|
||||
return await call_next(context)
|
||||
|
||||
async def on_list_tools(self, context: MiddlewareContext, call_next: CallNext):
|
||||
_principal_from_headers()
|
||||
return await call_next(context)
|
||||
|
||||
|
||||
mcp: FastMCP = FastMCP(
|
||||
name="nearle-catalogue",
|
||||
instructions=INSTRUCTIONS,
|
||||
version="1.0.0",
|
||||
middleware=[_AuthMiddleware()],
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
def _guard(what: str, fn, *args, **kwargs):
|
||||
"""
|
||||
Run a service call, turning failures into a ToolError the client can read.
|
||||
|
||||
Nearly every tool here reads Postgres, so the common failure is the database
|
||||
being unreachable. Left to propagate, that surfaces to the model as an
|
||||
unexplained error; named, the model can say what went wrong instead of
|
||||
retrying or inventing an answer.
|
||||
"""
|
||||
try:
|
||||
return fn(*args, **kwargs)
|
||||
except Exception as exc: # noqa: BLE001 - deliberately broad, see docstring
|
||||
logger.exception("MCP tool %s failed", what)
|
||||
raise ToolError(f"{what} failed: {exc}") from exc
|
||||
|
||||
|
||||
def _clip(n: Optional[int], default: int, maximum: int) -> int:
|
||||
"""Keep result counts sane - a model will happily ask for 10000 rows."""
|
||||
if n is None:
|
||||
return default
|
||||
return max(1, min(int(n), maximum))
|
||||
|
||||
|
||||
def _slim(product: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""
|
||||
Drop the fields a model has no use for.
|
||||
|
||||
Embeddings especially: a 384-float vector per product would swamp the
|
||||
context window and says nothing a model can reason about.
|
||||
"""
|
||||
return {
|
||||
k: v
|
||||
for k, v in product.items()
|
||||
if k not in {"embedding", "embedding_text", "distance"} and v not in (None, "", [], {})
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Catalog
|
||||
# ---------------------------------------------------------------------------
|
||||
@mcp.tool(
|
||||
annotations={"readOnlyHint": True},
|
||||
tags={"catalog"},
|
||||
)
|
||||
def search_products(
|
||||
query: str,
|
||||
brand: Optional[str] = None,
|
||||
category: Optional[str] = None,
|
||||
limit: Optional[int] = 10,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Search the catalog by meaning, not keywords.
|
||||
|
||||
The main entry point: use this to turn a description ("sugar-free biscuits",
|
||||
"something to replace butter") into concrete products. Returns each match
|
||||
with its brand and image_id, which the nutrition, store and recommendation
|
||||
tools take as input.
|
||||
|
||||
Args:
|
||||
query: What to look for, in natural language.
|
||||
brand: Restrict to one brand. Omit to search everything.
|
||||
category: Restrict to one category, e.g. "Dairy".
|
||||
limit: How many products to return (1-50).
|
||||
"""
|
||||
from app.services.rag_service import retrieve
|
||||
|
||||
top_k = _clip(limit, 10, 50)
|
||||
found = _guard("search_products", retrieve, query, brand=brand, top_k=top_k, category=category)
|
||||
return [
|
||||
_slim(
|
||||
{
|
||||
"brand": p.brand,
|
||||
"image_id": p.image_id,
|
||||
"title": p.title,
|
||||
"category": p.category,
|
||||
"description": p.description,
|
||||
"price_range": p.price_range,
|
||||
"selling_price": p.final_selling_price or p.selling_price,
|
||||
"size_variants": p.size_variants,
|
||||
"barcode": p.barcode,
|
||||
}
|
||||
)
|
||||
for p in found
|
||||
]
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
||||
def list_brands() -> List[str]:
|
||||
"""List every brand in the catalog.
|
||||
|
||||
Useful for grounding a brand name before passing it to another tool, since
|
||||
brand names must match the catalog's spelling.
|
||||
"""
|
||||
from app.services.vector_store import list_available_brands
|
||||
|
||||
return _guard("list_brands", list_available_brands)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
||||
def list_categories(brand: str) -> List[str]:
|
||||
"""List the product categories a brand sells.
|
||||
|
||||
Args:
|
||||
brand: Brand name, as returned by list_brands.
|
||||
"""
|
||||
from app.services.vector_store import list_categories_for_brand
|
||||
|
||||
return _guard("list_categories", list_categories_for_brand, brand)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
||||
def list_brand_products(
|
||||
brand: str,
|
||||
category: Optional[str] = None,
|
||||
limit: Optional[int] = 25,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""List a brand's products, newest catalog entries first.
|
||||
|
||||
Browsing, as opposed to search_products' semantic matching. Prefer
|
||||
search_products when the user described what they want rather than naming a
|
||||
brand outright.
|
||||
|
||||
Args:
|
||||
brand: Brand name, as returned by list_brands.
|
||||
category: Restrict to one category.
|
||||
limit: How many products to return (1-100).
|
||||
"""
|
||||
from app.services.vector_store import get_products_by_brand
|
||||
|
||||
rows = _guard(
|
||||
"list_brand_products",
|
||||
get_products_by_brand,
|
||||
brand,
|
||||
limit=_clip(limit, 25, 100),
|
||||
category=category,
|
||||
)
|
||||
return [_slim(r) for r in rows]
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
||||
def get_product(brand: str, image_id: str) -> Dict[str, Any]:
|
||||
"""Get the full catalog record for one product.
|
||||
|
||||
Args:
|
||||
brand: Brand name.
|
||||
image_id: The product key from search_products or list_brand_products.
|
||||
"""
|
||||
from app.services.vector_store import get_product_by_image_id
|
||||
|
||||
row = _guard("get_product", get_product_by_image_id, brand, image_id)
|
||||
if not row:
|
||||
raise ToolError(
|
||||
f"No product {image_id!r} for brand {brand!r}. "
|
||||
"Check the pair with search_products."
|
||||
)
|
||||
return _slim(row)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Nutrition
|
||||
# ---------------------------------------------------------------------------
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
|
||||
def get_nutrition(brand: str, image_id: str) -> Dict[str, Any]:
|
||||
"""Get nutrition facts, health score and dietary flags for one product.
|
||||
|
||||
Covers macros and key micronutrients per 100g, a 0-100 health score, diet
|
||||
tags (vegetarian, high-protein, ...) and declared allergens.
|
||||
|
||||
Args:
|
||||
brand: Brand name.
|
||||
image_id: The product key from search_products.
|
||||
"""
|
||||
from app.services.nutrition_db import get_full_nutrition
|
||||
|
||||
row = _guard("get_nutrition", get_full_nutrition, brand, image_id)
|
||||
if not row:
|
||||
raise ToolError(
|
||||
f"No nutrition data for {brand}/{image_id}. Not every catalog "
|
||||
"product has been enriched yet."
|
||||
)
|
||||
return _slim(row)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
|
||||
def find_healthier_alternatives(
|
||||
brand: str,
|
||||
image_id: str,
|
||||
limit: Optional[int] = 5,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Find comparable products with a better health score.
|
||||
|
||||
Answers "what should I buy instead of this?" - matches are drawn from the
|
||||
same category, so the suggestion is a real substitute rather than merely
|
||||
something healthier.
|
||||
|
||||
Args:
|
||||
brand: Brand name of the product to improve on.
|
||||
image_id: Product key of the product to improve on.
|
||||
limit: How many alternatives to return (1-20).
|
||||
"""
|
||||
from app.services.nutrition_alternatives_service import find_alternatives
|
||||
|
||||
return _guard(
|
||||
"find_healthier_alternatives",
|
||||
find_alternatives,
|
||||
brand,
|
||||
image_id,
|
||||
top_k=_clip(limit, 5, 20),
|
||||
)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
|
||||
def filter_products_by_nutrition(
|
||||
sort_by: str = "health_score",
|
||||
order: str = "desc",
|
||||
category: Optional[str] = None,
|
||||
diet_tag: Optional[str] = None,
|
||||
exclude_allergen: Optional[str] = None,
|
||||
limit: Optional[int] = 20,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Rank and filter products by a nutritional measure.
|
||||
|
||||
The tool for questions shaped like "highest protein snacks", "lowest sugar
|
||||
dairy", or "gluten-free products without nuts".
|
||||
|
||||
Args:
|
||||
sort_by: Field to rank by - health_score, protein_g, total_sugar_g,
|
||||
dietary_fiber_g, sodium_mg, calories_kcal, total_fat_g.
|
||||
order: "desc" for highest first, "asc" for lowest first.
|
||||
category: Restrict to one category.
|
||||
diet_tag: Keep only products carrying this tag, e.g. "High Protein".
|
||||
exclude_allergen: Drop products declaring this allergen, e.g. "Dairy".
|
||||
limit: How many products to return (1-100).
|
||||
"""
|
||||
from app.services.nutrition_db import query_products
|
||||
|
||||
return _guard(
|
||||
"filter_products_by_nutrition",
|
||||
query_products,
|
||||
sort_by=sort_by,
|
||||
order=order,
|
||||
category=category,
|
||||
diet_tag=diet_tag,
|
||||
exclude_allergen=exclude_allergen,
|
||||
limit=_clip(limit, 20, 100),
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stores
|
||||
# ---------------------------------------------------------------------------
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
|
||||
def list_stores() -> List[Dict[str, Any]]:
|
||||
"""List the retail stores, with city and tier.
|
||||
|
||||
Call this first for anything store-specific - the other store tools need a
|
||||
store_id from here.
|
||||
"""
|
||||
from app.services.store_db import list_stores as _list
|
||||
|
||||
return _guard("list_stores", _list)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
|
||||
def get_store_inventory(
|
||||
store_id: str,
|
||||
category: Optional[str] = None,
|
||||
in_stock_only: bool = False,
|
||||
limit: Optional[int] = 25,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""List what one store carries, with stock levels and its own pricing.
|
||||
|
||||
Prices vary per store, so this is the tool for "what does it cost at X",
|
||||
not the catalog price_range.
|
||||
|
||||
Args:
|
||||
store_id: Store identifier from list_stores.
|
||||
category: Restrict to one category.
|
||||
in_stock_only: Drop products currently out of stock.
|
||||
limit: How many products to return (1-100).
|
||||
"""
|
||||
from app.services.store_db import get_store_products
|
||||
|
||||
return _guard(
|
||||
"get_store_inventory",
|
||||
get_store_products,
|
||||
store_id,
|
||||
category=category,
|
||||
in_stock_only=in_stock_only,
|
||||
limit=_clip(limit, 25, 100),
|
||||
)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
|
||||
def get_store_discounts(store_id: str, limit: Optional[int] = 25) -> List[Dict[str, Any]]:
|
||||
"""List the discounts currently allocated at one store.
|
||||
|
||||
Args:
|
||||
store_id: Store identifier from list_stores.
|
||||
limit: How many discounts to return (1-100).
|
||||
"""
|
||||
from app.services.store_db import get_latest_discounts
|
||||
|
||||
return _guard(
|
||||
"get_store_discounts", get_latest_discounts, store_id, limit=_clip(limit, 25, 100)
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Analytics
|
||||
# ---------------------------------------------------------------------------
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
|
||||
def get_trending(
|
||||
window: str = "weekly",
|
||||
scope: str = "overall",
|
||||
scope_value: Optional[str] = None,
|
||||
limit: Optional[int] = 10,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""List what is selling fastest right now.
|
||||
|
||||
Args:
|
||||
window: Period to measure over - "daily", "weekly" or "monthly".
|
||||
scope: "overall", or "store"/"category" to narrow it.
|
||||
scope_value: The store_id or category name, when scope is not "overall".
|
||||
limit: How many products to return (1-50).
|
||||
"""
|
||||
from app.services.store_db import get_trending as _trending
|
||||
|
||||
return _guard(
|
||||
"get_trending",
|
||||
_trending,
|
||||
window_label=window,
|
||||
scope=scope,
|
||||
scope_value=scope_value,
|
||||
top_k=_clip(limit, 10, 50),
|
||||
)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
|
||||
def get_top_products(
|
||||
by: str = "revenue",
|
||||
limit: Optional[int] = 10,
|
||||
ascending: bool = False,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Rank products across every store by a sales measure.
|
||||
|
||||
Args:
|
||||
by: What to rank on - "revenue", "units" or "orders".
|
||||
limit: How many products to return (1-50).
|
||||
ascending: True ranks worst-performing first.
|
||||
"""
|
||||
from app.services.analytics_service import top_products
|
||||
|
||||
return _guard(
|
||||
"get_top_products", top_products, by=by, limit=_clip(limit, 10, 50), ascending=ascending
|
||||
)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
|
||||
def get_store_analytics(store_id: str) -> Dict[str, Any]:
|
||||
"""Get one store's sales dashboard - revenue, orders, top sellers, stock health.
|
||||
|
||||
Args:
|
||||
store_id: Store identifier from list_stores.
|
||||
"""
|
||||
from app.services.analytics_service import store_dashboard
|
||||
|
||||
return _guard("get_store_analytics", store_dashboard, store_id)
|
||||
|
||||
|
||||
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
||||
def get_recommendations(
|
||||
brand: str,
|
||||
image_id: str,
|
||||
limit: Optional[int] = 5,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""List products frequently bought together with this one.
|
||||
|
||||
Co-purchase based, so this answers "what else goes in the basket", which is
|
||||
a different question from find_healthier_alternatives' "what instead".
|
||||
|
||||
Args:
|
||||
brand: Brand name.
|
||||
image_id: Product key from search_products.
|
||||
limit: How many recommendations to return (1-20).
|
||||
"""
|
||||
from app.services.recommendation_service import recommend_for_product
|
||||
|
||||
return _guard(
|
||||
"get_recommendations",
|
||||
recommend_for_product,
|
||||
brand,
|
||||
image_id,
|
||||
top_k=_clip(limit, 5, 20),
|
||||
)
|
||||
|
||||
|
||||
def build_http_app(path: str = MCP_PATH):
|
||||
"""The ASGI app to mount on FastAPI. See app/main.py for the lifespan wiring."""
|
||||
return mcp.http_app(path="/", transport="http", stateless_http=True)
|
||||
@@ -42,6 +42,8 @@ actually resolves to real image bytes above a minimum size - this is what
|
||||
stops garbage/placeholder/expired URLs from reaching the S3 upload step
|
||||
and failing there silently.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Optional, List
|
||||
import requests
|
||||
from urllib.parse import urlparse
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List, Dict, Any, Optional
|
||||
import json
|
||||
import re
|
||||
|
||||
@@ -9,7 +9,10 @@ import re
|
||||
import time
|
||||
import psycopg
|
||||
|
||||
from app.infrastructure.settings import DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD
|
||||
from app.infrastructure.settings import (
|
||||
DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD,
|
||||
DB_CONNECT_TIMEOUT_SECONDS,
|
||||
)
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
|
||||
|
||||
@@ -93,7 +96,15 @@ def _connect() -> Optional[psycopg.Connection]:
|
||||
dbname=DB_NAME,
|
||||
user=DB_USER,
|
||||
password=DB_PASSWORD,
|
||||
autocommit=True
|
||||
autocommit=True,
|
||||
# Without this, a host that DROPS packets rather than refusing them
|
||||
# - a firewall, a wrong DB_HOST - blocks here until the OS gives up,
|
||||
# which is around 130 seconds on Linux. Every caller of _connect()
|
||||
# inherits that: /api/health stops answering, the container's
|
||||
# healthcheck times out, and the platform pulls the service out of
|
||||
# its load balancer. "Database unreachable" then presents as a Bad
|
||||
# Gateway on every route, including ones that never touch the DB.
|
||||
connect_timeout=DB_CONNECT_TIMEOUT_SECONDS,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Vector DB connection failed: {e}")
|
||||
|
||||
@@ -9,7 +9,8 @@
|
||||
# container layer, and so `ollama pull` model files persist outside Docker.
|
||||
#
|
||||
# Usage:
|
||||
# docker compose up -d
|
||||
# docker compose up -d # Postgres only (the default)
|
||||
# docker compose --profile full up -d # Postgres + the API, with volumes
|
||||
# # then in backend/.env: DB_HOST=localhost, DB_PORT=5432, DB_NAME=pgvector,
|
||||
# # DB_USER=postgres, DB_PASSWORD=<the value you set below>
|
||||
services:
|
||||
@@ -34,5 +35,53 @@ services:
|
||||
timeout: 5s
|
||||
retries: 10
|
||||
|
||||
# The API. Started only with `--profile full`, so the default
|
||||
# `docker compose up -d` still brings up Postgres alone as it always did.
|
||||
#
|
||||
# docker compose --profile full up -d --build
|
||||
#
|
||||
# Present mainly as the reference for the two volume mounts below - the paths
|
||||
# are the same ones to configure in Dokploy.
|
||||
backend:
|
||||
profiles: ["full"]
|
||||
build: .
|
||||
container_name: catalog_rag_backend
|
||||
restart: unless-stopped
|
||||
depends_on:
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
env_file: .env
|
||||
environment:
|
||||
# Inside a container `localhost` is the container itself, so the DB is
|
||||
# reached by service name over the compose network.
|
||||
DB_HOST: postgres
|
||||
DB_PORT: "5432"
|
||||
DB_PASSWORD: ${POSTGRES_PASSWORD:-changeme}
|
||||
# Ollama runs natively on the host, not in compose (see the note above).
|
||||
OLLAMA_BASE_URL: http://host.docker.internal:11434
|
||||
extra_hosts:
|
||||
- "host.docker.internal:host-gateway"
|
||||
# The container listens on 3000 and 8000 at once (see serve.py), so either
|
||||
# side of this mapping can change without touching the image. 8000 is
|
||||
# published because vite.config.js proxies /api to 127.0.0.1:8000.
|
||||
ports:
|
||||
- "8000:8000"
|
||||
volumes:
|
||||
# WITHOUT THESE TWO MOUNTS, a redeploy silently discards:
|
||||
# - every product added through the UI (POST /api/user/products/add and
|
||||
# /upload-file append to data/seed_catalogs/*.json), and
|
||||
# - every retrained model (the training endpoints write *.joblib).
|
||||
# Both directories live inside the image, so rebuilding it resets them to
|
||||
# whatever was committed to the repo.
|
||||
#
|
||||
# The image also carries a pristine copy at /app/.bundled, and the app
|
||||
# tops up anything missing on startup without overwriting what is already
|
||||
# there - so these work whether they are named volumes or bind mounts.
|
||||
# See app/infrastructure/persistence.py.
|
||||
- catalog_rag_data:/app/data
|
||||
- catalog_rag_artifacts:/app/app/intelligence/artifacts
|
||||
|
||||
volumes:
|
||||
catalog_rag_pgdata:
|
||||
catalog_rag_data:
|
||||
catalog_rag_artifacts:
|
||||
|
||||
@@ -14,6 +14,13 @@ python-dotenv>=1.0.1
|
||||
# deliberately no bcrypt/argon2/passlib dependency here.
|
||||
PyJWT>=2.9.0
|
||||
|
||||
# --- MCP (Model Context Protocol) ---
|
||||
# Serves the catalog as tools for AI clients at /mcp (see app/mcp_server.py),
|
||||
# mounted onto the FastAPI app so it needs no separate process or container.
|
||||
# Requires Python >=3.10; the Dockerfile's python:3.11-slim satisfies that, and
|
||||
# app/main.py degrades to serving the REST API alone if the import fails.
|
||||
fastmcp>=3.4.7
|
||||
|
||||
# --- Database / pgvector ---
|
||||
psycopg[binary]>=3.2.3
|
||||
pgvector>=0.2.5
|
||||
|
||||
82
scripts/check_deploy.sh
Executable file
82
scripts/check_deploy.sh
Executable file
@@ -0,0 +1,82 @@
|
||||
#!/usr/bin/env bash
|
||||
# Post-deploy smoke test.
|
||||
#
|
||||
# ./scripts/check_deploy.sh # https://mcp.nearle.ai.in
|
||||
# ./scripts/check_deploy.sh http://localhost:3000 # a local container
|
||||
#
|
||||
# Separates the three failures that all look alike from a browser:
|
||||
# 502 everywhere - the container is not running (it exited at startup)
|
||||
# 200 + database:false - the API is fine, Postgres is not
|
||||
# 200 + database:true - working
|
||||
set -uo pipefail
|
||||
|
||||
BASE="${1:-https://mcp.nearle.ai.in}"
|
||||
pass=0; fail=0
|
||||
|
||||
hit() { curl -sS -m 20 -o /tmp/_body -w '%{http_code}' "$BASE$1" 2>/dev/null || echo 000; }
|
||||
|
||||
check() { # path expected label
|
||||
local code; code=$(hit "$1")
|
||||
if [ "$code" = "$2" ]; then printf ' \033[32mPASS\033[0m %-16s %s\n' "$1" "$3"; pass=$((pass+1))
|
||||
else printf ' \033[31mFAIL\033[0m %-16s expected %s, got %s\n' "$1" "$2" "$code"; fail=$((fail+1)); fi
|
||||
}
|
||||
|
||||
echo "Checking $BASE"
|
||||
echo
|
||||
echo "Is the container running at all?"
|
||||
code=$(hit /)
|
||||
if [ "$code" = "502" ] || [ "$code" = "000" ]; then
|
||||
echo " FAIL / -> $code"
|
||||
echo
|
||||
echo " The container is not serving. It almost certainly exited at startup."
|
||||
echo " Read the Dokploy logs; settings.py names the missing value explicitly."
|
||||
echo " Confirm the image actually contains /app/.env - the build log must show"
|
||||
echo " a 'COPY .env .' step, and .dockerignore must not list .env."
|
||||
exit 1
|
||||
fi
|
||||
printf ' \033[32mPASS\033[0m %-16s container is up\n' "/"
|
||||
pass=$((pass+1))
|
||||
|
||||
echo
|
||||
echo "Routes that must not depend on the database:"
|
||||
check /docs 200 "interactive API docs"
|
||||
check /openapi.json 200 "OpenAPI schema"
|
||||
check /api/health 200 "health endpoint answers"
|
||||
|
||||
echo
|
||||
echo "Dependencies (reported in the body; 'false' does not mean the API is broken):"
|
||||
health=$(curl -sS -m 20 "$BASE/api/health" 2>/dev/null)
|
||||
db=$(printf '%s' "$health" | sed -n 's/.*"database":\([a-z]*\).*/\1/p')
|
||||
ol=$(printf '%s' "$health" | sed -n 's/.*"ollama":\([a-z]*\).*/\1/p')
|
||||
if [ "$db" = "true" ]; then printf ' \033[32mPASS\033[0m database connected\n'; pass=$((pass+1))
|
||||
else
|
||||
printf ' \033[33mWARN\033[0m database NOT connected\n'
|
||||
echo " The API works; catalog pages will be empty. Check DB_HOST/DB_USER/"
|
||||
echo " DB_PASSWORD/DB_NAME. A wrong password logs 'password authentication"
|
||||
echo " failed' rather than a timeout."
|
||||
fi
|
||||
[ "$ol" = "true" ] \
|
||||
&& printf ' \033[32mPASS\033[0m ollama connected\n' \
|
||||
|| printf ' \033[33mWARN\033[0m ollama not connected (/api/chat unavailable; expected if USE_OLLAMA=false)\n'
|
||||
|
||||
echo
|
||||
echo "Auth:"
|
||||
code=$(curl -sS -m 20 -o /dev/null -w '%{http_code}' -X POST "$BASE/api/auth/login" \
|
||||
-H 'Content-Type: application/json' -d '{"username":"admin","password":"definitely-not-the-password"}' 2>/dev/null)
|
||||
if [ "$code" = "401" ]; then printf ' \033[32mPASS\033[0m wrong password rejected (401)\n'; pass=$((pass+1))
|
||||
elif [ "$code" = "200" ]; then
|
||||
printf ' \033[31mFAIL\033[0m a WRONG PASSWORD WAS ACCEPTED - AUTH_ALLOW_ANY_LOGIN is true.\n'
|
||||
echo " Anyone who finds this host can sign in as admin. Set it to false."
|
||||
fail=$((fail+1))
|
||||
else printf ' \033[31mFAIL\033[0m login endpoint returned %s\n' "$code"; fail=$((fail+1)); fi
|
||||
|
||||
echo
|
||||
echo "MCP:"
|
||||
code=$(curl -sS -m 20 -o /dev/null -w '%{http_code}' "$BASE/api/mcp/info" 2>/dev/null)
|
||||
[ "$code" = "401" ] \
|
||||
&& { printf ' \033[32mPASS\033[0m /api/mcp/info guarded (401 without a token)\n'; pass=$((pass+1)); } \
|
||||
|| printf ' \033[33mWARN\033[0m /api/mcp/info returned %s (expected 401)\n' "$code"
|
||||
|
||||
echo
|
||||
echo "-------- $pass passed, $fail failed --------"
|
||||
[ "$fail" -eq 0 ]
|
||||
170
serve.py
Normal file
170
serve.py
Normal file
@@ -0,0 +1,170 @@
|
||||
"""
|
||||
Container entry point: serve the API on every port the platform might route to.
|
||||
|
||||
The frontend image answers on both 80 and 3000 (``listen 80; listen 3000;`` in
|
||||
nginx.conf) so it works whatever port the deployment is configured to hit. This
|
||||
does the same for the API, which otherwise has to guess: Dokploy routes a domain
|
||||
to one container port, this project's own README, vite.config.js proxy and
|
||||
docker-compose mapping all say 8000, and picking wrong produces a 502 with a
|
||||
perfectly healthy container behind it.
|
||||
|
||||
uvicorn's CLI binds a single ``--port``, but ``Server.run()`` accepts a list of
|
||||
already-bound sockets, so one process can listen on several. That is what this
|
||||
does - no extra worker, no second process to supervise.
|
||||
|
||||
Usage::
|
||||
|
||||
python serve.py # binds PORTS (default "3000,8000")
|
||||
PORT=8080 python serve.py # binds only 8080
|
||||
python serve.py --healthcheck # probe mode, used by HEALTHCHECK
|
||||
|
||||
Run it directly rather than through ``uvicorn app.main:app`` when you need the
|
||||
multi-port behaviour; the plain uvicorn command still works for local
|
||||
development where one port is all anyone wants.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
logger = logging.getLogger("serve")
|
||||
|
||||
# Both of the ports this project actually uses anywhere: 3000 because that is
|
||||
# what the platform routes a domain to by default, 8000 because the README,
|
||||
# the vite dev proxy and docker-compose all target it.
|
||||
DEFAULT_PORTS = "3000,8000"
|
||||
|
||||
HOST = os.getenv("HOST", "0.0.0.0")
|
||||
|
||||
|
||||
def configured_ports() -> list[int]:
|
||||
"""
|
||||
The ports to bind, in order.
|
||||
|
||||
``PORT`` wins over ``PORTS`` and is treated as an explicit single choice:
|
||||
setting it means "serve here", not "serve here as well". ``PORTS`` takes a
|
||||
comma-separated list for the both-at-once case.
|
||||
"""
|
||||
raw = os.getenv("PORT") or os.getenv("PORTS") or DEFAULT_PORTS
|
||||
|
||||
ports: list[int] = []
|
||||
for chunk in raw.split(","):
|
||||
chunk = chunk.strip()
|
||||
if not chunk:
|
||||
continue
|
||||
try:
|
||||
port = int(chunk)
|
||||
except ValueError:
|
||||
logger.warning("Ignoring non-numeric port %r in %r", chunk, raw)
|
||||
continue
|
||||
if not 1 <= port <= 65535:
|
||||
logger.warning("Ignoring out-of-range port %d", port)
|
||||
continue
|
||||
if port not in ports: # binding the same port twice would fail
|
||||
ports.append(port)
|
||||
|
||||
if not ports:
|
||||
logger.error("No usable port in %r - falling back to %s", raw, DEFAULT_PORTS)
|
||||
return [int(p) for p in DEFAULT_PORTS.split(",")]
|
||||
return ports
|
||||
|
||||
|
||||
def _bind(port: int) -> socket.socket | None:
|
||||
"""Bind one listening socket, or return None with the reason logged."""
|
||||
sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
||||
try:
|
||||
sock.bind((HOST, port))
|
||||
except OSError as exc:
|
||||
# Not fatal on its own. If the platform only routes to one of these,
|
||||
# losing the other (already in use, not permitted) should not take the
|
||||
# service down - _run() fails only when nothing at all is listening.
|
||||
logger.warning("Could not bind %s:%d - %s", HOST, port, exc)
|
||||
sock.close()
|
||||
return None
|
||||
sock.listen(2048)
|
||||
sock.set_inheritable(True)
|
||||
return sock
|
||||
|
||||
|
||||
def _run() -> int:
|
||||
import uvicorn
|
||||
|
||||
ports = configured_ports()
|
||||
sockets = [s for s in (_bind(p) for p in ports) if s is not None]
|
||||
|
||||
if not sockets:
|
||||
logger.error(
|
||||
"Could not bind any of %s. The API is not listening; exiting so the "
|
||||
"platform restarts or reports the container as failed.",
|
||||
", ".join(str(p) for p in ports),
|
||||
)
|
||||
return 1
|
||||
|
||||
bound = [s.getsockname()[1] for s in sockets]
|
||||
logger.info("Serving on %s port(s): %s", HOST, ", ".join(str(p) for p in bound))
|
||||
|
||||
config = uvicorn.Config(
|
||||
"app.main:app",
|
||||
# proxy_headers/forwarded_allow_ips: this container is never reached
|
||||
# directly - nginx, and Dokploy's Traefik, sit in front of it. Without
|
||||
# them uvicorn ignores X-Forwarded-Proto and reports every request as
|
||||
# plain http, so any redirect or generated absolute URL would downgrade
|
||||
# an https request.
|
||||
proxy_headers=True,
|
||||
forwarded_allow_ips="*",
|
||||
)
|
||||
uvicorn.Server(config).run(sockets=sockets)
|
||||
return 0
|
||||
|
||||
|
||||
def _healthcheck() -> int:
|
||||
"""
|
||||
Probe the API, passing if ANY configured port answers.
|
||||
|
||||
Shares configured_ports() with the server so the check cannot drift from
|
||||
what is actually bound - the reason this lives here rather than being a
|
||||
python -c one-liner in the Dockerfile.
|
||||
|
||||
Probes "/" rather than /api/health, and the distinction matters. This is a
|
||||
LIVENESS check: the only question it may ask is "is this process still
|
||||
serving HTTP", because the platform's answer to "no" is to stop routing
|
||||
traffic to the container.
|
||||
|
||||
/api/health is a READINESS report - it dials Postgres and Ollama to say
|
||||
whether they are reachable. Using it here couples the container's existence
|
||||
to its dependencies: an unreachable database made /api/health block for the
|
||||
OS TCP timeout, the check timed out, the container was marked unhealthy, and
|
||||
a service that was running perfectly well returned Bad Gateway on every
|
||||
route - including the ones that never touch the database. "/" is served from
|
||||
memory and does no I/O at all, so it can only fail if the app really is gone.
|
||||
"""
|
||||
ports = configured_ports()
|
||||
for port in ports:
|
||||
try:
|
||||
with urllib.request.urlopen(f"http://127.0.0.1:{port}/", timeout=8) as resp:
|
||||
if 200 <= resp.status < 500:
|
||||
return 0
|
||||
except urllib.error.HTTPError as exc:
|
||||
# An HTTP status - even 404 - means something is listening and
|
||||
# routing. That is exactly what liveness asks.
|
||||
if exc.code < 500:
|
||||
return 0
|
||||
except (urllib.error.URLError, OSError, ValueError):
|
||||
continue
|
||||
print(
|
||||
f"health: no response from any of {', '.join(str(p) for p in ports)}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s")
|
||||
if "--healthcheck" in sys.argv:
|
||||
sys.exit(_healthcheck())
|
||||
sys.exit(_run())
|
||||
Reference in New Issue
Block a user