diff --git a/.dockerignore b/.dockerignore index 6f2e1ef..829858b 100644 --- a/.dockerignore +++ b/.dockerignore @@ -3,6 +3,15 @@ __pycache__ *.pyc .pytest_cache *.log -.env .git tests + +# .env.production carries this deployment's configuration and the Dockerfile +# copies it to /app/.env inside the image, so it must NOT be ignored here. +# Dokploy's own .env (written from the Environment tab, empty when that tab is +# blank) is ignored instead - it is what silently overwrote the committed one. +# +# The generated sign-in passwords must never be in the image. +.env +.env.local +SIGNIN_PASSWORDS.txt diff --git a/.env.example b/.env.example index 4046cf4..3f239b1 100644 --- a/.env.example +++ b/.env.example @@ -16,6 +16,64 @@ # localhost URL points the backend at its own empty ports. Containers reach # each other by service name over the compose network instead. +# --- Ports (container only) ----------------------------------------------- +# The image serves 3000 and 8000 at once - 3000 because that is what Dokploy +# routes a domain to, 8000 because the README, the vite dev proxy and +# docker-compose all use it. Serving both means the container works whichever +# one the platform points at. +# +# PORTS the comma-separated pair to bind. PORT pins a single port instead and +# takes precedence, e.g. PORT=8080 serves only 8080. +# +# Neither affects running uvicorn directly for local development. +# PORTS=3000,8000 +# PORT=8080 + +# --- Persistence (IMPORTANT in Docker) ------------------------------------ +# The three directories the app WRITES to at runtime. The defaults point inside +# the repo/image and are right for local development; in a container each one +# needs a volume mounted on it, or a redeploy throws away everything written +# since the last build: +# +# SEED_CATALOG_DIR products added via POST /api/user/products/add and +# /upload-file are appended to the JSON files here +# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints +# DATA_DIR catalogs saved by the ingestion pipeline +# +# Mount these two paths in Dokploy (SEED_CATALOG_DIR sits inside DATA_DIR, so +# one mount covers both): +# +# /app/data +# /app/app/intelligence/artifacts +# +# Either a named volume or a bind mount works. The image keeps read-only copies +# of the bundled seed catalogs and pre-trained models at /app/.bundled, and the +# app restores whatever a freshly-mounted directory is missing on startup +# without overwriting anything already there - so a bind mount, which starts +# empty and would otherwise hide them, is safe. +# +# Leave all three unset unless the writable data belongs somewhere else. +# DATA_DIR=/app/data +# SEED_CATALOG_DIR=/app/data/seed_catalogs +# MODEL_ARTIFACTS_DIR=/app/app/intelligence/artifacts + +# --- CORS (REQUIRED when the API is on its own domain) -------------------- +# Comma-separated list of the exact browser origins allowed to call this API. +# Scheme and host both matter; no trailing slash, and no wildcard - the +# frontend sends an Authorization header, and browsers refuse a credentialed +# cross-origin request whose Allow-Origin is "*" (app/main.py logs and turns +# credentials off if it sees one, so a wildcard silently breaks every call). +# +# Production - the React app is served from catalogue.nearle.ai.in and calls +# the API at mcp.nearle.ai.in, so that frontend origin must be listed: +# +# API_CORS_ORIGINS=https://catalogue.nearle.ai.in +# +# Add http://localhost:5173 alongside it if you point a local Vite dev server +# at the deployed API. Server-to-server callers (MCP clients, scripts) are not +# affected by any of this - CORS is a browser rule; they use X-API-Key. +API_CORS_ORIGINS=http://localhost:5173,http://127.0.0.1:5173 + # --- Authentication (REQUIRED) ------------------------------------------- # The backend will not start without these while AUTH_ENABLED=true. Generate # all four lines, plus sign-in passwords, with: diff --git a/.env.production b/.env.production new file mode 100644 index 0000000..9dce744 --- /dev/null +++ b/.env.production @@ -0,0 +1,113 @@ +# Deployment configuration for mcp.nearle.ai.in. +# +# Committed at the repo owner's instruction so the deploy does not depend on +# re-entering config in the Dokploy UI. Everything needed to boot is here; no +# environment variables are required in Dokploy any more. +# +# A real environment variable still overrides anything set here - settings.py +# calls load_dotenv() without override=True, so the process environment wins. +# That is the escape hatch for changing a value without a commit. +# +# WHAT IS IN THIS FILE: live database, S3 and Google credentials, and the key +# that signs every access token. Anyone with read access to this repository has +# all of it, and git history keeps it after any rotation. + +# --- Ports ----------------------------------------------------------------- +# Dokploy routes the domain to 3000; 8000 is kept for the vite dev proxy and +# docker-compose. serve.py binds both. +PORTS=3000,8000 + +# --- CORS ------------------------------------------------------------------ +# The FRONTEND's origin, not this API's. Wrong value = the browser blocks every +# response while the server logs healthy 200s. +API_CORS_ORIGINS=https://catalogue.nearle.ai.in + +# --- Authentication -------------------------------------------------------- +AUTH_ENABLED=true + +# Freshly generated for this deployment - deliberately NOT the values from the +# development .env. Those hashes are for the passwords DevAdmin!2026 and +# DevUser!2026, which are sitting in plaintext in test_login_fix.py in this very +# repository: shipping them would publish working admin credentials. +# Sign-in passwords for the hashes below are in SIGNIN_PASSWORDS.txt (ignored). +AUTH_SECRET_KEY=4Kmyr4Cjf_kdUIq_4EGxo5vFHfCT5_uKVR3eouszB8Le6F0n45m7eDY94_KJoqSz +AUTH_ADMIN_USERNAME=admin +AUTH_ADMIN_PASSWORD_HASH=pbkdf2_sha256$600000$Xa07unPO4LeTU4bz04eh7Q==$KmZ2ZBrclJ0z0sDiCoKOrfTO2UK8e7hjsZZ6HB2PY9o= +AUTH_USER_USERNAME=user +AUTH_USER_PASSWORD_HASH=pbkdf2_sha256$600000$65VMpqwUSyFzCqnhlwBqgQ==$sMVnar+Hnp5ZmXcObFNI3jJMxqnVrW8naLZNFFh0KCw= + +AUTH_TOKEN_TTL_MINUTES=720 +AUTH_MAX_LOGIN_ATTEMPTS=10 +AUTH_LOCKOUT_SECONDS=300 + +# MUST stay false here. The development .env has this true, where it is a +# convenience: it skips the password check entirely, so any username signs in +# and `admin` gets the admin pages. On a host published to the internet it means +# anyone who finds mcp.nearle.ai.in signs in as admin by typing anything at all. +AUTH_ALLOW_ANY_LOGIN=false + +# Machine consumers. Empty: MCP clients authenticate with a login token instead. +API_KEYS= + +# --- Postgres / pgvector --------------------------------------------------- +# DB_NAME is not set in the development .env, so it falls back to settings.py's +# default. Stated explicitly here so the deployment does not depend on that +# default staying the same. +USE_PGVECTOR=true +DB_HOST=31.97.228.132 +DB_PORT=6054 +DB_NAME=pgvector +DB_USER=admin +DB_PASSWORD=Package@321# + +# --- Embeddings ------------------------------------------------------------ +USE_EMBEDDINGS=true +EMBEDDINGS_MODEL=sentence-transformers/all-MiniLM-L6-v2 +EMBEDDINGS_DIM=384 + +# --- Ollama (local LLM, powers /api/chat) ---------------------------------- +# Off, because the development value (http://localhost:11434) cannot work from +# inside a container: there, localhost is the container itself, not the VPS +# host. Left on with nothing listening, /api/chat fails AND every healthcheck +# takes ~3s longer, because the health handler probes Ollama with a 3s timeout. +# +# To enable: set USE_OLLAMA=true and point OLLAMA_BASE_URL at something the +# container can actually reach - http://host.docker.internal:11434 with a +# host-gateway mapping, the VPS's LAN IP, or an ollama service name. +USE_OLLAMA=false +OLLAMA_BASE_URL=http://host.docker.internal:11434 +OLLAMA_MODEL_NAME=qwen2.5:1.5b +OLLAMA_TIMEOUT_SECONDS=120 + +# --- DigitalOcean Spaces (product image storage) --------------------------- +USE_S3=true +S3_ACCESS_KEY=DO801G8Q8JAZKF49U3WJ +S3_SECRET_KEY=lBQExYfkVqH+ybmGVmQH5MkThBbrIohA/VQLgcPUvug +S3_ENDPOINT=https://nearle.sgp1.digitaloceanspaces.com +S3_BUCKET=nearle +S3_REGION=sgp1 + +# --- Google Custom Search (optional image source) -------------------------- +USE_GOOGLE_CSE=true +GOOGLE_API_KEY=AIzaSyBY4pIO_Fp5FCMqeVxDNcfalzdWNHJWVn0 +GOOGLE_CSE_ID=9745cbd96dd164562 + +# --- Open-source image sources (no key needed) ----------------------------- +USE_DDG_IMAGES=true +USE_OPEN_FACTS=true +USE_WIKIMEDIA=true +# The Playwright browser binary is NOT installed in the image (see Dockerfile), +# so this tier is skipped at runtime regardless. false stops it being attempted. +USE_PLAYWRIGHT_FALLBACK=false + +MIN_IMAGE_BYTES=3000 + +# --- Product validation ---------------------------------------------------- +ENABLE_PRODUCT_VALIDATION=true +VALIDATION_REJECT_THRESHOLD=0.35 +VALIDATION_REVIEW_THRESHOLD=0.70 + +# --- RAG ------------------------------------------------------------------- +RAG_DEFAULT_TOP_K=5 +RAG_MAX_TOP_K=15 +RAG_MAX_CONTEXT_CHARS=4000 diff --git a/.gitignore b/.gitignore index a87448d..05969e2 100644 --- a/.gitignore +++ b/.gitignore @@ -1,6 +1,26 @@ -# Secrets - never commit +# .env.production is committed deliberately, at the repo owner's instruction, so +# the deployment does not depend on re-entering config in the Dokploy UI. +# +# It is NOT named .env, and that matters: Dokploy writes its own .env into the +# build context from the service's Environment tab after cloning, so a committed +# .env is silently replaced (with an empty file when that tab is blank). +# The Dockerfile copies .env.production to /app/.env inside the image. +# +# The two database secrets are NOT in it - they are set as Dokploy environment +# variables, which override the file (settings.py calls load_dotenv() without +# override=True, so the process environment wins). +# +# The auth secrets ARE in it. AUTH_SECRET_KEY signs every access token, so +# anyone with read access to this repository can mint an admin token, and git +# history keeps it after any rotation. Regenerate with +# `python scripts/make_auth_secrets.py` if that stops being acceptable. +!.env.production .env +# Local overrides and the generated sign-in passwords stay out of git. +.env.local +SIGNIN_PASSWORDS.txt + # Python __pycache__/ *.pyc diff --git a/Dockerfile b/Dockerfile index eabbd5d..89ed8c0 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,25 +1,116 @@ -FROM python:3.11-slim +# Multi-stage, mirroring catalogue_frontend/Dockerfile: a build stage that +# resolves dependencies, then a clean runtime stage that copies in only the +# result. There it is `npm ci` -> dist/; here it is pip -> a virtualenv. +# ---- Build stage ---- +FROM python:3.11-slim AS build WORKDIR /app +# Dependencies land in a self-contained venv so the runtime stage can take that +# one directory and leave pip, its HTTP cache and the downloaded wheels behind. +# +# requirements.txt is copied on its own, ahead of the source, for the same +# reason the frontend stage copies package.json before the rest of the app: +# this layer is cached on the file's checksum, so editing a router does not +# reinstall torch. +# # psycopg[binary] avoids needing libpq-dev; sentence-transformers/scikit-learn/ -# scipy all ship prebuilt wheels for this image, so no extra build toolchain -# is needed. Playwright's Python package installs, but its browser binary is -# NOT installed here - it's only a last-resort image-search fallback -# (see requirements.txt); run `playwright install chromium` in the container -# if you need that specific fallback tier. +# scipy all ship prebuilt wheels for this image, so no compiler is needed and +# this stage installs no build toolchain. Playwright's Python package installs, +# but its browser binary is NOT installed here - it's only a last-resort +# image-search fallback (see requirements.txt); run `playwright install +# chromium` in the container if you need that specific fallback tier. COPY requirements.txt . -RUN pip install --no-cache-dir -r requirements.txt +RUN python -m venv /opt/venv \ + && /opt/venv/bin/pip install --no-cache-dir -r requirements.txt + + +# ---- Runtime stage ---- +FROM python:3.11-slim AS runtime +WORKDIR /app + +# PATH: putting the venv first is what makes a bare `python`/`uvicorn` resolve +# to it - there is no "activate" step in a container. +# PYTHONUNBUFFERED: without it Dokploy's log view stays empty until a buffer +# happens to fill, so startup errors surface minutes after the container died. +# PYTHONDONTWRITEBYTECODE: no .pyc to write into a read-only-ish image layer. +ENV PATH="/opt/venv/bin:$PATH" +ENV PYTHONUNBUFFERED=1 +ENV PYTHONDONTWRITEBYTECODE=1 + +COPY --from=build /opt/venv /opt/venv COPY app ./app COPY cli ./cli COPY scripts ./scripts COPY data ./data +COPY serve.py . -EXPOSE 8000 -# --proxy-headers/--forwarded-allow-ips: this container is never reached -# directly - nginx (and, in prod, Caddy) sit in front of it. Without these, -# uvicorn ignores X-Forwarded-Proto and reports every request as plain http, -# so any redirect or generated absolute URL would downgrade an https request. -CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000", \ - "--proxy-headers", "--forwarded-allow-ips", "*"] +# The deployment's configuration, landing at /app/.env because settings.py +# resolves it from the backend root - app/infrastructure/settings.py takes +# parents[2], which is /app here - so it must sit next to app/, not inside it. +# +# The source file is named .env.production, NOT .env, and that detail is the +# whole point. Dokploy writes its own .env into the build context from the +# service's Environment tab AFTER cloning the repository. With that tab empty it +# writes an empty file, overwriting the committed one - so `COPY .env .` copied +# a zero-byte file, the container started with no configuration at all, and the +# platform reported only a Bad Gateway. The checkout showed it plainly: every +# file timestamped 08:33, and .env alone at 08:34, 0 bytes. +# +# Dokploy does not manage .env.production, so it survives. Anything set in the +# Environment tab still wins at runtime, because settings.py calls load_dotenv() +# without override=True and the process environment takes precedence. +COPY .env.production .env + +# Pristine copies of everything the app also WRITES to, kept at a path that is +# never mounted over. +# +# /app/data/seed_catalogs and /app/app/intelligence/artifacts both need volumes +# (products added through the UI are appended to the first, retrained models +# are written to the second - otherwise a redeploy throws both away). But +# mounting a volume there hides the copies shipped in this image: a *named* +# volume is seeded from the image on first use, a *bind* mount is not, and +# Dokploy offers both. A bind mount would leave the API with zero seed catalogs +# and zero trained models, with nothing in the logs saying why. +# +# So keep a second copy here. On startup restore_bundled_assets() +# (app/infrastructure/persistence.py) copies in whatever the mounted directory +# is missing, and never overwrites what is already there. +RUN mkdir -p /app/.bundled \ + && cp -a /app/data/seed_catalogs /app/.bundled/seed_catalogs \ + && cp -a /app/app/intelligence/artifacts /app/.bundled/artifacts + +# Declared so `docker run` without an explicit -v still gets an anonymous +# volume rather than writing into the container layer. Dokploy (and the compose +# file) name them properly; this is the floor, not the recommended setup. +VOLUME ["/app/data", "/app/app/intelligence/artifacts"] + +# Answers on BOTH ports, the same way the frontend image does (nginx.conf has +# `listen 80; listen 3000;`). 3000 is what Dokploy routes a domain to; 8000 is +# what this project's README, the vite dev proxy and docker-compose all use. +# Serving both means the container works whichever one the platform is pointed +# at, instead of returning 502 from a perfectly healthy process. +# +# serve.py binds both sockets and hands them to one uvicorn - see the note +# there. To pin a single port, set PORT (PORT=8080 serves only 8080); to change +# the pair, set PORTS. +ENV PORTS=3000,8000 +EXPOSE 3000 8000 + +# Liveness only, and passes if EITHER port answers. +# +# It probes "/", which is served from memory, NOT /api/health, which dials +# Postgres and Ollama. That is the whole point: the platform's response to a +# failed healthcheck is to stop routing traffic, so this may only ask "is the +# process still serving HTTP". Tying it to the database meant an unreachable +# Postgres blocked the handler for the OS TCP timeout, the check timed out, the +# container was marked unhealthy, and a perfectly healthy API returned Bad +# Gateway on every route. Use /api/health to ask whether dependencies are up; +# it reports them in the body and always answers 200. +HEALTHCHECK --interval=30s --timeout=10s --start-period=40s --retries=3 \ + CMD ["python", "serve.py", "--healthcheck"] + +# Exec form: python is PID 1, so Docker's SIGTERM reaches it directly and a +# redeploy shuts down cleanly instead of waiting out the 10s kill timeout. +CMD ["python", "serve.py"] diff --git a/README.md b/README.md index 15a41de..393c018 100644 --- a/README.md +++ b/README.md @@ -69,6 +69,105 @@ permission check; `user` holds the product/store/inventory permissions. `AUTH_ENABLED=false` disables all of it for local work — never in a deployment. See the Authentication section of `../DEPLOYMENT.md` for the full endpoint map. +## MCP server + +The catalog is exposed to AI clients over the Model Context Protocol at `/mcp`, +mounted onto this same FastAPI app (`app/mcp_server.py`) - no separate process +or container, so it deploys with the API and answers on the same host. + +**15 tools, all read-only.** Catalog search and browse, nutrition facts and +health scores, healthier alternatives, per-store inventory and pricing, +discounts, trending and sales analytics. None of the write or compute endpoints +are reachable through MCP: a tool list is chosen from by a model rather than by +a person, and "retrain the models" is not something to leave one tool call away. + +**Auth is the app's own access token** - there is no separate MCP credential. +Clients send `Authorization: Bearer ` from `POST /api/auth/login`, and +the same `Principal` and expiry apply as on the REST side. Note the practical +consequence: tokens expire after `AUTH_TOKEN_TTL_MINUTES` (12h by default), so a +long-running client has to refresh. Raise the TTL if that is a problem. + +```jsonc +{ + "mcpServers": { + "nearle-catalogue": { + "type": "http", + "url": "https://mcp.nearle.ai.in/mcp", + "headers": { "Authorization": "Bearer " } + } + } +} +``` + +The admin UI has an inspector at `/mcp` (React route, admin-only) listing every +tool with its arguments and a console to run one against live data. It is backed +by `GET /api/mcp/info` and `POST /api/mcp/tools/{name}` rather than by the MCP +endpoint itself - the browser would otherwise need a full MCP client to render a +tool list. + +`fastmcp` requires Python 3.10+. The image is 3.11; on an older interpreter the +import fails, `app/main.py` logs a warning, and the REST API starts without the +MCP endpoint rather than failing outright. + +## Ports + +The container answers on **3000 and 8000 at the same time**, the same way the +frontend image answers on 80 and 3000. 3000 is what Dokploy routes a domain to; +8000 is what this README, the vite dev proxy and `docker-compose.yml` use. Both +being live means the deployment works whichever one it is pointed at, instead of +returning 502 from a healthy container. + +`serve.py` is what makes that possible - uvicorn's CLI binds a single `--port`, +but `Server.run()` accepts a list of pre-bound sockets, so it is still one +process. If one port is unavailable it logs and carries on with the other; it +exits non-zero only when nothing is listening. + +```bash +python serve.py # 3000 and 8000 +PORT=8080 python serve.py # only 8080 (PORT pins a single port) +PORTS=80,3000 python serve.py # a different pair +``` + +For local development `uvicorn app.main:app --reload --port 8000` is still the +normal thing to run - one port is all you need, and it gives you autoreload. + +## Persistence: the two volumes a deployment needs + +Most state lives in Postgres, but three things are written to the filesystem, +and in a container those live inside the image - so a redeploy rebuilds the +image and silently discards them: + +| Path | Written by | +|---|---| +| `/app/data/seed_catalogs` | `POST /api/user/products/add`, `/upload-file` - every product added through the UI is appended to the brand's JSON | +| `/app/app/intelligence/artifacts` | the training endpoints - every retrained `*.joblib` model | +| `/app/data` | catalogs saved by the ingestion pipeline | + +Mount a volume on each (the first is inside the third, so two mounts cover all +three): + +``` +/app/data +/app/app/intelligence/artifacts +``` + +In Dokploy, add both under the service's **Volumes**. Named volume or bind +mount, either is fine - `docker compose --profile full up -d` shows the same +two mounts as named volumes. + +Bind mounts normally break this pattern, because they start empty and hide the +seed catalogs and pre-trained models the image ships with. They are safe here: +the image keeps read-only copies at `/app/.bundled`, and on startup +`app/infrastructure/persistence.py` copies in whatever the mounted directory is +missing. It never overwrites an existing file, so a user-added product always +survives the next redeploy rather than being reverted to the bundled catalog. + +If you leave the volumes off, the API still runs and logs a warning naming the +directories that will be lost. + +Override the locations with `DATA_DIR`, `SEED_CATALOG_DIR` and +`MODEL_ARTIFACTS_DIR` if the writable data belongs somewhere else. + ## Pulling the local LLM (one-time) ```bash diff --git a/app/api/routers/mcp_info.py b/app/api/routers/mcp_info.py new file mode 100644 index 0000000..2855edd --- /dev/null +++ b/app/api/routers/mcp_info.py @@ -0,0 +1,167 @@ +""" +REST inspector for the MCP server. + +The MCP endpoint itself speaks streamable HTTP with session handling, which a +browser cannot usefully talk to without a full MCP client implementation. Rather +than ship one in React, these two endpoints expose what the admin page needs - +what tools exist, and what one returns - as ordinary JSON over the REST API the +frontend already authenticates against. + +Admin-only. The tool inventory describes the whole catalog API surface, and the +invoke endpoint executes a tool; neither is something to leave open just because +the tools happen to be read-only. +""" +from __future__ import annotations + +import logging +from typing import Any, Dict, List, Optional + +from fastapi import APIRouter, Body, Depends, HTTPException, Request + +from app.api.deps import require_admin + +logger = logging.getLogger(__name__) +router = APIRouter(prefix="/mcp", tags=["mcp"]) + + +def _server(): + """The FastMCP instance, or None when the optional dependency is absent.""" + try: + from app.mcp_server import mcp + + return mcp + except Exception: # pragma: no cover - depends on fastmcp being installed + return None + + +def _public_url(request: Request, path: str) -> str: + """ + The URL an external MCP client should connect to. + + Built from the forwarded headers rather than a configured constant so it is + right on whichever host the request arrived on - the API's own domain, the + frontend's nginx, or localhost in development. uvicorn runs with + --proxy-headers, so request.url already reflects the external scheme. + """ + base = str(request.base_url).rstrip("/") + return f"{base}{path}" + + +@router.get("/info", dependencies=[Depends(require_admin)]) +async def mcp_info(request: Request) -> Dict[str, Any]: + """Describe the MCP server and list its tools, for the admin page.""" + mcp = _server() + if mcp is None: + return { + "enabled": False, + "reason": ( + "The fastmcp package is not installed, or failed to import. It " + "requires Python 3.10 or newer." + ), + "tools": [], + } + + from app.mcp_server import MCP_PATH + + try: + # run_middleware=False: the auth middleware reads an Authorization header + # from an MCP request context that does not exist on this REST call. The + # caller is already established as an admin by the route dependency. + tools = await mcp.list_tools(run_middleware=False) + except Exception as exc: # noqa: BLE001 + logger.exception("Could not list MCP tools") + raise HTTPException(status_code=500, detail=f"Could not list MCP tools: {exc}") + + described: List[Dict[str, Any]] = [] + for tool in sorted(tools, key=lambda t: t.name): + # Server-side FunctionTool exposes its JSON Schema as `parameters`. + # `inputSchema` is the wire-format name, present on the client-side Tool + # an MCP client receives - checked second so this keeps working if a + # future version converges on one name. + schema = getattr(tool, "parameters", None) or getattr(tool, "inputSchema", None) or {} + properties = schema.get("properties", {}) or {} + required = set(schema.get("required", []) or []) + described.append( + { + "name": tool.name, + "description": (tool.description or "").strip(), + "tags": sorted(getattr(tool, "tags", set()) or []), + "parameters": [ + { + "name": pname, + "type": pinfo.get("type") or "any", + "description": (pinfo.get("description") or "").strip(), + "required": pname in required, + "default": pinfo.get("default"), + } + for pname, pinfo in properties.items() + ], + } + ) + + return { + "enabled": True, + "name": mcp.name, + "version": getattr(mcp, "version", None), + "url": _public_url(request, MCP_PATH), + "transport": "http", + "auth": "Bearer token from POST /api/auth/login", + "tool_count": len(described), + "tools": described, + } + + +@router.post("/tools/{tool_name}", dependencies=[Depends(require_admin)]) +async def invoke_tool( + tool_name: str, + arguments: Optional[Dict[str, Any]] = Body(default=None), +) -> Dict[str, Any]: + """ + Run one MCP tool and return its result - the admin page's "try it" console. + + Exists so the server can be checked from the browser without installing an + MCP client. Every tool is read-only, so this executes the same call an AI + client would make, with the same code path. + """ + mcp = _server() + if mcp is None: + raise HTTPException(status_code=503, detail="The MCP server is not available.") + + try: + # Same reasoning as above: this request authenticated through the REST + # guard, so the MCP auth middleware would have no headers to inspect. + tools = {t.name: t for t in await mcp.list_tools(run_middleware=False)} + except Exception as exc: # noqa: BLE001 + raise HTTPException(status_code=500, detail=f"Could not list MCP tools: {exc}") + + if tool_name not in tools: + raise HTTPException( + status_code=404, + detail=f"No MCP tool named '{tool_name}'. Known tools: {', '.join(sorted(tools))}.", + ) + + try: + result = await mcp.call_tool(tool_name, arguments or {}, run_middleware=False) + except Exception as exc: # noqa: BLE001 + # A tool raising ToolError is a normal outcome worth showing verbatim - + # "no nutrition data for this product" is an answer, not a server fault. + return {"tool": tool_name, "ok": False, "error": str(exc)} + + return {"tool": tool_name, "ok": True, "result": _jsonable(result)} + + +def _jsonable(result: Any) -> Any: + """Reduce a ToolResult to something FastAPI can serialise.""" + for attr in ("structured_content", "data"): + value = getattr(result, attr, None) + if value is not None: + return value + + content = getattr(result, "content", None) + if content is None: + return result + out = [] + for block in content: + text = getattr(block, "text", None) + out.append(text if text is not None else str(block)) + return out diff --git a/app/api/routers/upload.py b/app/api/routers/upload.py index c668cff..0721212 100644 --- a/app/api/routers/upload.py +++ b/app/api/routers/upload.py @@ -1,6 +1,7 @@ from __future__ import annotations import io +import json import uuid import logging import pandas as pd diff --git a/app/api/routers/user_products.py b/app/api/routers/user_products.py index 3b19017..57324dc 100644 --- a/app/api/routers/user_products.py +++ b/app/api/routers/user_products.py @@ -7,6 +7,7 @@ from pydantic import BaseModel, Field from fastapi import APIRouter, Depends, HTTPException, File, UploadFile from app.api.deps import require_permission +from app.infrastructure.settings import SEED_CATALOG_DIR from app.services.vector_store import ( upsert_brand_products, diff --git a/app/core/catalog_engine.py b/app/core/catalog_engine.py index 7955bf0..fd5fed4 100644 --- a/app/core/catalog_engine.py +++ b/app/core/catalog_engine.py @@ -19,7 +19,7 @@ sys.path.append(str(Path(__file__).parent.parent)) from app.services.ollama_service import fetch_brand_catalog_with_gemini, fetch_brand_catalog_exhaustive, fetch_product_details from app.services.image_search import find_all_image_urls, find_product_quantity_openfacts -from app.infrastructure.settings import USE_OLLAMA +from app.infrastructure.settings import DATA_DIR, USE_OLLAMA from app.services.embeddings_service import embed_texts from app.services.vector_store import ensure_brand_schema, upsert_brand_products, get_existing_product_image_id from app.services.s3_service import s3_service @@ -948,20 +948,30 @@ class ProductCatalogEngine: return catalog def save_catalog(self, catalog: Dict[str, Any], filename: str = None) -> str: - """Save catalog to JSON file inside data/ folder""" + """Save catalog to JSON file inside DATA_DIR. + + Resolved against DATA_DIR rather than a bare relative "data/" path: the + latter depends on the process's working directory, so the same call + landed in a different place depending on whether the app was started + from the repo root, from backend/, or by the container's uvicorn - and + only one of those is the directory with a volume mounted on it. + """ if not filename: brand = catalog.get('brand', 'unknown') storage_brand = resolve_parent_brand(brand) safe_brand = storage_brand.replace(' ', '_') ts = catalog.get('generation_timestamp', 'latest') - filename = f"data/catalog_{safe_brand}_{ts}.json" - - Path("data").mkdir(exist_ok=True) - with open(filename, 'w', encoding='utf-8') as f: + filename = f"catalog_{safe_brand}_{ts}.json" + + path = Path(filename) + if not path.is_absolute(): + path = DATA_DIR / path + path.parent.mkdir(parents=True, exist_ok=True) + with open(path, 'w', encoding='utf-8') as f: json.dump(catalog, f, indent=2, ensure_ascii=False) - - logger.info(f"💾 Catalog saved to: {filename}") - return filename + + logger.info(f"💾 Catalog saved to: {path}") + return str(path) # Global engine instance catalog_engine = ProductCatalogEngine() diff --git a/app/infrastructure/persistence.py b/app/infrastructure/persistence.py new file mode 100644 index 0000000..a9ecb66 --- /dev/null +++ b/app/infrastructure/persistence.py @@ -0,0 +1,146 @@ +""" +First-run restore of the assets bundled into the container image. + +The problem this solves +---------------------- +Three directories are written to at runtime - the seed catalogs appended to by +``POST /api/user/products/add``, the generated catalogs from the ingestion +pipeline, and the ``*.joblib`` bundles the training endpoints produce. In a +container all three sit inside the image, so a redeploy silently discards +every one of them. They need to be on a volume. + +Mounting a volume over them introduces the opposite problem. A *named* Docker +volume is seeded from the image the first time it is used, but a *bind* mount +starts empty and merely hides what the image had underneath. Dokploy offers +both and neither announces which you picked, so a bind mount on ``/app/data`` +would leave the API running with no seed catalogs: the next product added would +write a JSON file containing only that product, and every model would report as +untrained. + +The fix is to keep a pristine copy inside the image at a path nobody mounts +(``BUNDLED_ASSETS_DIR``, populated by the Dockerfile) and top up the writable +directory from it on startup. + +Existing files are never overwritten. That is the whole contract: the bundle +supplies what is missing, and anything the running app has already written wins +over the copy baked into the image. Without that rule every redeploy would +revert user-added products back to the bundled catalog. +""" +from __future__ import annotations + +import logging +import shutil +from pathlib import Path +from typing import Tuple + +from app.infrastructure.settings import ( + BUNDLED_ASSETS_DIR, + DATA_DIR, + MODEL_ARTIFACTS_DIR, + SEED_CATALOG_DIR, +) + +logger = logging.getLogger(__name__) + +# (subdirectory under BUNDLED_ASSETS_DIR, writable destination) +_BUNDLES: Tuple[Tuple[str, Path], ...] = ( + ("seed_catalogs", SEED_CATALOG_DIR), + ("artifacts", MODEL_ARTIFACTS_DIR), +) + + +def _restore_one(source: Path, destination: Path) -> int: + """Copy files missing from ``destination``. Returns how many were copied.""" + if not source.is_dir(): + return 0 + + destination.mkdir(parents=True, exist_ok=True) + copied = 0 + for item in sorted(source.iterdir()): + if not item.is_file(): + continue + target = destination / item.name + if target.exists(): + continue + try: + # copy2 rather than copy: it preserves mtime, so "is this artifact + # older than the data it was trained on" stays answerable. + shutil.copy2(item, target) + copied += 1 + except OSError as exc: + logger.warning("Could not restore %s -> %s: %s", item, target, exc) + return copied + + +def restore_bundled_assets() -> None: + """ + Top up the writable directories from the image's read-only bundle. + + Safe to call on every boot: it is a no-op once the volume is populated, and + a no-op outside Docker where BUNDLED_ASSETS_DIR does not exist. + """ + # Create these regardless. A volume mounted at DATA_DIR arrives empty, and + # the ingestion pipeline writes into it without creating it first. + for path in (DATA_DIR, SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR): + try: + path.mkdir(parents=True, exist_ok=True) + except OSError as exc: + logger.error( + "Cannot create writable directory %s: %s. Uploads and trained " + "models will fail to save - check the volume's permissions.", + path, + exc, + ) + + if not BUNDLED_ASSETS_DIR.is_dir(): + logger.debug( + "No bundled asset directory at %s - nothing to restore.", BUNDLED_ASSETS_DIR + ) + return + + for name, destination in _BUNDLES: + copied = _restore_one(BUNDLED_ASSETS_DIR / name, destination) + if copied: + logger.info( + "Restored %d bundled file(s) into %s (first run on this volume).", + copied, + destination, + ) + + _warn_if_not_persistent() + + +def _warn_if_not_persistent() -> None: + """ + Point out that the writable directories are still inside the image. + + Only reachable when BUNDLED_ASSETS_DIR exists, i.e. in the container. If the + destinations were never mounted, everything written to them is lost on the + next redeploy - which looks exactly like the app quietly ignoring uploads, + hours later and with nothing in the logs to connect it to. + """ + unmounted = [p for p in (SEED_CATALOG_DIR, MODEL_ARTIFACTS_DIR) if not _is_mount(p)] + if unmounted: + logger.warning( + "These directories are written at runtime but do not look like " + "mount points: %s. Anything saved there - products added through " + "the UI, retrained models - will be discarded on the next " + "redeploy. Mount a volume on each (see backend/README.md).", + ", ".join(str(p) for p in unmounted), + ) + + +def _is_mount(path: Path) -> bool: + """ + Whether ``path`` sits on a different device than its parent. + + A mounted volume shows up as a device-number change. Falls back to True on + error so a probe failure produces silence rather than a false alarm telling + somebody their correctly-mounted volume is broken. + """ + try: + if path.is_mount(): + return True + return path.stat().st_dev != path.parent.stat().st_dev + except OSError: + return True diff --git a/app/infrastructure/settings.py b/app/infrastructure/settings.py index b180edb..cd94d62 100644 --- a/app/infrastructure/settings.py +++ b/app/infrastructure/settings.py @@ -55,6 +55,52 @@ def _require(name: str, *, feature_flag: str) -> str: return value +# --------------------------------------------------------------------------- +# Writable data directories (persistence) +# --------------------------------------------------------------------------- +# Everything the running app WRITES lives under one of these three paths. They +# are settings rather than hard-coded paths because in a container they must be +# mounted on a volume - otherwise every product added through the UI and every +# retrained model is discarded the next time the image is redeployed. +# +# DATA_DIR generated catalogs (catalog_engine.save_catalog) +# SEED_CATALOG_DIR per-brand JSON catalogs, appended to by +# POST /api/user/products/add and /upload-file +# MODEL_ARTIFACTS_DIR *.joblib bundles written by the training endpoints +# +# See BUNDLED_ASSETS_DIR below for how the read-only copies shipped inside the +# image get into these directories the first time a volume is mounted. +_BACKEND_ROOT = Path(__file__).resolve().parents[2] + + +def _dir(name: str, default: Path) -> Path: + raw = os.getenv(name, "").strip() + return Path(raw).expanduser() if raw else default + + +DATA_DIR = _dir("DATA_DIR", _BACKEND_ROOT / "data") +SEED_CATALOG_DIR = _dir("SEED_CATALOG_DIR", DATA_DIR / "seed_catalogs") +MODEL_ARTIFACTS_DIR = _dir( + "MODEL_ARTIFACTS_DIR", _BACKEND_ROOT / "app" / "intelligence" / "artifacts" +) + +# Pristine copies of the bundled seed catalogs and pre-trained models, placed +# here by the Dockerfile at a path that is never itself mounted over. +# +# This exists because the two ways of mounting a volume behave differently, and +# the difference is silent. Docker copies the image's content into a *named* +# volume the first time it is used, but a *bind* mount starts empty and simply +# hides whatever the image had at that path. Mounting a bind mount on /app/data +# would therefore leave the app with no seed catalogs at all: the next product +# added would write a fresh JSON file containing only that one product, and the +# ML endpoints would report no trained models. +# +# So the image keeps a second, unmounted copy, and restore_bundled_assets() +# (app/infrastructure/persistence.py) fills in whatever the writable directory +# is missing at startup. Empty/absent outside Docker, where nothing is mounted +# and the defaults above already point at the real files. +BUNDLED_ASSETS_DIR = _dir("BUNDLED_ASSETS_DIR", Path("/app/.bundled")) + # --------------------------------------------------------------------------- # Ollama (local LLM) # --------------------------------------------------------------------------- @@ -81,6 +127,14 @@ DB_PORT = os.getenv("DB_PORT", "5432") DB_NAME = os.getenv("DB_NAME", "pgvector") DB_USER = os.getenv("DB_USER", "postgres") DB_PASSWORD = _require("DB_PASSWORD", feature_flag="USE_PGVECTOR") if USE_PGVECTOR else os.getenv("DB_PASSWORD", "") +# How long to wait for the TCP connect before giving up. Matters more than it +# looks: a host that DROPS packets (a firewall, a typo'd DB_HOST) otherwise +# blocks until the OS timeout - about 130 seconds on Linux - and every request +# that touches the database inherits that wait, including /api/health. A short +# ceiling turns "the database is unreachable" into a fast, honest error instead +# of a hung worker and a container the platform decides is unhealthy. +DB_CONNECT_TIMEOUT_SECONDS = int(os.getenv("DB_CONNECT_TIMEOUT_SECONDS", "5")) + DATABASE_URL = os.getenv( "DATABASE_URL", f"postgresql://{DB_USER}:{DB_PASSWORD}@{DB_HOST}:{DB_PORT}/{DB_NAME}", diff --git a/app/intelligence/model_utils.py b/app/intelligence/model_utils.py index 759518f..eeaf5a8 100644 --- a/app/intelligence/model_utils.py +++ b/app/intelligence/model_utils.py @@ -13,9 +13,15 @@ from datetime import datetime, timezone from pathlib import Path from typing import Any, Dict, List, Optional +from app.infrastructure.settings import MODEL_ARTIFACTS_DIR + logger = logging.getLogger(__name__) -ARTIFACTS_DIR = Path(__file__).resolve().parent / "artifacts" +# From settings so it can be pointed at a mounted volume: retraining writes +# *.joblib here, and inside the image those are discarded on the next redeploy, +# silently reverting every model to the version baked in at build time. +# Defaults to this package's own artifacts/ directory. +ARTIFACTS_DIR = MODEL_ARTIFACTS_DIR ARTIFACTS_DIR.mkdir(parents=True, exist_ok=True) diff --git a/app/main.py b/app/main.py index e764438..8dc51b2 100644 --- a/app/main.py +++ b/app/main.py @@ -9,6 +9,7 @@ Run with: from __future__ import annotations import logging +import os import threading from contextlib import asynccontextmanager from pathlib import Path @@ -18,11 +19,12 @@ from fastapi.middleware.cors import CORSMiddleware from fastapi.staticfiles import StaticFiles from fastapi.responses import FileResponse +from app.infrastructure.persistence import restore_bundled_assets from app.infrastructure.settings import API_CORS_ORIGINS from app.api.routers import health, brands, search, chat, catalog, system from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin from app.api.routers import nutrition, nutrition_admin, upload -from app.api.routers import auth, user_products, admin_train +from app.api.routers import auth, user_products, admin_train, mcp_info from app.services.store_db import ensure_store_intelligence_schema from app.services.nutrition_db import ensure_nutrition_schema @@ -32,6 +34,18 @@ logging.basicConfig( ) logger = logging.getLogger(__name__) +try: + from app.mcp_server import MCP_PATH, build_http_app as build_mcp_app + + _mcp_app = build_mcp_app() +except Exception as e: # pragma: no cover - depends on an optional dependency + # fastmcp needs Python >=3.10. The image is 3.11, but an older local + # interpreter should lose the MCP endpoint, not the whole API. + logging.getLogger(__name__).warning("MCP server unavailable: %s", e) + _mcp_app = None + MCP_PATH = "/mcp" + + @asynccontextmanager async def lifespan(_app: FastAPI): """ @@ -43,6 +57,15 @@ async def lifespan(_app: FastAPI): healthcheck pass while that settles. """ + # Runs before the thread below, and synchronously: the seed catalogs and + # model artifacts have to be in place before the first request can read + # them, and it is a handful of file copies on first boot, nothing on every + # boot after that. + try: + restore_bundled_assets() + except Exception as e: + logger.warning("Could not restore bundled assets: %s", e) + def _async_init(): try: ensure_store_intelligence_schema() @@ -58,7 +81,17 @@ async def lifespan(_app: FastAPI): logger.warning("Startup background init warning: %s", e) threading.Thread(target=_async_init, daemon=True).start() - yield + + # The mounted MCP app carries its own lifespan, which starts the session + # manager its request handler depends on. Mounting a sub-app does NOT run + # that lifespan - only the outermost app's is executed - so without this + # the endpoint exists, accepts a connection, and then fails on the first + # message with a session manager that was never started. + if _mcp_app is not None: + async with _mcp_app.lifespan(_mcp_app): + yield + else: + yield app = FastAPI( @@ -98,6 +131,24 @@ app.add_middleware( allow_headers=["*"], ) +# A wrong origin list fails only in the browser, as an opaque "blocked by CORS" +# with a perfectly healthy 200 in the server log - so state the effective list +# at startup, where it can actually be compared against the frontend's URL. +logger.info( + "CORS allowed origins: %s", + ", ".join(API_CORS_ORIGINS) if API_CORS_ORIGINS else "(none - same-origin only)", +) +if API_CORS_ORIGINS and all( + o.startswith(("http://localhost", "http://127.0.0.1")) for o in API_CORS_ORIGINS +): + logger.warning( + "API_CORS_ORIGINS lists only localhost origins (%s). If this API is " + "published on a domain and the frontend is served from a different one, " + "every browser request will be blocked. Set API_CORS_ORIGINS to the " + "frontend's exact origin, e.g. https://catalogue.nearle.ai.in", + ", ".join(API_CORS_ORIGINS), + ) + app.include_router(health.router, prefix="/api") app.include_router(auth.router, prefix="/api") app.include_router(user_products.router, prefix="/api") @@ -116,10 +167,44 @@ app.include_router(store_admin.router, prefix="/api") app.include_router(nutrition.router, prefix="/api") app.include_router(nutrition_admin.router, prefix="/api") app.include_router(upload.router, prefix="/api") +app.include_router(mcp_info.router, prefix="/api") + +# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients, +# not part of the REST surface, and the catch-all SPA route below only skips +# paths it recognises. Mounted rather than routed because it is a whole ASGI +# app with its own request handling. +if _mcp_app is not None: + app.mount(MCP_PATH, _mcp_app) + logger.info("MCP server mounted at %s", MCP_PATH) + +# Serve built frontend static files if dist exists (single-port unified +# deployment). Off in the normal setup: the React app is served by its own +# nginx on catalogue.nearle.ai.in and this API answers on +# mcp.nearle.ai.in, so no dist/ is present here and the JSON root +# handler at the bottom of this file is what responds to /. +# +# The candidates cover both repo layouts - the sibling checkout is named +# `catalogue_frontend`, and only `frontend` was checked before, so this branch +# could never activate even when a build was sitting right next to it. +# FRONTEND_DIST_DIR overrides both when the build lands somewhere else. +_dist_override = os.getenv("FRONTEND_DIST_DIR", "").strip() +_repo_root = Path(__file__).resolve().parents[2] +_dist_candidates = ( + [Path(_dist_override)] + if _dist_override + else [ + _repo_root / "catalogue_frontend" / "dist", + _repo_root / "frontend" / "dist", + Path(__file__).resolve().parents[1] / "frontend" / "dist", + ] +) +FRONTEND_DIST = next( + (p for p in _dist_candidates if (p / "assets").exists()), + _dist_candidates[0], +) -# Serve built frontend static files if dist exists (single-port unified deployment) -FRONTEND_DIST = Path(__file__).resolve().parents[2] / "frontend" / "dist" if FRONTEND_DIST.exists() and (FRONTEND_DIST / "assets").exists(): + logger.info("Serving built frontend from %s", FRONTEND_DIST) app.mount("/assets", StaticFiles(directory=str(FRONTEND_DIST / "assets")), name="assets") @app.get("/{full_path:path}") @@ -130,7 +215,10 @@ if FRONTEND_DIST.exists() and (FRONTEND_DIST / "assets").exists(): # endpoint would look like a successful call to any client. 404 is the # honest answer, and it matters now that the API is also reachable # directly at api.{$DOMAIN} rather than only behind the frontend. - if full_path.startswith(("api", "docs", "redoc", "openapi.json")): + # "mcp" is in this list for the case where the MCP app failed to load: + # the mount would be absent, and without this the SPA fallback would + # answer an MCP client with index.html and a 200. + if full_path.startswith(("api", "docs", "redoc", "openapi.json", "mcp")): raise HTTPException(status_code=404, detail="Not found") file_path = FRONTEND_DIST / full_path if file_path.exists() and file_path.is_file(): diff --git a/app/mcp_server.py b/app/mcp_server.py new file mode 100644 index 0000000..f8bb094 --- /dev/null +++ b/app/mcp_server.py @@ -0,0 +1,522 @@ +""" +MCP server exposing the catalog as tools for AI clients. + +Mounted onto the FastAPI app at /mcp (see app/main.py), so it ships in the same +container and answers on the same host - mcp.nearle.ai.in/mcp. + +Scope: reads only. Every tool here maps to a service function the REST API +already exposes through a GET. None of the write or compute endpoints - catalog +generation, ML training, the upload endpoints, product creation - are reachable +through MCP, because a tool list is chosen from by a model rather than by a +person, and "retrain the models" is not a thing to leave one tool call away. +Adding a write tool later is deliberate work, not an oversight to correct. + +Authentication reuses the access token from POST /api/auth/login. There is no +separate MCP credential: the same Principal, the same expiry, the same +revocation story as the REST API, and nothing new to keep secret. Clients send +it as an Authorization: Bearer header, which _AuthMiddleware checks once for +every tool call and tool listing. + +The tools are hand-written rather than generated from the OpenAPI schema. All +63 routes would technically work, but a model picks a tool by reading its +description, and 63 near-identical auto-generated entries is a worse starting +point than a dozen written to be chosen between. +""" +from __future__ import annotations + +import logging +from typing import Any, Dict, List, Optional + +from fastmcp import FastMCP +from fastmcp.exceptions import ToolError +from fastmcp.server.dependencies import get_http_headers +from fastmcp.server.middleware import CallNext, Middleware, MiddlewareContext + +from app.infrastructure.security import AuthError, Principal, anonymous_principal, decode_access_token +from app.infrastructure.settings import AUTH_ENABLED + +logger = logging.getLogger(__name__) + +MCP_PATH = "/mcp" + +INSTRUCTIONS = """\ +Read-only access to an Indian FMCG product catalog: products and brands, \ +nutrition facts and health scores, per-store inventory and pricing, discounts, \ +and sales analytics. + +Products are identified by a (brand, image_id) pair - image_id is the SKU-like \ +product key, not an image URL. Get one from search_products or \ +list_brand_products before calling any tool that takes an image_id. + +Nutrition figures are per 100g unless the product states otherwise, and health \ +scores run 0-100, higher being better. Prices are in INR. +""" + + +# --------------------------------------------------------------------------- +# Authentication +# --------------------------------------------------------------------------- +def _principal_from_headers() -> Principal: + """ + Resolve the caller from the Authorization header of the current request. + + Raises ToolError rather than AuthError: ToolError is what reaches the client + as a readable message instead of an opaque internal failure. + """ + if not AUTH_ENABLED: + return anonymous_principal() + + # include={"authorization"} is required, not optional: get_http_headers() + # strips the Authorization header by default, so that it is not forwarded to + # downstream services by accident. Without this the header is invisible here + # and every request looks unauthenticated - including valid ones. + headers = get_http_headers(include={"authorization"}) + raw = headers.get("authorization", "") # keys are lowercased by the helper + scheme, _, token = raw.partition(" ") + + if scheme.lower() != "bearer" or not token.strip(): + raise ToolError( + "Not authenticated. Send an Authorization: Bearer header. " + "Get a token from POST /api/auth/login." + ) + try: + return decode_access_token(token.strip()) + except AuthError as exc: + raise ToolError(str(exc)) from exc + + +class _AuthMiddleware(Middleware): + """ + One authentication check covering every tool. + + Sitting here rather than at the top of each tool means a tool added later + cannot be left unguarded by forgetting a line - the same reason the REST + side puts its guard in a route dependency instead of the handler body. + + Listing is checked as well as calling. An unauthenticated client learning + the exact shape of the catalog API is a smaller problem than an + unauthenticated call, but it is not nothing, and there is no reason to + publish it. + """ + + async def on_call_tool(self, context: MiddlewareContext, call_next: CallNext): + principal = _principal_from_headers() + name = getattr(context.message, "name", "?") + logger.info("MCP tool call %r by %s (%s)", name, principal.username, principal.kind) + return await call_next(context) + + async def on_list_tools(self, context: MiddlewareContext, call_next: CallNext): + _principal_from_headers() + return await call_next(context) + + +mcp: FastMCP = FastMCP( + name="nearle-catalogue", + instructions=INSTRUCTIONS, + version="1.0.0", + middleware=[_AuthMiddleware()], +) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- +def _guard(what: str, fn, *args, **kwargs): + """ + Run a service call, turning failures into a ToolError the client can read. + + Nearly every tool here reads Postgres, so the common failure is the database + being unreachable. Left to propagate, that surfaces to the model as an + unexplained error; named, the model can say what went wrong instead of + retrying or inventing an answer. + """ + try: + return fn(*args, **kwargs) + except Exception as exc: # noqa: BLE001 - deliberately broad, see docstring + logger.exception("MCP tool %s failed", what) + raise ToolError(f"{what} failed: {exc}") from exc + + +def _clip(n: Optional[int], default: int, maximum: int) -> int: + """Keep result counts sane - a model will happily ask for 10000 rows.""" + if n is None: + return default + return max(1, min(int(n), maximum)) + + +def _slim(product: Dict[str, Any]) -> Dict[str, Any]: + """ + Drop the fields a model has no use for. + + Embeddings especially: a 384-float vector per product would swamp the + context window and says nothing a model can reason about. + """ + return { + k: v + for k, v in product.items() + if k not in {"embedding", "embedding_text", "distance"} and v not in (None, "", [], {}) + } + + +# --------------------------------------------------------------------------- +# Catalog +# --------------------------------------------------------------------------- +@mcp.tool( + annotations={"readOnlyHint": True}, + tags={"catalog"}, +) +def search_products( + query: str, + brand: Optional[str] = None, + category: Optional[str] = None, + limit: Optional[int] = 10, +) -> List[Dict[str, Any]]: + """Search the catalog by meaning, not keywords. + + The main entry point: use this to turn a description ("sugar-free biscuits", + "something to replace butter") into concrete products. Returns each match + with its brand and image_id, which the nutrition, store and recommendation + tools take as input. + + Args: + query: What to look for, in natural language. + brand: Restrict to one brand. Omit to search everything. + category: Restrict to one category, e.g. "Dairy". + limit: How many products to return (1-50). + """ + from app.services.rag_service import retrieve + + top_k = _clip(limit, 10, 50) + found = _guard("search_products", retrieve, query, brand=brand, top_k=top_k, category=category) + return [ + _slim( + { + "brand": p.brand, + "image_id": p.image_id, + "title": p.title, + "category": p.category, + "description": p.description, + "price_range": p.price_range, + "selling_price": p.final_selling_price or p.selling_price, + "size_variants": p.size_variants, + "barcode": p.barcode, + } + ) + for p in found + ] + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"}) +def list_brands() -> List[str]: + """List every brand in the catalog. + + Useful for grounding a brand name before passing it to another tool, since + brand names must match the catalog's spelling. + """ + from app.services.vector_store import list_available_brands + + return _guard("list_brands", list_available_brands) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"}) +def list_categories(brand: str) -> List[str]: + """List the product categories a brand sells. + + Args: + brand: Brand name, as returned by list_brands. + """ + from app.services.vector_store import list_categories_for_brand + + return _guard("list_categories", list_categories_for_brand, brand) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"}) +def list_brand_products( + brand: str, + category: Optional[str] = None, + limit: Optional[int] = 25, +) -> List[Dict[str, Any]]: + """List a brand's products, newest catalog entries first. + + Browsing, as opposed to search_products' semantic matching. Prefer + search_products when the user described what they want rather than naming a + brand outright. + + Args: + brand: Brand name, as returned by list_brands. + category: Restrict to one category. + limit: How many products to return (1-100). + """ + from app.services.vector_store import get_products_by_brand + + rows = _guard( + "list_brand_products", + get_products_by_brand, + brand, + limit=_clip(limit, 25, 100), + category=category, + ) + return [_slim(r) for r in rows] + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"}) +def get_product(brand: str, image_id: str) -> Dict[str, Any]: + """Get the full catalog record for one product. + + Args: + brand: Brand name. + image_id: The product key from search_products or list_brand_products. + """ + from app.services.vector_store import get_product_by_image_id + + row = _guard("get_product", get_product_by_image_id, brand, image_id) + if not row: + raise ToolError( + f"No product {image_id!r} for brand {brand!r}. " + "Check the pair with search_products." + ) + return _slim(row) + + +# --------------------------------------------------------------------------- +# Nutrition +# --------------------------------------------------------------------------- +@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"}) +def get_nutrition(brand: str, image_id: str) -> Dict[str, Any]: + """Get nutrition facts, health score and dietary flags for one product. + + Covers macros and key micronutrients per 100g, a 0-100 health score, diet + tags (vegetarian, high-protein, ...) and declared allergens. + + Args: + brand: Brand name. + image_id: The product key from search_products. + """ + from app.services.nutrition_db import get_full_nutrition + + row = _guard("get_nutrition", get_full_nutrition, brand, image_id) + if not row: + raise ToolError( + f"No nutrition data for {brand}/{image_id}. Not every catalog " + "product has been enriched yet." + ) + return _slim(row) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"}) +def find_healthier_alternatives( + brand: str, + image_id: str, + limit: Optional[int] = 5, +) -> List[Dict[str, Any]]: + """Find comparable products with a better health score. + + Answers "what should I buy instead of this?" - matches are drawn from the + same category, so the suggestion is a real substitute rather than merely + something healthier. + + Args: + brand: Brand name of the product to improve on. + image_id: Product key of the product to improve on. + limit: How many alternatives to return (1-20). + """ + from app.services.nutrition_alternatives_service import find_alternatives + + return _guard( + "find_healthier_alternatives", + find_alternatives, + brand, + image_id, + top_k=_clip(limit, 5, 20), + ) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"}) +def filter_products_by_nutrition( + sort_by: str = "health_score", + order: str = "desc", + category: Optional[str] = None, + diet_tag: Optional[str] = None, + exclude_allergen: Optional[str] = None, + limit: Optional[int] = 20, +) -> List[Dict[str, Any]]: + """Rank and filter products by a nutritional measure. + + The tool for questions shaped like "highest protein snacks", "lowest sugar + dairy", or "gluten-free products without nuts". + + Args: + sort_by: Field to rank by - health_score, protein_g, total_sugar_g, + dietary_fiber_g, sodium_mg, calories_kcal, total_fat_g. + order: "desc" for highest first, "asc" for lowest first. + category: Restrict to one category. + diet_tag: Keep only products carrying this tag, e.g. "High Protein". + exclude_allergen: Drop products declaring this allergen, e.g. "Dairy". + limit: How many products to return (1-100). + """ + from app.services.nutrition_db import query_products + + return _guard( + "filter_products_by_nutrition", + query_products, + sort_by=sort_by, + order=order, + category=category, + diet_tag=diet_tag, + exclude_allergen=exclude_allergen, + limit=_clip(limit, 20, 100), + ) + + +# --------------------------------------------------------------------------- +# Stores +# --------------------------------------------------------------------------- +@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"}) +def list_stores() -> List[Dict[str, Any]]: + """List the retail stores, with city and tier. + + Call this first for anything store-specific - the other store tools need a + store_id from here. + """ + from app.services.store_db import list_stores as _list + + return _guard("list_stores", _list) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"}) +def get_store_inventory( + store_id: str, + category: Optional[str] = None, + in_stock_only: bool = False, + limit: Optional[int] = 25, +) -> List[Dict[str, Any]]: + """List what one store carries, with stock levels and its own pricing. + + Prices vary per store, so this is the tool for "what does it cost at X", + not the catalog price_range. + + Args: + store_id: Store identifier from list_stores. + category: Restrict to one category. + in_stock_only: Drop products currently out of stock. + limit: How many products to return (1-100). + """ + from app.services.store_db import get_store_products + + return _guard( + "get_store_inventory", + get_store_products, + store_id, + category=category, + in_stock_only=in_stock_only, + limit=_clip(limit, 25, 100), + ) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"}) +def get_store_discounts(store_id: str, limit: Optional[int] = 25) -> List[Dict[str, Any]]: + """List the discounts currently allocated at one store. + + Args: + store_id: Store identifier from list_stores. + limit: How many discounts to return (1-100). + """ + from app.services.store_db import get_latest_discounts + + return _guard( + "get_store_discounts", get_latest_discounts, store_id, limit=_clip(limit, 25, 100) + ) + + +# --------------------------------------------------------------------------- +# Analytics +# --------------------------------------------------------------------------- +@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"}) +def get_trending( + window: str = "weekly", + scope: str = "overall", + scope_value: Optional[str] = None, + limit: Optional[int] = 10, +) -> List[Dict[str, Any]]: + """List what is selling fastest right now. + + Args: + window: Period to measure over - "daily", "weekly" or "monthly". + scope: "overall", or "store"/"category" to narrow it. + scope_value: The store_id or category name, when scope is not "overall". + limit: How many products to return (1-50). + """ + from app.services.store_db import get_trending as _trending + + return _guard( + "get_trending", + _trending, + window_label=window, + scope=scope, + scope_value=scope_value, + top_k=_clip(limit, 10, 50), + ) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"}) +def get_top_products( + by: str = "revenue", + limit: Optional[int] = 10, + ascending: bool = False, +) -> List[Dict[str, Any]]: + """Rank products across every store by a sales measure. + + Args: + by: What to rank on - "revenue", "units" or "orders". + limit: How many products to return (1-50). + ascending: True ranks worst-performing first. + """ + from app.services.analytics_service import top_products + + return _guard( + "get_top_products", top_products, by=by, limit=_clip(limit, 10, 50), ascending=ascending + ) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"}) +def get_store_analytics(store_id: str) -> Dict[str, Any]: + """Get one store's sales dashboard - revenue, orders, top sellers, stock health. + + Args: + store_id: Store identifier from list_stores. + """ + from app.services.analytics_service import store_dashboard + + return _guard("get_store_analytics", store_dashboard, store_id) + + +@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"}) +def get_recommendations( + brand: str, + image_id: str, + limit: Optional[int] = 5, +) -> List[Dict[str, Any]]: + """List products frequently bought together with this one. + + Co-purchase based, so this answers "what else goes in the basket", which is + a different question from find_healthier_alternatives' "what instead". + + Args: + brand: Brand name. + image_id: Product key from search_products. + limit: How many recommendations to return (1-20). + """ + from app.services.recommendation_service import recommend_for_product + + return _guard( + "get_recommendations", + recommend_for_product, + brand, + image_id, + top_k=_clip(limit, 5, 20), + ) + + +def build_http_app(path: str = MCP_PATH): + """The ASGI app to mount on FastAPI. See app/main.py for the lifespan wiring.""" + return mcp.http_app(path="/", transport="http", stateless_http=True) diff --git a/app/services/image_search.py b/app/services/image_search.py index 2bccee4..5d696dd 100644 --- a/app/services/image_search.py +++ b/app/services/image_search.py @@ -42,6 +42,8 @@ actually resolves to real image bytes above a minimum size - this is what stops garbage/placeholder/expired URLs from reaching the S3 upload step and failing there silently. """ +from __future__ import annotations + from typing import Optional, List import requests from urllib.parse import urlparse diff --git a/app/services/ollama_service.py b/app/services/ollama_service.py index 7ee1632..dddcbd5 100644 --- a/app/services/ollama_service.py +++ b/app/services/ollama_service.py @@ -1,3 +1,5 @@ +from __future__ import annotations + from typing import List, Dict, Any, Optional import json import re diff --git a/app/services/vector_store.py b/app/services/vector_store.py index 070b1fd..560a3ac 100644 --- a/app/services/vector_store.py +++ b/app/services/vector_store.py @@ -9,7 +9,10 @@ import re import time import psycopg -from app.infrastructure.settings import DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD +from app.infrastructure.settings import ( + DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD, + DB_CONNECT_TIMEOUT_SECONDS, +) from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand @@ -93,7 +96,15 @@ def _connect() -> Optional[psycopg.Connection]: dbname=DB_NAME, user=DB_USER, password=DB_PASSWORD, - autocommit=True + autocommit=True, + # Without this, a host that DROPS packets rather than refusing them + # - a firewall, a wrong DB_HOST - blocks here until the OS gives up, + # which is around 130 seconds on Linux. Every caller of _connect() + # inherits that: /api/health stops answering, the container's + # healthcheck times out, and the platform pulls the service out of + # its load balancer. "Database unreachable" then presents as a Bad + # Gateway on every route, including ones that never touch the DB. + connect_timeout=DB_CONNECT_TIMEOUT_SECONDS, ) except Exception as e: logger.error(f"Vector DB connection failed: {e}") diff --git a/docker-compose.yml b/docker-compose.yml index 3a007b2..f924573 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -9,7 +9,8 @@ # container layer, and so `ollama pull` model files persist outside Docker. # # Usage: -# docker compose up -d +# docker compose up -d # Postgres only (the default) +# docker compose --profile full up -d # Postgres + the API, with volumes # # then in backend/.env: DB_HOST=localhost, DB_PORT=5432, DB_NAME=pgvector, # # DB_USER=postgres, DB_PASSWORD= services: @@ -34,5 +35,53 @@ services: timeout: 5s retries: 10 + # The API. Started only with `--profile full`, so the default + # `docker compose up -d` still brings up Postgres alone as it always did. + # + # docker compose --profile full up -d --build + # + # Present mainly as the reference for the two volume mounts below - the paths + # are the same ones to configure in Dokploy. + backend: + profiles: ["full"] + build: . + container_name: catalog_rag_backend + restart: unless-stopped + depends_on: + postgres: + condition: service_healthy + env_file: .env + environment: + # Inside a container `localhost` is the container itself, so the DB is + # reached by service name over the compose network. + DB_HOST: postgres + DB_PORT: "5432" + DB_PASSWORD: ${POSTGRES_PASSWORD:-changeme} + # Ollama runs natively on the host, not in compose (see the note above). + OLLAMA_BASE_URL: http://host.docker.internal:11434 + extra_hosts: + - "host.docker.internal:host-gateway" + # The container listens on 3000 and 8000 at once (see serve.py), so either + # side of this mapping can change without touching the image. 8000 is + # published because vite.config.js proxies /api to 127.0.0.1:8000. + ports: + - "8000:8000" + volumes: + # WITHOUT THESE TWO MOUNTS, a redeploy silently discards: + # - every product added through the UI (POST /api/user/products/add and + # /upload-file append to data/seed_catalogs/*.json), and + # - every retrained model (the training endpoints write *.joblib). + # Both directories live inside the image, so rebuilding it resets them to + # whatever was committed to the repo. + # + # The image also carries a pristine copy at /app/.bundled, and the app + # tops up anything missing on startup without overwriting what is already + # there - so these work whether they are named volumes or bind mounts. + # See app/infrastructure/persistence.py. + - catalog_rag_data:/app/data + - catalog_rag_artifacts:/app/app/intelligence/artifacts + volumes: catalog_rag_pgdata: + catalog_rag_data: + catalog_rag_artifacts: diff --git a/requirements.txt b/requirements.txt index a85a840..ffa93d6 100644 --- a/requirements.txt +++ b/requirements.txt @@ -14,6 +14,13 @@ python-dotenv>=1.0.1 # deliberately no bcrypt/argon2/passlib dependency here. PyJWT>=2.9.0 +# --- MCP (Model Context Protocol) --- +# Serves the catalog as tools for AI clients at /mcp (see app/mcp_server.py), +# mounted onto the FastAPI app so it needs no separate process or container. +# Requires Python >=3.10; the Dockerfile's python:3.11-slim satisfies that, and +# app/main.py degrades to serving the REST API alone if the import fails. +fastmcp>=3.4.7 + # --- Database / pgvector --- psycopg[binary]>=3.2.3 pgvector>=0.2.5 diff --git a/scripts/check_deploy.sh b/scripts/check_deploy.sh new file mode 100755 index 0000000..bb1631b --- /dev/null +++ b/scripts/check_deploy.sh @@ -0,0 +1,82 @@ +#!/usr/bin/env bash +# Post-deploy smoke test. +# +# ./scripts/check_deploy.sh # https://mcp.nearle.ai.in +# ./scripts/check_deploy.sh http://localhost:3000 # a local container +# +# Separates the three failures that all look alike from a browser: +# 502 everywhere - the container is not running (it exited at startup) +# 200 + database:false - the API is fine, Postgres is not +# 200 + database:true - working +set -uo pipefail + +BASE="${1:-https://mcp.nearle.ai.in}" +pass=0; fail=0 + +hit() { curl -sS -m 20 -o /tmp/_body -w '%{http_code}' "$BASE$1" 2>/dev/null || echo 000; } + +check() { # path expected label + local code; code=$(hit "$1") + if [ "$code" = "$2" ]; then printf ' \033[32mPASS\033[0m %-16s %s\n' "$1" "$3"; pass=$((pass+1)) + else printf ' \033[31mFAIL\033[0m %-16s expected %s, got %s\n' "$1" "$2" "$code"; fail=$((fail+1)); fi +} + +echo "Checking $BASE" +echo +echo "Is the container running at all?" +code=$(hit /) +if [ "$code" = "502" ] || [ "$code" = "000" ]; then + echo " FAIL / -> $code" + echo + echo " The container is not serving. It almost certainly exited at startup." + echo " Read the Dokploy logs; settings.py names the missing value explicitly." + echo " Confirm the image actually contains /app/.env - the build log must show" + echo " a 'COPY .env .' step, and .dockerignore must not list .env." + exit 1 +fi +printf ' \033[32mPASS\033[0m %-16s container is up\n' "/" +pass=$((pass+1)) + +echo +echo "Routes that must not depend on the database:" +check /docs 200 "interactive API docs" +check /openapi.json 200 "OpenAPI schema" +check /api/health 200 "health endpoint answers" + +echo +echo "Dependencies (reported in the body; 'false' does not mean the API is broken):" +health=$(curl -sS -m 20 "$BASE/api/health" 2>/dev/null) +db=$(printf '%s' "$health" | sed -n 's/.*"database":\([a-z]*\).*/\1/p') +ol=$(printf '%s' "$health" | sed -n 's/.*"ollama":\([a-z]*\).*/\1/p') +if [ "$db" = "true" ]; then printf ' \033[32mPASS\033[0m database connected\n'; pass=$((pass+1)) +else + printf ' \033[33mWARN\033[0m database NOT connected\n' + echo " The API works; catalog pages will be empty. Check DB_HOST/DB_USER/" + echo " DB_PASSWORD/DB_NAME. A wrong password logs 'password authentication" + echo " failed' rather than a timeout." +fi +[ "$ol" = "true" ] \ + && printf ' \033[32mPASS\033[0m ollama connected\n' \ + || printf ' \033[33mWARN\033[0m ollama not connected (/api/chat unavailable; expected if USE_OLLAMA=false)\n' + +echo +echo "Auth:" +code=$(curl -sS -m 20 -o /dev/null -w '%{http_code}' -X POST "$BASE/api/auth/login" \ + -H 'Content-Type: application/json' -d '{"username":"admin","password":"definitely-not-the-password"}' 2>/dev/null) +if [ "$code" = "401" ]; then printf ' \033[32mPASS\033[0m wrong password rejected (401)\n'; pass=$((pass+1)) +elif [ "$code" = "200" ]; then + printf ' \033[31mFAIL\033[0m a WRONG PASSWORD WAS ACCEPTED - AUTH_ALLOW_ANY_LOGIN is true.\n' + echo " Anyone who finds this host can sign in as admin. Set it to false." + fail=$((fail+1)) +else printf ' \033[31mFAIL\033[0m login endpoint returned %s\n' "$code"; fail=$((fail+1)); fi + +echo +echo "MCP:" +code=$(curl -sS -m 20 -o /dev/null -w '%{http_code}' "$BASE/api/mcp/info" 2>/dev/null) +[ "$code" = "401" ] \ + && { printf ' \033[32mPASS\033[0m /api/mcp/info guarded (401 without a token)\n'; pass=$((pass+1)); } \ + || printf ' \033[33mWARN\033[0m /api/mcp/info returned %s (expected 401)\n' "$code" + +echo +echo "-------- $pass passed, $fail failed --------" +[ "$fail" -eq 0 ] diff --git a/serve.py b/serve.py new file mode 100644 index 0000000..79f8780 --- /dev/null +++ b/serve.py @@ -0,0 +1,170 @@ +""" +Container entry point: serve the API on every port the platform might route to. + +The frontend image answers on both 80 and 3000 (``listen 80; listen 3000;`` in +nginx.conf) so it works whatever port the deployment is configured to hit. This +does the same for the API, which otherwise has to guess: Dokploy routes a domain +to one container port, this project's own README, vite.config.js proxy and +docker-compose mapping all say 8000, and picking wrong produces a 502 with a +perfectly healthy container behind it. + +uvicorn's CLI binds a single ``--port``, but ``Server.run()`` accepts a list of +already-bound sockets, so one process can listen on several. That is what this +does - no extra worker, no second process to supervise. + +Usage:: + + python serve.py # binds PORTS (default "3000,8000") + PORT=8080 python serve.py # binds only 8080 + python serve.py --healthcheck # probe mode, used by HEALTHCHECK + +Run it directly rather than through ``uvicorn app.main:app`` when you need the +multi-port behaviour; the plain uvicorn command still works for local +development where one port is all anyone wants. +""" +from __future__ import annotations + +import logging +import os +import socket +import sys +import urllib.error +import urllib.request + +logger = logging.getLogger("serve") + +# Both of the ports this project actually uses anywhere: 3000 because that is +# what the platform routes a domain to by default, 8000 because the README, +# the vite dev proxy and docker-compose all target it. +DEFAULT_PORTS = "3000,8000" + +HOST = os.getenv("HOST", "0.0.0.0") + + +def configured_ports() -> list[int]: + """ + The ports to bind, in order. + + ``PORT`` wins over ``PORTS`` and is treated as an explicit single choice: + setting it means "serve here", not "serve here as well". ``PORTS`` takes a + comma-separated list for the both-at-once case. + """ + raw = os.getenv("PORT") or os.getenv("PORTS") or DEFAULT_PORTS + + ports: list[int] = [] + for chunk in raw.split(","): + chunk = chunk.strip() + if not chunk: + continue + try: + port = int(chunk) + except ValueError: + logger.warning("Ignoring non-numeric port %r in %r", chunk, raw) + continue + if not 1 <= port <= 65535: + logger.warning("Ignoring out-of-range port %d", port) + continue + if port not in ports: # binding the same port twice would fail + ports.append(port) + + if not ports: + logger.error("No usable port in %r - falling back to %s", raw, DEFAULT_PORTS) + return [int(p) for p in DEFAULT_PORTS.split(",")] + return ports + + +def _bind(port: int) -> socket.socket | None: + """Bind one listening socket, or return None with the reason logged.""" + sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + try: + sock.bind((HOST, port)) + except OSError as exc: + # Not fatal on its own. If the platform only routes to one of these, + # losing the other (already in use, not permitted) should not take the + # service down - _run() fails only when nothing at all is listening. + logger.warning("Could not bind %s:%d - %s", HOST, port, exc) + sock.close() + return None + sock.listen(2048) + sock.set_inheritable(True) + return sock + + +def _run() -> int: + import uvicorn + + ports = configured_ports() + sockets = [s for s in (_bind(p) for p in ports) if s is not None] + + if not sockets: + logger.error( + "Could not bind any of %s. The API is not listening; exiting so the " + "platform restarts or reports the container as failed.", + ", ".join(str(p) for p in ports), + ) + return 1 + + bound = [s.getsockname()[1] for s in sockets] + logger.info("Serving on %s port(s): %s", HOST, ", ".join(str(p) for p in bound)) + + config = uvicorn.Config( + "app.main:app", + # proxy_headers/forwarded_allow_ips: this container is never reached + # directly - nginx, and Dokploy's Traefik, sit in front of it. Without + # them uvicorn ignores X-Forwarded-Proto and reports every request as + # plain http, so any redirect or generated absolute URL would downgrade + # an https request. + proxy_headers=True, + forwarded_allow_ips="*", + ) + uvicorn.Server(config).run(sockets=sockets) + return 0 + + +def _healthcheck() -> int: + """ + Probe the API, passing if ANY configured port answers. + + Shares configured_ports() with the server so the check cannot drift from + what is actually bound - the reason this lives here rather than being a + python -c one-liner in the Dockerfile. + + Probes "/" rather than /api/health, and the distinction matters. This is a + LIVENESS check: the only question it may ask is "is this process still + serving HTTP", because the platform's answer to "no" is to stop routing + traffic to the container. + + /api/health is a READINESS report - it dials Postgres and Ollama to say + whether they are reachable. Using it here couples the container's existence + to its dependencies: an unreachable database made /api/health block for the + OS TCP timeout, the check timed out, the container was marked unhealthy, and + a service that was running perfectly well returned Bad Gateway on every + route - including the ones that never touch the database. "/" is served from + memory and does no I/O at all, so it can only fail if the app really is gone. + """ + ports = configured_ports() + for port in ports: + try: + with urllib.request.urlopen(f"http://127.0.0.1:{port}/", timeout=8) as resp: + if 200 <= resp.status < 500: + return 0 + except urllib.error.HTTPError as exc: + # An HTTP status - even 404 - means something is listening and + # routing. That is exactly what liveness asks. + if exc.code < 500: + return 0 + except (urllib.error.URLError, OSError, ValueError): + continue + print( + f"health: no response from any of {', '.join(str(p) for p in ports)}", + file=sys.stderr, + ) + return 1 + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO, format="%(levelname)s: %(message)s") + if "--healthcheck" in sys.argv: + sys.exit(_healthcheck()) + sys.exit(_run())