Files
catalogue_backend/app/mcp_server.py
Suriyakumarvijayanayagam 00216ae834 Point the docs at the real API domain, mcp.nearle.ai.in
The deployment landed on mcp.nearle.ai.in, not the mcp.catalogue.nearle.ai.in
these references were written against. Only comments, README and .env.example
are affected - nothing reads the hostname at runtime - but a wrong host in the
connection snippet is a wrong host somebody pastes into an MCP client.

API_CORS_ORIGINS is unchanged: it takes the FRONTEND's origin
(catalogue.nearle.ai.in), not the API's, so moving the API does not affect it.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-13 13:29:34 +05:30

523 lines
18 KiB
Python

"""
MCP server exposing the catalog as tools for AI clients.
Mounted onto the FastAPI app at /mcp (see app/main.py), so it ships in the same
container and answers on the same host - mcp.nearle.ai.in/mcp.
Scope: reads only. Every tool here maps to a service function the REST API
already exposes through a GET. None of the write or compute endpoints - catalog
generation, ML training, the upload endpoints, product creation - are reachable
through MCP, because a tool list is chosen from by a model rather than by a
person, and "retrain the models" is not a thing to leave one tool call away.
Adding a write tool later is deliberate work, not an oversight to correct.
Authentication reuses the access token from POST /api/auth/login. There is no
separate MCP credential: the same Principal, the same expiry, the same
revocation story as the REST API, and nothing new to keep secret. Clients send
it as an Authorization: Bearer header, which _AuthMiddleware checks once for
every tool call and tool listing.
The tools are hand-written rather than generated from the OpenAPI schema. All
63 routes would technically work, but a model picks a tool by reading its
description, and 63 near-identical auto-generated entries is a worse starting
point than a dozen written to be chosen between.
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List, Optional
from fastmcp import FastMCP
from fastmcp.exceptions import ToolError
from fastmcp.server.dependencies import get_http_headers
from fastmcp.server.middleware import CallNext, Middleware, MiddlewareContext
from app.infrastructure.security import AuthError, Principal, anonymous_principal, decode_access_token
from app.infrastructure.settings import AUTH_ENABLED
logger = logging.getLogger(__name__)
MCP_PATH = "/mcp"
INSTRUCTIONS = """\
Read-only access to an Indian FMCG product catalog: products and brands, \
nutrition facts and health scores, per-store inventory and pricing, discounts, \
and sales analytics.
Products are identified by a (brand, image_id) pair - image_id is the SKU-like \
product key, not an image URL. Get one from search_products or \
list_brand_products before calling any tool that takes an image_id.
Nutrition figures are per 100g unless the product states otherwise, and health \
scores run 0-100, higher being better. Prices are in INR.
"""
# ---------------------------------------------------------------------------
# Authentication
# ---------------------------------------------------------------------------
def _principal_from_headers() -> Principal:
"""
Resolve the caller from the Authorization header of the current request.
Raises ToolError rather than AuthError: ToolError is what reaches the client
as a readable message instead of an opaque internal failure.
"""
if not AUTH_ENABLED:
return anonymous_principal()
# include={"authorization"} is required, not optional: get_http_headers()
# strips the Authorization header by default, so that it is not forwarded to
# downstream services by accident. Without this the header is invisible here
# and every request looks unauthenticated - including valid ones.
headers = get_http_headers(include={"authorization"})
raw = headers.get("authorization", "") # keys are lowercased by the helper
scheme, _, token = raw.partition(" ")
if scheme.lower() != "bearer" or not token.strip():
raise ToolError(
"Not authenticated. Send an Authorization: Bearer <token> header. "
"Get a token from POST /api/auth/login."
)
try:
return decode_access_token(token.strip())
except AuthError as exc:
raise ToolError(str(exc)) from exc
class _AuthMiddleware(Middleware):
"""
One authentication check covering every tool.
Sitting here rather than at the top of each tool means a tool added later
cannot be left unguarded by forgetting a line - the same reason the REST
side puts its guard in a route dependency instead of the handler body.
Listing is checked as well as calling. An unauthenticated client learning
the exact shape of the catalog API is a smaller problem than an
unauthenticated call, but it is not nothing, and there is no reason to
publish it.
"""
async def on_call_tool(self, context: MiddlewareContext, call_next: CallNext):
principal = _principal_from_headers()
name = getattr(context.message, "name", "?")
logger.info("MCP tool call %r by %s (%s)", name, principal.username, principal.kind)
return await call_next(context)
async def on_list_tools(self, context: MiddlewareContext, call_next: CallNext):
_principal_from_headers()
return await call_next(context)
mcp: FastMCP = FastMCP(
name="nearle-catalogue",
instructions=INSTRUCTIONS,
version="1.0.0",
middleware=[_AuthMiddleware()],
)
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def _guard(what: str, fn, *args, **kwargs):
"""
Run a service call, turning failures into a ToolError the client can read.
Nearly every tool here reads Postgres, so the common failure is the database
being unreachable. Left to propagate, that surfaces to the model as an
unexplained error; named, the model can say what went wrong instead of
retrying or inventing an answer.
"""
try:
return fn(*args, **kwargs)
except Exception as exc: # noqa: BLE001 - deliberately broad, see docstring
logger.exception("MCP tool %s failed", what)
raise ToolError(f"{what} failed: {exc}") from exc
def _clip(n: Optional[int], default: int, maximum: int) -> int:
"""Keep result counts sane - a model will happily ask for 10000 rows."""
if n is None:
return default
return max(1, min(int(n), maximum))
def _slim(product: Dict[str, Any]) -> Dict[str, Any]:
"""
Drop the fields a model has no use for.
Embeddings especially: a 384-float vector per product would swamp the
context window and says nothing a model can reason about.
"""
return {
k: v
for k, v in product.items()
if k not in {"embedding", "embedding_text", "distance"} and v not in (None, "", [], {})
}
# ---------------------------------------------------------------------------
# Catalog
# ---------------------------------------------------------------------------
@mcp.tool(
annotations={"readOnlyHint": True},
tags={"catalog"},
)
def search_products(
query: str,
brand: Optional[str] = None,
category: Optional[str] = None,
limit: Optional[int] = 10,
) -> List[Dict[str, Any]]:
"""Search the catalog by meaning, not keywords.
The main entry point: use this to turn a description ("sugar-free biscuits",
"something to replace butter") into concrete products. Returns each match
with its brand and image_id, which the nutrition, store and recommendation
tools take as input.
Args:
query: What to look for, in natural language.
brand: Restrict to one brand. Omit to search everything.
category: Restrict to one category, e.g. "Dairy".
limit: How many products to return (1-50).
"""
from app.services.rag_service import retrieve
top_k = _clip(limit, 10, 50)
found = _guard("search_products", retrieve, query, brand=brand, top_k=top_k, category=category)
return [
_slim(
{
"brand": p.brand,
"image_id": p.image_id,
"title": p.title,
"category": p.category,
"description": p.description,
"price_range": p.price_range,
"selling_price": p.final_selling_price or p.selling_price,
"size_variants": p.size_variants,
"barcode": p.barcode,
}
)
for p in found
]
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def list_brands() -> List[str]:
"""List every brand in the catalog.
Useful for grounding a brand name before passing it to another tool, since
brand names must match the catalog's spelling.
"""
from app.services.vector_store import list_available_brands
return _guard("list_brands", list_available_brands)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def list_categories(brand: str) -> List[str]:
"""List the product categories a brand sells.
Args:
brand: Brand name, as returned by list_brands.
"""
from app.services.vector_store import list_categories_for_brand
return _guard("list_categories", list_categories_for_brand, brand)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def list_brand_products(
brand: str,
category: Optional[str] = None,
limit: Optional[int] = 25,
) -> List[Dict[str, Any]]:
"""List a brand's products, newest catalog entries first.
Browsing, as opposed to search_products' semantic matching. Prefer
search_products when the user described what they want rather than naming a
brand outright.
Args:
brand: Brand name, as returned by list_brands.
category: Restrict to one category.
limit: How many products to return (1-100).
"""
from app.services.vector_store import get_products_by_brand
rows = _guard(
"list_brand_products",
get_products_by_brand,
brand,
limit=_clip(limit, 25, 100),
category=category,
)
return [_slim(r) for r in rows]
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def get_product(brand: str, image_id: str) -> Dict[str, Any]:
"""Get the full catalog record for one product.
Args:
brand: Brand name.
image_id: The product key from search_products or list_brand_products.
"""
from app.services.vector_store import get_product_by_image_id
row = _guard("get_product", get_product_by_image_id, brand, image_id)
if not row:
raise ToolError(
f"No product {image_id!r} for brand {brand!r}. "
"Check the pair with search_products."
)
return _slim(row)
# ---------------------------------------------------------------------------
# Nutrition
# ---------------------------------------------------------------------------
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
def get_nutrition(brand: str, image_id: str) -> Dict[str, Any]:
"""Get nutrition facts, health score and dietary flags for one product.
Covers macros and key micronutrients per 100g, a 0-100 health score, diet
tags (vegetarian, high-protein, ...) and declared allergens.
Args:
brand: Brand name.
image_id: The product key from search_products.
"""
from app.services.nutrition_db import get_full_nutrition
row = _guard("get_nutrition", get_full_nutrition, brand, image_id)
if not row:
raise ToolError(
f"No nutrition data for {brand}/{image_id}. Not every catalog "
"product has been enriched yet."
)
return _slim(row)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
def find_healthier_alternatives(
brand: str,
image_id: str,
limit: Optional[int] = 5,
) -> List[Dict[str, Any]]:
"""Find comparable products with a better health score.
Answers "what should I buy instead of this?" - matches are drawn from the
same category, so the suggestion is a real substitute rather than merely
something healthier.
Args:
brand: Brand name of the product to improve on.
image_id: Product key of the product to improve on.
limit: How many alternatives to return (1-20).
"""
from app.services.nutrition_alternatives_service import find_alternatives
return _guard(
"find_healthier_alternatives",
find_alternatives,
brand,
image_id,
top_k=_clip(limit, 5, 20),
)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
def filter_products_by_nutrition(
sort_by: str = "health_score",
order: str = "desc",
category: Optional[str] = None,
diet_tag: Optional[str] = None,
exclude_allergen: Optional[str] = None,
limit: Optional[int] = 20,
) -> List[Dict[str, Any]]:
"""Rank and filter products by a nutritional measure.
The tool for questions shaped like "highest protein snacks", "lowest sugar
dairy", or "gluten-free products without nuts".
Args:
sort_by: Field to rank by - health_score, protein_g, total_sugar_g,
dietary_fiber_g, sodium_mg, calories_kcal, total_fat_g.
order: "desc" for highest first, "asc" for lowest first.
category: Restrict to one category.
diet_tag: Keep only products carrying this tag, e.g. "High Protein".
exclude_allergen: Drop products declaring this allergen, e.g. "Dairy".
limit: How many products to return (1-100).
"""
from app.services.nutrition_db import query_products
return _guard(
"filter_products_by_nutrition",
query_products,
sort_by=sort_by,
order=order,
category=category,
diet_tag=diet_tag,
exclude_allergen=exclude_allergen,
limit=_clip(limit, 20, 100),
)
# ---------------------------------------------------------------------------
# Stores
# ---------------------------------------------------------------------------
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
def list_stores() -> List[Dict[str, Any]]:
"""List the retail stores, with city and tier.
Call this first for anything store-specific - the other store tools need a
store_id from here.
"""
from app.services.store_db import list_stores as _list
return _guard("list_stores", _list)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
def get_store_inventory(
store_id: str,
category: Optional[str] = None,
in_stock_only: bool = False,
limit: Optional[int] = 25,
) -> List[Dict[str, Any]]:
"""List what one store carries, with stock levels and its own pricing.
Prices vary per store, so this is the tool for "what does it cost at X",
not the catalog price_range.
Args:
store_id: Store identifier from list_stores.
category: Restrict to one category.
in_stock_only: Drop products currently out of stock.
limit: How many products to return (1-100).
"""
from app.services.store_db import get_store_products
return _guard(
"get_store_inventory",
get_store_products,
store_id,
category=category,
in_stock_only=in_stock_only,
limit=_clip(limit, 25, 100),
)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
def get_store_discounts(store_id: str, limit: Optional[int] = 25) -> List[Dict[str, Any]]:
"""List the discounts currently allocated at one store.
Args:
store_id: Store identifier from list_stores.
limit: How many discounts to return (1-100).
"""
from app.services.store_db import get_latest_discounts
return _guard(
"get_store_discounts", get_latest_discounts, store_id, limit=_clip(limit, 25, 100)
)
# ---------------------------------------------------------------------------
# Analytics
# ---------------------------------------------------------------------------
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
def get_trending(
window: str = "weekly",
scope: str = "overall",
scope_value: Optional[str] = None,
limit: Optional[int] = 10,
) -> List[Dict[str, Any]]:
"""List what is selling fastest right now.
Args:
window: Period to measure over - "daily", "weekly" or "monthly".
scope: "overall", or "store"/"category" to narrow it.
scope_value: The store_id or category name, when scope is not "overall".
limit: How many products to return (1-50).
"""
from app.services.store_db import get_trending as _trending
return _guard(
"get_trending",
_trending,
window_label=window,
scope=scope,
scope_value=scope_value,
top_k=_clip(limit, 10, 50),
)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
def get_top_products(
by: str = "revenue",
limit: Optional[int] = 10,
ascending: bool = False,
) -> List[Dict[str, Any]]:
"""Rank products across every store by a sales measure.
Args:
by: What to rank on - "revenue", "units" or "orders".
limit: How many products to return (1-50).
ascending: True ranks worst-performing first.
"""
from app.services.analytics_service import top_products
return _guard(
"get_top_products", top_products, by=by, limit=_clip(limit, 10, 50), ascending=ascending
)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
def get_store_analytics(store_id: str) -> Dict[str, Any]:
"""Get one store's sales dashboard - revenue, orders, top sellers, stock health.
Args:
store_id: Store identifier from list_stores.
"""
from app.services.analytics_service import store_dashboard
return _guard("get_store_analytics", store_dashboard, store_id)
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
def get_recommendations(
brand: str,
image_id: str,
limit: Optional[int] = 5,
) -> List[Dict[str, Any]]:
"""List products frequently bought together with this one.
Co-purchase based, so this answers "what else goes in the basket", which is
a different question from find_healthier_alternatives' "what instead".
Args:
brand: Brand name.
image_id: Product key from search_products.
limit: How many recommendations to return (1-20).
"""
from app.services.recommendation_service import recommend_for_product
return _guard(
"get_recommendations",
recommend_for_product,
brand,
image_id,
top_k=_clip(limit, 5, 20),
)
def build_http_app(path: str = MCP_PATH):
"""The ASGI app to mount on FastAPI. See app/main.py for the lifespan wiring."""
return mcp.http_app(path="/", transport="http", stateless_http=True)