The deployment landed on mcp.nearle.ai.in, not the mcp.catalogue.nearle.ai.in these references were written against. Only comments, README and .env.example are affected - nothing reads the hostname at runtime - but a wrong host in the connection snippet is a wrong host somebody pastes into an MCP client. API_CORS_ORIGINS is unchanged: it takes the FRONTEND's origin (catalogue.nearle.ai.in), not the API's, so moving the API does not affect it. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
523 lines
18 KiB
Python
523 lines
18 KiB
Python
"""
|
|
MCP server exposing the catalog as tools for AI clients.
|
|
|
|
Mounted onto the FastAPI app at /mcp (see app/main.py), so it ships in the same
|
|
container and answers on the same host - mcp.nearle.ai.in/mcp.
|
|
|
|
Scope: reads only. Every tool here maps to a service function the REST API
|
|
already exposes through a GET. None of the write or compute endpoints - catalog
|
|
generation, ML training, the upload endpoints, product creation - are reachable
|
|
through MCP, because a tool list is chosen from by a model rather than by a
|
|
person, and "retrain the models" is not a thing to leave one tool call away.
|
|
Adding a write tool later is deliberate work, not an oversight to correct.
|
|
|
|
Authentication reuses the access token from POST /api/auth/login. There is no
|
|
separate MCP credential: the same Principal, the same expiry, the same
|
|
revocation story as the REST API, and nothing new to keep secret. Clients send
|
|
it as an Authorization: Bearer header, which _AuthMiddleware checks once for
|
|
every tool call and tool listing.
|
|
|
|
The tools are hand-written rather than generated from the OpenAPI schema. All
|
|
63 routes would technically work, but a model picks a tool by reading its
|
|
description, and 63 near-identical auto-generated entries is a worse starting
|
|
point than a dozen written to be chosen between.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
from fastmcp import FastMCP
|
|
from fastmcp.exceptions import ToolError
|
|
from fastmcp.server.dependencies import get_http_headers
|
|
from fastmcp.server.middleware import CallNext, Middleware, MiddlewareContext
|
|
|
|
from app.infrastructure.security import AuthError, Principal, anonymous_principal, decode_access_token
|
|
from app.infrastructure.settings import AUTH_ENABLED
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
MCP_PATH = "/mcp"
|
|
|
|
INSTRUCTIONS = """\
|
|
Read-only access to an Indian FMCG product catalog: products and brands, \
|
|
nutrition facts and health scores, per-store inventory and pricing, discounts, \
|
|
and sales analytics.
|
|
|
|
Products are identified by a (brand, image_id) pair - image_id is the SKU-like \
|
|
product key, not an image URL. Get one from search_products or \
|
|
list_brand_products before calling any tool that takes an image_id.
|
|
|
|
Nutrition figures are per 100g unless the product states otherwise, and health \
|
|
scores run 0-100, higher being better. Prices are in INR.
|
|
"""
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Authentication
|
|
# ---------------------------------------------------------------------------
|
|
def _principal_from_headers() -> Principal:
|
|
"""
|
|
Resolve the caller from the Authorization header of the current request.
|
|
|
|
Raises ToolError rather than AuthError: ToolError is what reaches the client
|
|
as a readable message instead of an opaque internal failure.
|
|
"""
|
|
if not AUTH_ENABLED:
|
|
return anonymous_principal()
|
|
|
|
# include={"authorization"} is required, not optional: get_http_headers()
|
|
# strips the Authorization header by default, so that it is not forwarded to
|
|
# downstream services by accident. Without this the header is invisible here
|
|
# and every request looks unauthenticated - including valid ones.
|
|
headers = get_http_headers(include={"authorization"})
|
|
raw = headers.get("authorization", "") # keys are lowercased by the helper
|
|
scheme, _, token = raw.partition(" ")
|
|
|
|
if scheme.lower() != "bearer" or not token.strip():
|
|
raise ToolError(
|
|
"Not authenticated. Send an Authorization: Bearer <token> header. "
|
|
"Get a token from POST /api/auth/login."
|
|
)
|
|
try:
|
|
return decode_access_token(token.strip())
|
|
except AuthError as exc:
|
|
raise ToolError(str(exc)) from exc
|
|
|
|
|
|
class _AuthMiddleware(Middleware):
|
|
"""
|
|
One authentication check covering every tool.
|
|
|
|
Sitting here rather than at the top of each tool means a tool added later
|
|
cannot be left unguarded by forgetting a line - the same reason the REST
|
|
side puts its guard in a route dependency instead of the handler body.
|
|
|
|
Listing is checked as well as calling. An unauthenticated client learning
|
|
the exact shape of the catalog API is a smaller problem than an
|
|
unauthenticated call, but it is not nothing, and there is no reason to
|
|
publish it.
|
|
"""
|
|
|
|
async def on_call_tool(self, context: MiddlewareContext, call_next: CallNext):
|
|
principal = _principal_from_headers()
|
|
name = getattr(context.message, "name", "?")
|
|
logger.info("MCP tool call %r by %s (%s)", name, principal.username, principal.kind)
|
|
return await call_next(context)
|
|
|
|
async def on_list_tools(self, context: MiddlewareContext, call_next: CallNext):
|
|
_principal_from_headers()
|
|
return await call_next(context)
|
|
|
|
|
|
mcp: FastMCP = FastMCP(
|
|
name="nearle-catalogue",
|
|
instructions=INSTRUCTIONS,
|
|
version="1.0.0",
|
|
middleware=[_AuthMiddleware()],
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Helpers
|
|
# ---------------------------------------------------------------------------
|
|
def _guard(what: str, fn, *args, **kwargs):
|
|
"""
|
|
Run a service call, turning failures into a ToolError the client can read.
|
|
|
|
Nearly every tool here reads Postgres, so the common failure is the database
|
|
being unreachable. Left to propagate, that surfaces to the model as an
|
|
unexplained error; named, the model can say what went wrong instead of
|
|
retrying or inventing an answer.
|
|
"""
|
|
try:
|
|
return fn(*args, **kwargs)
|
|
except Exception as exc: # noqa: BLE001 - deliberately broad, see docstring
|
|
logger.exception("MCP tool %s failed", what)
|
|
raise ToolError(f"{what} failed: {exc}") from exc
|
|
|
|
|
|
def _clip(n: Optional[int], default: int, maximum: int) -> int:
|
|
"""Keep result counts sane - a model will happily ask for 10000 rows."""
|
|
if n is None:
|
|
return default
|
|
return max(1, min(int(n), maximum))
|
|
|
|
|
|
def _slim(product: Dict[str, Any]) -> Dict[str, Any]:
|
|
"""
|
|
Drop the fields a model has no use for.
|
|
|
|
Embeddings especially: a 384-float vector per product would swamp the
|
|
context window and says nothing a model can reason about.
|
|
"""
|
|
return {
|
|
k: v
|
|
for k, v in product.items()
|
|
if k not in {"embedding", "embedding_text", "distance"} and v not in (None, "", [], {})
|
|
}
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Catalog
|
|
# ---------------------------------------------------------------------------
|
|
@mcp.tool(
|
|
annotations={"readOnlyHint": True},
|
|
tags={"catalog"},
|
|
)
|
|
def search_products(
|
|
query: str,
|
|
brand: Optional[str] = None,
|
|
category: Optional[str] = None,
|
|
limit: Optional[int] = 10,
|
|
) -> List[Dict[str, Any]]:
|
|
"""Search the catalog by meaning, not keywords.
|
|
|
|
The main entry point: use this to turn a description ("sugar-free biscuits",
|
|
"something to replace butter") into concrete products. Returns each match
|
|
with its brand and image_id, which the nutrition, store and recommendation
|
|
tools take as input.
|
|
|
|
Args:
|
|
query: What to look for, in natural language.
|
|
brand: Restrict to one brand. Omit to search everything.
|
|
category: Restrict to one category, e.g. "Dairy".
|
|
limit: How many products to return (1-50).
|
|
"""
|
|
from app.services.rag_service import retrieve
|
|
|
|
top_k = _clip(limit, 10, 50)
|
|
found = _guard("search_products", retrieve, query, brand=brand, top_k=top_k, category=category)
|
|
return [
|
|
_slim(
|
|
{
|
|
"brand": p.brand,
|
|
"image_id": p.image_id,
|
|
"title": p.title,
|
|
"category": p.category,
|
|
"description": p.description,
|
|
"price_range": p.price_range,
|
|
"selling_price": p.final_selling_price or p.selling_price,
|
|
"size_variants": p.size_variants,
|
|
"barcode": p.barcode,
|
|
}
|
|
)
|
|
for p in found
|
|
]
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
|
def list_brands() -> List[str]:
|
|
"""List every brand in the catalog.
|
|
|
|
Useful for grounding a brand name before passing it to another tool, since
|
|
brand names must match the catalog's spelling.
|
|
"""
|
|
from app.services.vector_store import list_available_brands
|
|
|
|
return _guard("list_brands", list_available_brands)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
|
def list_categories(brand: str) -> List[str]:
|
|
"""List the product categories a brand sells.
|
|
|
|
Args:
|
|
brand: Brand name, as returned by list_brands.
|
|
"""
|
|
from app.services.vector_store import list_categories_for_brand
|
|
|
|
return _guard("list_categories", list_categories_for_brand, brand)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
|
def list_brand_products(
|
|
brand: str,
|
|
category: Optional[str] = None,
|
|
limit: Optional[int] = 25,
|
|
) -> List[Dict[str, Any]]:
|
|
"""List a brand's products, newest catalog entries first.
|
|
|
|
Browsing, as opposed to search_products' semantic matching. Prefer
|
|
search_products when the user described what they want rather than naming a
|
|
brand outright.
|
|
|
|
Args:
|
|
brand: Brand name, as returned by list_brands.
|
|
category: Restrict to one category.
|
|
limit: How many products to return (1-100).
|
|
"""
|
|
from app.services.vector_store import get_products_by_brand
|
|
|
|
rows = _guard(
|
|
"list_brand_products",
|
|
get_products_by_brand,
|
|
brand,
|
|
limit=_clip(limit, 25, 100),
|
|
category=category,
|
|
)
|
|
return [_slim(r) for r in rows]
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
|
def get_product(brand: str, image_id: str) -> Dict[str, Any]:
|
|
"""Get the full catalog record for one product.
|
|
|
|
Args:
|
|
brand: Brand name.
|
|
image_id: The product key from search_products or list_brand_products.
|
|
"""
|
|
from app.services.vector_store import get_product_by_image_id
|
|
|
|
row = _guard("get_product", get_product_by_image_id, brand, image_id)
|
|
if not row:
|
|
raise ToolError(
|
|
f"No product {image_id!r} for brand {brand!r}. "
|
|
"Check the pair with search_products."
|
|
)
|
|
return _slim(row)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Nutrition
|
|
# ---------------------------------------------------------------------------
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
|
|
def get_nutrition(brand: str, image_id: str) -> Dict[str, Any]:
|
|
"""Get nutrition facts, health score and dietary flags for one product.
|
|
|
|
Covers macros and key micronutrients per 100g, a 0-100 health score, diet
|
|
tags (vegetarian, high-protein, ...) and declared allergens.
|
|
|
|
Args:
|
|
brand: Brand name.
|
|
image_id: The product key from search_products.
|
|
"""
|
|
from app.services.nutrition_db import get_full_nutrition
|
|
|
|
row = _guard("get_nutrition", get_full_nutrition, brand, image_id)
|
|
if not row:
|
|
raise ToolError(
|
|
f"No nutrition data for {brand}/{image_id}. Not every catalog "
|
|
"product has been enriched yet."
|
|
)
|
|
return _slim(row)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
|
|
def find_healthier_alternatives(
|
|
brand: str,
|
|
image_id: str,
|
|
limit: Optional[int] = 5,
|
|
) -> List[Dict[str, Any]]:
|
|
"""Find comparable products with a better health score.
|
|
|
|
Answers "what should I buy instead of this?" - matches are drawn from the
|
|
same category, so the suggestion is a real substitute rather than merely
|
|
something healthier.
|
|
|
|
Args:
|
|
brand: Brand name of the product to improve on.
|
|
image_id: Product key of the product to improve on.
|
|
limit: How many alternatives to return (1-20).
|
|
"""
|
|
from app.services.nutrition_alternatives_service import find_alternatives
|
|
|
|
return _guard(
|
|
"find_healthier_alternatives",
|
|
find_alternatives,
|
|
brand,
|
|
image_id,
|
|
top_k=_clip(limit, 5, 20),
|
|
)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"nutrition"})
|
|
def filter_products_by_nutrition(
|
|
sort_by: str = "health_score",
|
|
order: str = "desc",
|
|
category: Optional[str] = None,
|
|
diet_tag: Optional[str] = None,
|
|
exclude_allergen: Optional[str] = None,
|
|
limit: Optional[int] = 20,
|
|
) -> List[Dict[str, Any]]:
|
|
"""Rank and filter products by a nutritional measure.
|
|
|
|
The tool for questions shaped like "highest protein snacks", "lowest sugar
|
|
dairy", or "gluten-free products without nuts".
|
|
|
|
Args:
|
|
sort_by: Field to rank by - health_score, protein_g, total_sugar_g,
|
|
dietary_fiber_g, sodium_mg, calories_kcal, total_fat_g.
|
|
order: "desc" for highest first, "asc" for lowest first.
|
|
category: Restrict to one category.
|
|
diet_tag: Keep only products carrying this tag, e.g. "High Protein".
|
|
exclude_allergen: Drop products declaring this allergen, e.g. "Dairy".
|
|
limit: How many products to return (1-100).
|
|
"""
|
|
from app.services.nutrition_db import query_products
|
|
|
|
return _guard(
|
|
"filter_products_by_nutrition",
|
|
query_products,
|
|
sort_by=sort_by,
|
|
order=order,
|
|
category=category,
|
|
diet_tag=diet_tag,
|
|
exclude_allergen=exclude_allergen,
|
|
limit=_clip(limit, 20, 100),
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Stores
|
|
# ---------------------------------------------------------------------------
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
|
|
def list_stores() -> List[Dict[str, Any]]:
|
|
"""List the retail stores, with city and tier.
|
|
|
|
Call this first for anything store-specific - the other store tools need a
|
|
store_id from here.
|
|
"""
|
|
from app.services.store_db import list_stores as _list
|
|
|
|
return _guard("list_stores", _list)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
|
|
def get_store_inventory(
|
|
store_id: str,
|
|
category: Optional[str] = None,
|
|
in_stock_only: bool = False,
|
|
limit: Optional[int] = 25,
|
|
) -> List[Dict[str, Any]]:
|
|
"""List what one store carries, with stock levels and its own pricing.
|
|
|
|
Prices vary per store, so this is the tool for "what does it cost at X",
|
|
not the catalog price_range.
|
|
|
|
Args:
|
|
store_id: Store identifier from list_stores.
|
|
category: Restrict to one category.
|
|
in_stock_only: Drop products currently out of stock.
|
|
limit: How many products to return (1-100).
|
|
"""
|
|
from app.services.store_db import get_store_products
|
|
|
|
return _guard(
|
|
"get_store_inventory",
|
|
get_store_products,
|
|
store_id,
|
|
category=category,
|
|
in_stock_only=in_stock_only,
|
|
limit=_clip(limit, 25, 100),
|
|
)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"stores"})
|
|
def get_store_discounts(store_id: str, limit: Optional[int] = 25) -> List[Dict[str, Any]]:
|
|
"""List the discounts currently allocated at one store.
|
|
|
|
Args:
|
|
store_id: Store identifier from list_stores.
|
|
limit: How many discounts to return (1-100).
|
|
"""
|
|
from app.services.store_db import get_latest_discounts
|
|
|
|
return _guard(
|
|
"get_store_discounts", get_latest_discounts, store_id, limit=_clip(limit, 25, 100)
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Analytics
|
|
# ---------------------------------------------------------------------------
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
|
|
def get_trending(
|
|
window: str = "weekly",
|
|
scope: str = "overall",
|
|
scope_value: Optional[str] = None,
|
|
limit: Optional[int] = 10,
|
|
) -> List[Dict[str, Any]]:
|
|
"""List what is selling fastest right now.
|
|
|
|
Args:
|
|
window: Period to measure over - "daily", "weekly" or "monthly".
|
|
scope: "overall", or "store"/"category" to narrow it.
|
|
scope_value: The store_id or category name, when scope is not "overall".
|
|
limit: How many products to return (1-50).
|
|
"""
|
|
from app.services.store_db import get_trending as _trending
|
|
|
|
return _guard(
|
|
"get_trending",
|
|
_trending,
|
|
window_label=window,
|
|
scope=scope,
|
|
scope_value=scope_value,
|
|
top_k=_clip(limit, 10, 50),
|
|
)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
|
|
def get_top_products(
|
|
by: str = "revenue",
|
|
limit: Optional[int] = 10,
|
|
ascending: bool = False,
|
|
) -> List[Dict[str, Any]]:
|
|
"""Rank products across every store by a sales measure.
|
|
|
|
Args:
|
|
by: What to rank on - "revenue", "units" or "orders".
|
|
limit: How many products to return (1-50).
|
|
ascending: True ranks worst-performing first.
|
|
"""
|
|
from app.services.analytics_service import top_products
|
|
|
|
return _guard(
|
|
"get_top_products", top_products, by=by, limit=_clip(limit, 10, 50), ascending=ascending
|
|
)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"analytics"})
|
|
def get_store_analytics(store_id: str) -> Dict[str, Any]:
|
|
"""Get one store's sales dashboard - revenue, orders, top sellers, stock health.
|
|
|
|
Args:
|
|
store_id: Store identifier from list_stores.
|
|
"""
|
|
from app.services.analytics_service import store_dashboard
|
|
|
|
return _guard("get_store_analytics", store_dashboard, store_id)
|
|
|
|
|
|
@mcp.tool(annotations={"readOnlyHint": True}, tags={"catalog"})
|
|
def get_recommendations(
|
|
brand: str,
|
|
image_id: str,
|
|
limit: Optional[int] = 5,
|
|
) -> List[Dict[str, Any]]:
|
|
"""List products frequently bought together with this one.
|
|
|
|
Co-purchase based, so this answers "what else goes in the basket", which is
|
|
a different question from find_healthier_alternatives' "what instead".
|
|
|
|
Args:
|
|
brand: Brand name.
|
|
image_id: Product key from search_products.
|
|
limit: How many recommendations to return (1-20).
|
|
"""
|
|
from app.services.recommendation_service import recommend_for_product
|
|
|
|
return _guard(
|
|
"get_recommendations",
|
|
recommend_for_product,
|
|
brand,
|
|
image_id,
|
|
top_k=_clip(limit, 5, 20),
|
|
)
|
|
|
|
|
|
def build_http_app(path: str = MCP_PATH):
|
|
"""The ASGI app to mount on FastAPI. See app/main.py for the lifespan wiring."""
|
|
return mcp.http_app(path="/", transport="http", stateless_http=True)
|