Files
sriram c7e4d59188 Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-01 12:17:42 +05:30

125 lines
5.7 KiB
Python

"""Link a listing to its canonical product (one real-world variant).
From most to least certain:
1. GTIN - same barcode -> auto
2. MPN - same manufacturer part number (laptops) -> auto
3. variant key - same brand, model, RAM and storage -> auto
4. fuzzy - model names ≥ AUTO_RATIO similar AND every
hard attribute (RAM, storage, processor) equal -> auto
≥ REVIEW_RATIO -> pending (review queue)
5. otherwise a new product is created for the variant.
A listing that states too little to identify a variant (no storage on a
phone title, for example) is stored but not linked to any product.
"""
from __future__ import annotations
from dataclasses import dataclass
from decimal import Decimal
from typing import List, Optional
from rapidfuzz import fuzz
from app.electronics.normalise.title_parser import laptop_line, processor_is_specific
AUTO_RATIO = 92
REVIEW_RATIO = 85
@dataclass
class MatchDecision:
product_id: Optional[int] # None -> create a new product
method: str
confidence: float
review_status: str
def _eq(a: Optional[Decimal], b: Optional[Decimal]) -> bool:
if a is None or b is None:
return a is None and b is None
return Decimal(a) == Decimal(b)
def _number_tokens(model_norm: str) -> set:
"""Tokens that carry a digit ("s25", "a37", "15", "2a", "8a"). Two models
whose number tokens differ are different products, however similar the
rest of the name is: Galaxy S25 vs S26, A27 vs A37, iPhone 15 vs 16."""
return {t for t in (model_norm or "").split() if any(ch.isdigit() for ch in t)}
def _lines_compatible(a: str, b: str) -> bool:
"""One model line is the other plus/minus extra words, and they agree on
every number token they both carry ("15" is not "15s", "slim 3" is not
"slim 5")."""
ta, tb = set(a.split()), set(b.split())
return bool(ta and tb) and (ta <= tb or tb <= ta)
def decide(listing, candidates: List[dict]) -> Optional[MatchDecision]:
"""`listing` is a models.Listing with variant_key/model_norm set;
`candidates` are product rows of the same brand and category."""
if not listing.variant_key:
return None
if listing.gtin:
for c in candidates:
if c.get("gtin") and c["gtin"] == listing.gtin:
return MatchDecision(c["id"], "gtin", 0.99, "auto")
if listing.model_number:
for c in candidates:
if c.get("mpn") and c["mpn"].lower() == listing.model_number.lower():
return MatchDecision(c["id"], "mpn", 0.97, "auto")
for c in candidates:
if c["variant_key"] == listing.variant_key:
return MatchDecision(c["id"], "variant_key", 0.95, "auto")
# Laptops: the same configuration (exact CPU model, RAM, storage) within a
# compatible model line - "ideapad slim 3" and "ideapad slim 3 15amn8" -
# is the same product. Two compatible candidates means the title is too
# vague to choose ("pavilion" vs "pavilion 14" and "pavilion 15"): review.
if listing.category == "laptops" and processor_is_specific(listing.processor) \
and listing.ram_gb is not None and listing.storage_gb is not None:
line = laptop_line(listing.model_norm)
same_config = [
c for c in candidates
if c.get("processor") == listing.processor
and _eq(c.get("ram_gb"), listing.ram_gb) and _eq(c.get("storage_gb"), listing.storage_gb)
and _lines_compatible(line, laptop_line(c["model_norm"]))
]
if len(same_config) == 1:
return MatchDecision(same_config[0]["id"], "variant_key", 0.85, "auto")
if len(same_config) > 1:
return MatchDecision(same_config[0]["id"], "variant_key", 0.6, "pending")
# One side does not state the RAM ("Apple iPhone 15 (128 GB)"). Same model
# and storage with exactly one candidate is the same variant; with several
# candidates it is ambiguous and goes to review.
same_model = [c for c in candidates
if c["model_norm"] == listing.model_norm and _eq(c.get("storage_gb"), listing.storage_gb)
and (c.get("ram_gb") is None) != (listing.ram_gb is None)]
if len(same_model) == 1:
return MatchDecision(same_model[0]["id"], "variant_key", 0.85, "auto")
if len(same_model) > 1:
return MatchDecision(same_model[0]["id"], "variant_key", 0.6, "pending")
best, best_score = None, 0.0
for c in candidates:
if not (_eq(c.get("storage_gb"), listing.storage_gb) and _eq(c.get("ram_gb"), listing.ram_gb)):
continue
if listing.category == "laptops" and (c.get("processor") or listing.processor) and c.get("processor") != listing.processor:
continue
if _number_tokens(c["model_norm"]) != _number_tokens(listing.model_norm):
continue
score = fuzz.token_set_ratio(c["model_norm"], listing.model_norm or "")
# token_set_ratio treats "galaxy s24" and "galaxy s24 ultra" as a
# subset match (100). Different words mean different models.
extra = set((listing.model_norm or "").split()) ^ set(c["model_norm"].split())
if extra:
score = min(score, fuzz.ratio(c["model_norm"], listing.model_norm or ""))
if score > best_score:
best, best_score = c, score
if best is not None and best_score >= AUTO_RATIO:
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.9, 2), "auto")
if best is not None and best_score >= REVIEW_RATIO:
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.8, 2), "pending")
return MatchDecision(None, "variant_key", 0.9, "auto")