Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

@@ -0,0 +1,124 @@
"""Link a listing to its canonical product (one real-world variant).
From most to least certain:
1. GTIN - same barcode -> auto
2. MPN - same manufacturer part number (laptops) -> auto
3. variant key - same brand, model, RAM and storage -> auto
4. fuzzy - model names ≥ AUTO_RATIO similar AND every
hard attribute (RAM, storage, processor) equal -> auto
≥ REVIEW_RATIO -> pending (review queue)
5. otherwise a new product is created for the variant.
A listing that states too little to identify a variant (no storage on a
phone title, for example) is stored but not linked to any product.
"""
from __future__ import annotations
from dataclasses import dataclass
from decimal import Decimal
from typing import List, Optional
from rapidfuzz import fuzz
from app.electronics.normalise.title_parser import laptop_line, processor_is_specific
AUTO_RATIO = 92
REVIEW_RATIO = 85
@dataclass
class MatchDecision:
product_id: Optional[int] # None -> create a new product
method: str
confidence: float
review_status: str
def _eq(a: Optional[Decimal], b: Optional[Decimal]) -> bool:
if a is None or b is None:
return a is None and b is None
return Decimal(a) == Decimal(b)
def _number_tokens(model_norm: str) -> set:
"""Tokens that carry a digit ("s25", "a37", "15", "2a", "8a"). Two models
whose number tokens differ are different products, however similar the
rest of the name is: Galaxy S25 vs S26, A27 vs A37, iPhone 15 vs 16."""
return {t for t in (model_norm or "").split() if any(ch.isdigit() for ch in t)}
def _lines_compatible(a: str, b: str) -> bool:
"""One model line is the other plus/minus extra words, and they agree on
every number token they both carry ("15" is not "15s", "slim 3" is not
"slim 5")."""
ta, tb = set(a.split()), set(b.split())
return bool(ta and tb) and (ta <= tb or tb <= ta)
def decide(listing, candidates: List[dict]) -> Optional[MatchDecision]:
"""`listing` is a models.Listing with variant_key/model_norm set;
`candidates` are product rows of the same brand and category."""
if not listing.variant_key:
return None
if listing.gtin:
for c in candidates:
if c.get("gtin") and c["gtin"] == listing.gtin:
return MatchDecision(c["id"], "gtin", 0.99, "auto")
if listing.model_number:
for c in candidates:
if c.get("mpn") and c["mpn"].lower() == listing.model_number.lower():
return MatchDecision(c["id"], "mpn", 0.97, "auto")
for c in candidates:
if c["variant_key"] == listing.variant_key:
return MatchDecision(c["id"], "variant_key", 0.95, "auto")
# Laptops: the same configuration (exact CPU model, RAM, storage) within a
# compatible model line - "ideapad slim 3" and "ideapad slim 3 15amn8" -
# is the same product. Two compatible candidates means the title is too
# vague to choose ("pavilion" vs "pavilion 14" and "pavilion 15"): review.
if listing.category == "laptops" and processor_is_specific(listing.processor) \
and listing.ram_gb is not None and listing.storage_gb is not None:
line = laptop_line(listing.model_norm)
same_config = [
c for c in candidates
if c.get("processor") == listing.processor
and _eq(c.get("ram_gb"), listing.ram_gb) and _eq(c.get("storage_gb"), listing.storage_gb)
and _lines_compatible(line, laptop_line(c["model_norm"]))
]
if len(same_config) == 1:
return MatchDecision(same_config[0]["id"], "variant_key", 0.85, "auto")
if len(same_config) > 1:
return MatchDecision(same_config[0]["id"], "variant_key", 0.6, "pending")
# One side does not state the RAM ("Apple iPhone 15 (128 GB)"). Same model
# and storage with exactly one candidate is the same variant; with several
# candidates it is ambiguous and goes to review.
same_model = [c for c in candidates
if c["model_norm"] == listing.model_norm and _eq(c.get("storage_gb"), listing.storage_gb)
and (c.get("ram_gb") is None) != (listing.ram_gb is None)]
if len(same_model) == 1:
return MatchDecision(same_model[0]["id"], "variant_key", 0.85, "auto")
if len(same_model) > 1:
return MatchDecision(same_model[0]["id"], "variant_key", 0.6, "pending")
best, best_score = None, 0.0
for c in candidates:
if not (_eq(c.get("storage_gb"), listing.storage_gb) and _eq(c.get("ram_gb"), listing.ram_gb)):
continue
if listing.category == "laptops" and (c.get("processor") or listing.processor) and c.get("processor") != listing.processor:
continue
if _number_tokens(c["model_norm"]) != _number_tokens(listing.model_norm):
continue
score = fuzz.token_set_ratio(c["model_norm"], listing.model_norm or "")
# token_set_ratio treats "galaxy s24" and "galaxy s24 ultra" as a
# subset match (100). Different words mean different models.
extra = set((listing.model_norm or "").split()) ^ set(c["model_norm"].split())
if extra:
score = min(score, fuzz.ratio(c["model_norm"], listing.model_norm or ""))
if score > best_score:
best, best_score = c, score
if best is not None and best_score >= AUTO_RATIO:
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.9, 2), "auto")
if best is not None and best_score >= REVIEW_RATIO:
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.8, 2), "pending")
return MatchDecision(None, "variant_key", 0.9, "auto")

View File

@@ -0,0 +1,105 @@
"""Rebuild canonical products for a category from the listings already stored.
Products and listing links are derived data: every fact lives on the listing
(title, snippet evidence, specs, URL). When the parsing or matching rules
improve, this re-runs them over the stored listings - no network requests -
and keeps each image attached to the listing it was found on.
Review decisions (approved/rejected links) are lost, because the products they
pointed at are rebuilt; uncertain matches simply come back to the queue.
"""
from __future__ import annotations
import logging
from decimal import Decimal
from typing import Dict, List
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect, transaction
from app.electronics.match.matcher import decide
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import fill_from_context, parse_title, variant_key
logger = logging.getLogger(__name__)
_ORDER = {"brand_official": 0, "scraped_page": 1, "search_snippet": 2}
def _listing_from_row(row: dict, category: str) -> Listing:
parsed = parse_title(row["title"], category, expected_brand=row["brand_slug"])
snippet = ""
if row["source_type"] == "search_snippet" and " — " in row["evidence_text"]:
snippet = row["evidence_text"].split(" — ", 1)[1].split(" || ", 1)[0]
raw = row["specs_raw"] or {}
spec_texts = tuple(str(v) for k, v in raw.items() if "processor" in k.lower() or "cpu" in k.lower())
spec_texts += (str((row["specs"] or {}).get("processor") or ""),)
fill_from_context(parsed, category, snippet=snippet, spec_texts=spec_texts)
l = Listing(
site_domain=row["domain"], source_sku=row["source_sku"], source_url=row["source_url"],
source_type=row["source_type"], brand_slug=row["brand_slug"], category=category,
title=row["title"], evidence_text=row["evidence_text"], confidence=float(row["confidence"]),
parser=row["parser"], family=parsed.brand.family if parsed.brand else row["family"],
model=parsed.model, model_number=row["model_number"] or parsed.mpn,
ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb, colour=row["colour"],
gtin=row["gtin"], specs=row["specs"] or {},
)
l.model_norm, l.processor = parsed.model_norm, parsed.processor
l.variant_key = variant_key(parsed, category) if parsed.brand else None
return l
def rematch(category: str) -> Dict[str, int]:
stats: Dict[str, int] = {"listings": 0, "linked": 0, "pending": 0, "unlinked": 0, "products": 0, "images": 0}
with connect() as conn:
rows = conn.execute(
"""
SELECT l.*, b.slug AS brand_slug, s.domain
FROM elec.source_listing l
JOIN elec.brand b ON b.id = l.brand_id
JOIN elec.site s ON s.id = l.site_id
JOIN elec.category c ON c.id = l.category_id
WHERE c.slug = %s
""",
(category,),
).fetchall()
images = conn.execute(
"""
SELECT i.url, i.source_listing_id, i.source_type, i.rank FROM elec.product_image i
JOIN elec.product p ON p.id = i.product_id JOIN elec.category c ON c.id = p.category_id
WHERE c.slug = %s
""",
(category,),
).fetchall()
with transaction() as conn:
# Maps and images cascade from the products.
conn.execute(
"DELETE FROM elec.product p USING elec.category c WHERE c.id = p.category_id AND c.slug = %s",
(category,),
)
ids = repo.id_maps()
product_of_listing: Dict[int, int] = {}
rows.sort(key=lambda r: (_ORDER.get(r["source_type"], 9), r["id"]))
for row in rows:
stats["listings"] += 1
listing = _listing_from_row(row, category)
decision = decide(listing, repo.product_candidates(listing.brand_slug, category))
if decision is None:
stats["unlinked"] += 1
continue
product_id = decision.product_id or repo.create_product(listing, ids)
stats["products"] += decision.product_id is None
repo.map_listing(row["id"], product_id, decision.method, decision.confidence, decision.review_status)
product_of_listing[row["id"]] = product_id
if decision.review_status == "pending":
stats["pending"] += 1
else:
stats["linked"] += 1
repo.merge_product_specs(product_id, listing.specs, {}, listing.source_url)
for img in images:
pid = product_of_listing.get(img["source_listing_id"])
if pid is not None:
repo.add_image(pid, img["url"], img["source_listing_id"], img["source_type"], img["rank"])
stats["images"] += 1
stats.update({f"products_{k}": v for k, v in repo.refresh_verification().items()})
return stats