Files
sriram c7e4d59188 Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-01 12:17:42 +05:30

106 lines
4.8 KiB
Python

"""Rebuild canonical products for a category from the listings already stored.
Products and listing links are derived data: every fact lives on the listing
(title, snippet evidence, specs, URL). When the parsing or matching rules
improve, this re-runs them over the stored listings - no network requests -
and keeps each image attached to the listing it was found on.
Review decisions (approved/rejected links) are lost, because the products they
pointed at are rebuilt; uncertain matches simply come back to the queue.
"""
from __future__ import annotations
import logging
from decimal import Decimal
from typing import Dict, List
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect, transaction
from app.electronics.match.matcher import decide
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import fill_from_context, parse_title, variant_key
logger = logging.getLogger(__name__)
_ORDER = {"brand_official": 0, "scraped_page": 1, "search_snippet": 2}
def _listing_from_row(row: dict, category: str) -> Listing:
parsed = parse_title(row["title"], category, expected_brand=row["brand_slug"])
snippet = ""
if row["source_type"] == "search_snippet" and " — " in row["evidence_text"]:
snippet = row["evidence_text"].split(" — ", 1)[1].split(" || ", 1)[0]
raw = row["specs_raw"] or {}
spec_texts = tuple(str(v) for k, v in raw.items() if "processor" in k.lower() or "cpu" in k.lower())
spec_texts += (str((row["specs"] or {}).get("processor") or ""),)
fill_from_context(parsed, category, snippet=snippet, spec_texts=spec_texts)
l = Listing(
site_domain=row["domain"], source_sku=row["source_sku"], source_url=row["source_url"],
source_type=row["source_type"], brand_slug=row["brand_slug"], category=category,
title=row["title"], evidence_text=row["evidence_text"], confidence=float(row["confidence"]),
parser=row["parser"], family=parsed.brand.family if parsed.brand else row["family"],
model=parsed.model, model_number=row["model_number"] or parsed.mpn,
ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb, colour=row["colour"],
gtin=row["gtin"], specs=row["specs"] or {},
)
l.model_norm, l.processor = parsed.model_norm, parsed.processor
l.variant_key = variant_key(parsed, category) if parsed.brand else None
return l
def rematch(category: str) -> Dict[str, int]:
stats: Dict[str, int] = {"listings": 0, "linked": 0, "pending": 0, "unlinked": 0, "products": 0, "images": 0}
with connect() as conn:
rows = conn.execute(
"""
SELECT l.*, b.slug AS brand_slug, s.domain
FROM elec.source_listing l
JOIN elec.brand b ON b.id = l.brand_id
JOIN elec.site s ON s.id = l.site_id
JOIN elec.category c ON c.id = l.category_id
WHERE c.slug = %s
""",
(category,),
).fetchall()
images = conn.execute(
"""
SELECT i.url, i.source_listing_id, i.source_type, i.rank FROM elec.product_image i
JOIN elec.product p ON p.id = i.product_id JOIN elec.category c ON c.id = p.category_id
WHERE c.slug = %s
""",
(category,),
).fetchall()
with transaction() as conn:
# Maps and images cascade from the products.
conn.execute(
"DELETE FROM elec.product p USING elec.category c WHERE c.id = p.category_id AND c.slug = %s",
(category,),
)
ids = repo.id_maps()
product_of_listing: Dict[int, int] = {}
rows.sort(key=lambda r: (_ORDER.get(r["source_type"], 9), r["id"]))
for row in rows:
stats["listings"] += 1
listing = _listing_from_row(row, category)
decision = decide(listing, repo.product_candidates(listing.brand_slug, category))
if decision is None:
stats["unlinked"] += 1
continue
product_id = decision.product_id or repo.create_product(listing, ids)
stats["products"] += decision.product_id is None
repo.map_listing(row["id"], product_id, decision.method, decision.confidence, decision.review_status)
product_of_listing[row["id"]] = product_id
if decision.review_status == "pending":
stats["pending"] += 1
else:
stats["linked"] += 1
repo.merge_product_specs(product_id, listing.specs, {}, listing.source_url)
for img in images:
pid = product_of_listing.get(img["source_listing_id"])
if pid is not None:
repo.add_image(pid, img["url"], img["source_listing_id"], img["source_type"], img["rank"])
stats["images"] += 1
stats.update({f"products_{k}": v for k, v in repo.refresh_verification().items()})
return stats