Files
sriram c7e4d59188 Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-01 12:17:42 +05:30

262 lines
12 KiB
Python

"""Command line for the electronics pipeline. Run from backend/:
python -m app.electronics.cli migrate
python -m app.electronics.cli seed-reference
python -m app.electronics.cli probe [--site croma.com] [--all]
python -m app.electronics.cli collect --category mobiles --brand samsung --brand xiaomi --limit 15
python -m app.electronics.cli reviews [--category mobiles]
python -m app.electronics.cli report
python -m app.electronics.cli review [--approve ID | --reject ID]
python -m app.electronics.cli verify-grounding
"""
from __future__ import annotations
import json
import logging
import re
from decimal import Decimal
from typing import List, Optional
import typer
app = typer.Typer(add_completion=False, help="Electronics catalogue: search-first, evidence-backed collection.")
def _setup_logging(verbose: bool) -> None:
logging.basicConfig(level=logging.DEBUG if verbose else logging.INFO,
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
for noisy in ("httpx", "httpcore", "primp", "ddgs", "urllib3", "sentence_transformers"):
logging.getLogger(noisy).setLevel(logging.WARNING)
@app.command()
def migrate() -> None:
"""Create/upgrade the elec schema in the local electronics_catalog database."""
from app.electronics.db.migrate import run_migrations
applied = run_migrations()
typer.echo(f"Applied: {', '.join(applied) if applied else 'nothing (up to date)'}")
@app.command("seed-reference")
def seed_reference() -> None:
"""Load brands, aliases, categories and sites from reference/*.yaml."""
from app.electronics.db import repository as repo
from app.electronics.reference import load_reference
typer.echo(json.dumps(repo.seed_reference(load_reference())))
@app.command()
def probe(site: List[str] = typer.Option([], "--site", help="Domain(s) to probe; default all probe-policy sites"),
include_official: bool = typer.Option(False, "--official", help="Also probe brand official sites"),
verbose: bool = False) -> None:
"""Grade sites A/B/C: may they be scraped, or only searched?"""
_setup_logging(verbose)
from app.electronics.db import repository as repo
from app.electronics.net.polite_client import PoliteClient
from app.electronics.probe.site_probe import probe_site
from app.electronics.reference import load_reference
from app.electronics.search.engine import SearchEngine
ref = load_reference()
if site:
targets = [s for s in ref.sites.values() if s.domain in site]
else:
targets = [s for s in ref.sites.values() if include_official or s.kind != "brand_official"]
run_id = repo.start_run("probe", {"sites": [s.domain for s in targets]})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
engine = SearchEngine(budget=len(targets) * 3)
results = {}
try:
for s in targets:
res = probe_site(s, client, engine)
repo.set_probe_result(s.domain, res["outcome"], res["robots_allowed"], res["evidence"])
results[s.domain] = res["outcome"]
typer.echo(f"{s.name:28} {s.domain:24} {res['outcome']} {res['evidence'].get('reason')}")
finally:
client.close()
repo.finish_run(run_id, "done", {"grades": results})
@app.command()
def collect(category: str = typer.Option(..., help="mobiles | laptops"),
brand: List[str] = typer.Option([], "--brand", help="Brand slug(s); default all brands of the category"),
limit: int = typer.Option(15, help="Max models per brand"),
expand: int = typer.Option(8, help="Models per brand looked up on other platforms"),
budget: int = typer.Option(200, help="Max search queries this run"),
no_fetch: bool = typer.Option(False, "--no-fetch", help="Search results only; fetch no pages"),
no_llm: bool = typer.Option(False, "--no-llm", help="Deterministic spec parsing only"),
no_embed: bool = typer.Option(False, "--no-embed"),
reprobe: bool = False,
verbose: bool = False) -> None:
"""Discover and collect real listings for allow-listed brands."""
_setup_logging(verbose)
from app.electronics.collector import Collector, RunOptions
from app.electronics.reference import load_reference
ref = load_reference()
if category not in ref.categories:
raise typer.BadParameter(f"unknown category {category!r}; use one of {list(ref.categories)}")
brands = brand or [b.slug for b in ref.brands_for(category)]
unknown = [b for b in brands if b not in ref.brands or category not in ref.brands[b].categories]
if unknown:
raise typer.BadParameter(f"not allow-listed for {category}: {unknown}")
opts = RunOptions(category=category, brands=brands, max_products_per_brand=limit, expand_per_brand=expand,
search_budget=budget, use_llm=not no_llm, fetch_pages=not no_fetch, reprobe=reprobe)
stats = Collector(opts, progress=typer.echo).run(embed=not no_embed)
typer.echo(json.dumps(stats, indent=2, sort_keys=True))
@app.command()
def prices(limit: int = typer.Option(40, help="Max Google queries (free tier: 100/day)"),
category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)")) -> None:
"""Fill missing prices on search-only platforms from Google's structured data (no site fetches)."""
_setup_logging(False)
from app.electronics.price_lookup import lookup_prices
stats = lookup_prices(limit=limit, category=category, progress=typer.echo)
typer.echo(json.dumps(stats, indent=2, default=str))
if stats.get("error"):
typer.echo("\nGoogle search is not usable yet: " + str(stats["error"]))
raise typer.Exit(code=1)
@app.command()
def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
no_embed: bool = typer.Option(False, "--no-embed")) -> None:
"""Rebuild products from stored listings with the current matching rules (no network)."""
_setup_logging(False)
from app.electronics.match.rematch import rematch as run_rematch
stats = run_rematch(category)
if not no_embed:
from app.electronics.collector import embed_verified_products
try:
stats["embedded"] = embed_verified_products()
except Exception as exc: # noqa: BLE001
typer.echo(f"Embedding skipped: {exc}")
typer.echo(json.dumps(stats, indent=2))
@app.command()
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
limit: int = typer.Option(200, help="Max product pages to re-read"),
verbose: bool = False) -> None:
"""Re-read ratings and customer reviews from the product pages already on file.
Only pages the collector itself reads (scraped / brand official listings)
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
rating or review is stored only when the page's own schema.org data states
it; nothing is generated.
"""
_setup_logging(verbose)
from rapidfuzz import fuzz
from app.electronics.db import repository as repo
from app.electronics.extract.jsonld import extract_products
from app.electronics.net.polite_client import PoliteClient
rows = repo.listings_for_review_backfill(category)[:limit]
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
status, error = "done", None
try:
for row in rows:
stats["pages"] += 1
res = client.get(row["source_url"])
if not res.ok:
continue
stats["pages_ok"] += 1
products = extract_products(res.text)
# The same product the listing was stored from: its SKU, else its name.
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
if match is None:
title = (row["title"] or "").lower()
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
scored = [sp for sp in scored if sp[0] >= 85]
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
if match is None:
stats["no_matching_product"] += 1
continue
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
stats["rated"] += 1
if match.get("reviews"):
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
except Exception as exc: # noqa: BLE001
status, error = "failed", repr(exc)
raise
finally:
client.close()
repo.finish_run(run_id, status, stats, error)
typer.echo(json.dumps(stats, indent=2))
@app.command()
def report() -> None:
"""Counts per brand/category and per site."""
from app.electronics.db.connection import connect
with connect() as conn:
typer.echo("Sites:")
for r in conn.execute("SELECT name, domain, policy, probe_outcome, breaker_until FROM elec.site "
"WHERE kind <> 'brand_official' ORDER BY name"):
typer.echo(f" {r['name']:22} {r['policy']:9} grade={r['probe_outcome'] or '-'}"
f"{' breaker until ' + str(r['breaker_until']) if r['breaker_until'] else ''}")
typer.echo("\nProducts by status:")
for r in conn.execute("SELECT verification_status, count(*) n FROM elec.product GROUP BY 1"):
typer.echo(f" {r['verification_status']:12} {r['n']}")
typer.echo("\nVerified catalogue (brand / category):")
for r in conn.execute("SELECT * FROM elec.v_brand_summary ORDER BY category, brand"):
typer.echo(f" {r['brand']:10} {r['category']:8} products={r['product_count']:3} "
f"price ₹{r['min_price']}–₹{r['max_price']} max_platforms={r['max_platforms']}")
typer.echo("\nListings by site and source type:")
for r in conn.execute("SELECT s.name, l.source_type, count(*) n, count(l.price) priced "
"FROM elec.source_listing l JOIN elec.site s ON s.id = l.site_id "
"GROUP BY 1, 2 ORDER BY 1, 2"):
typer.echo(f" {r['name']:22} {r['source_type']:15} {r['n']:4} (with price: {r['priced']})")
@app.command()
def review(approve: Optional[int] = typer.Option(None, help="listing id to approve"),
reject: Optional[int] = typer.Option(None, help="listing id to reject")) -> None:
"""Show uncertain listing-to-product matches, or approve/reject one."""
from app.electronics.db import repository as repo
if approve or reject:
ok = repo.set_review(approve or reject, approve is not None)
refreshed = repo.refresh_verification()
typer.echo(f"{'updated' if ok else 'nothing pending for that listing'}; products: {refreshed}")
return
for r in repo.review_queue():
typer.echo(f"[{r['listing_id']}] {r['site']}: {r['listing_title']}\n -> {r['product']} "
f"({r['method']}, {r['confidence']}) {r['source_url']}")
@app.command("verify-grounding")
def verify_grounding(sample: int = 100) -> None:
"""Audit: every stored price must appear in the evidence text stored with it."""
from app.electronics.db import repository as repo
bad = 0
rows = repo.grounding_sample(sample)
for r in rows:
digits = re.sub(r"\D", "", r["evidence_text"].replace(".00", ""))
price = r["price"]
whole = str(int(price)) if price == price.to_integral() else str(price)
if whole.replace(".", "") not in digits:
bad += 1
typer.echo(f"NOT GROUNDED listing {r['id']}: price {price} not in evidence ({r['source_url']})")
typer.echo(f"Checked {len(rows)} priced listings; {bad} without evidence.")
raise typer.Exit(code=1 if bad else 0)
if __name__ == "__main__":
app()