Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
261
backend/app/electronics/cli.py
Normal file
261
backend/app/electronics/cli.py
Normal file
@@ -0,0 +1,261 @@
|
||||
"""Command line for the electronics pipeline. Run from backend/:
|
||||
|
||||
python -m app.electronics.cli migrate
|
||||
python -m app.electronics.cli seed-reference
|
||||
python -m app.electronics.cli probe [--site croma.com] [--all]
|
||||
python -m app.electronics.cli collect --category mobiles --brand samsung --brand xiaomi --limit 15
|
||||
python -m app.electronics.cli reviews [--category mobiles]
|
||||
python -m app.electronics.cli report
|
||||
python -m app.electronics.cli review [--approve ID | --reject ID]
|
||||
python -m app.electronics.cli verify-grounding
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from decimal import Decimal
|
||||
from typing import List, Optional
|
||||
|
||||
import typer
|
||||
|
||||
app = typer.Typer(add_completion=False, help="Electronics catalogue: search-first, evidence-backed collection.")
|
||||
|
||||
|
||||
def _setup_logging(verbose: bool) -> None:
|
||||
logging.basicConfig(level=logging.DEBUG if verbose else logging.INFO,
|
||||
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
||||
for noisy in ("httpx", "httpcore", "primp", "ddgs", "urllib3", "sentence_transformers"):
|
||||
logging.getLogger(noisy).setLevel(logging.WARNING)
|
||||
|
||||
|
||||
@app.command()
|
||||
def migrate() -> None:
|
||||
"""Create/upgrade the elec schema in the local electronics_catalog database."""
|
||||
from app.electronics.db.migrate import run_migrations
|
||||
|
||||
applied = run_migrations()
|
||||
typer.echo(f"Applied: {', '.join(applied) if applied else 'nothing (up to date)'}")
|
||||
|
||||
|
||||
@app.command("seed-reference")
|
||||
def seed_reference() -> None:
|
||||
"""Load brands, aliases, categories and sites from reference/*.yaml."""
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
typer.echo(json.dumps(repo.seed_reference(load_reference())))
|
||||
|
||||
|
||||
@app.command()
|
||||
def probe(site: List[str] = typer.Option([], "--site", help="Domain(s) to probe; default all probe-policy sites"),
|
||||
include_official: bool = typer.Option(False, "--official", help="Also probe brand official sites"),
|
||||
verbose: bool = False) -> None:
|
||||
"""Grade sites A/B/C: may they be scraped, or only searched?"""
|
||||
_setup_logging(verbose)
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
from app.electronics.probe.site_probe import probe_site
|
||||
from app.electronics.reference import load_reference
|
||||
from app.electronics.search.engine import SearchEngine
|
||||
|
||||
ref = load_reference()
|
||||
if site:
|
||||
targets = [s for s in ref.sites.values() if s.domain in site]
|
||||
else:
|
||||
targets = [s for s in ref.sites.values() if include_official or s.kind != "brand_official"]
|
||||
run_id = repo.start_run("probe", {"sites": [s.domain for s in targets]})
|
||||
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
||||
r.outcome, r.robots_allowed))
|
||||
engine = SearchEngine(budget=len(targets) * 3)
|
||||
results = {}
|
||||
try:
|
||||
for s in targets:
|
||||
res = probe_site(s, client, engine)
|
||||
repo.set_probe_result(s.domain, res["outcome"], res["robots_allowed"], res["evidence"])
|
||||
results[s.domain] = res["outcome"]
|
||||
typer.echo(f"{s.name:28} {s.domain:24} {res['outcome']} {res['evidence'].get('reason')}")
|
||||
finally:
|
||||
client.close()
|
||||
repo.finish_run(run_id, "done", {"grades": results})
|
||||
|
||||
|
||||
@app.command()
|
||||
def collect(category: str = typer.Option(..., help="mobiles | laptops"),
|
||||
brand: List[str] = typer.Option([], "--brand", help="Brand slug(s); default all brands of the category"),
|
||||
limit: int = typer.Option(15, help="Max models per brand"),
|
||||
expand: int = typer.Option(8, help="Models per brand looked up on other platforms"),
|
||||
budget: int = typer.Option(200, help="Max search queries this run"),
|
||||
no_fetch: bool = typer.Option(False, "--no-fetch", help="Search results only; fetch no pages"),
|
||||
no_llm: bool = typer.Option(False, "--no-llm", help="Deterministic spec parsing only"),
|
||||
no_embed: bool = typer.Option(False, "--no-embed"),
|
||||
reprobe: bool = False,
|
||||
verbose: bool = False) -> None:
|
||||
"""Discover and collect real listings for allow-listed brands."""
|
||||
_setup_logging(verbose)
|
||||
from app.electronics.collector import Collector, RunOptions
|
||||
from app.electronics.reference import load_reference
|
||||
|
||||
ref = load_reference()
|
||||
if category not in ref.categories:
|
||||
raise typer.BadParameter(f"unknown category {category!r}; use one of {list(ref.categories)}")
|
||||
brands = brand or [b.slug for b in ref.brands_for(category)]
|
||||
unknown = [b for b in brands if b not in ref.brands or category not in ref.brands[b].categories]
|
||||
if unknown:
|
||||
raise typer.BadParameter(f"not allow-listed for {category}: {unknown}")
|
||||
opts = RunOptions(category=category, brands=brands, max_products_per_brand=limit, expand_per_brand=expand,
|
||||
search_budget=budget, use_llm=not no_llm, fetch_pages=not no_fetch, reprobe=reprobe)
|
||||
stats = Collector(opts, progress=typer.echo).run(embed=not no_embed)
|
||||
typer.echo(json.dumps(stats, indent=2, sort_keys=True))
|
||||
|
||||
|
||||
@app.command()
|
||||
def prices(limit: int = typer.Option(40, help="Max Google queries (free tier: 100/day)"),
|
||||
category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)")) -> None:
|
||||
"""Fill missing prices on search-only platforms from Google's structured data (no site fetches)."""
|
||||
_setup_logging(False)
|
||||
from app.electronics.price_lookup import lookup_prices
|
||||
|
||||
stats = lookup_prices(limit=limit, category=category, progress=typer.echo)
|
||||
typer.echo(json.dumps(stats, indent=2, default=str))
|
||||
if stats.get("error"):
|
||||
typer.echo("\nGoogle search is not usable yet: " + str(stats["error"]))
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
|
||||
@app.command()
|
||||
def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
|
||||
no_embed: bool = typer.Option(False, "--no-embed")) -> None:
|
||||
"""Rebuild products from stored listings with the current matching rules (no network)."""
|
||||
_setup_logging(False)
|
||||
from app.electronics.match.rematch import rematch as run_rematch
|
||||
|
||||
stats = run_rematch(category)
|
||||
if not no_embed:
|
||||
from app.electronics.collector import embed_verified_products
|
||||
|
||||
try:
|
||||
stats["embedded"] = embed_verified_products()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
typer.echo(f"Embedding skipped: {exc}")
|
||||
typer.echo(json.dumps(stats, indent=2))
|
||||
|
||||
|
||||
@app.command()
|
||||
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
|
||||
limit: int = typer.Option(200, help="Max product pages to re-read"),
|
||||
verbose: bool = False) -> None:
|
||||
"""Re-read ratings and customer reviews from the product pages already on file.
|
||||
|
||||
Only pages the collector itself reads (scraped / brand official listings)
|
||||
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
|
||||
rating or review is stored only when the page's own schema.org data states
|
||||
it; nothing is generated.
|
||||
"""
|
||||
_setup_logging(verbose)
|
||||
from rapidfuzz import fuzz
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
|
||||
rows = repo.listings_for_review_backfill(category)[:limit]
|
||||
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
|
||||
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
||||
r.outcome, r.robots_allowed))
|
||||
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
|
||||
status, error = "done", None
|
||||
try:
|
||||
for row in rows:
|
||||
stats["pages"] += 1
|
||||
res = client.get(row["source_url"])
|
||||
if not res.ok:
|
||||
continue
|
||||
stats["pages_ok"] += 1
|
||||
products = extract_products(res.text)
|
||||
# The same product the listing was stored from: its SKU, else its name.
|
||||
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
|
||||
if match is None:
|
||||
title = (row["title"] or "").lower()
|
||||
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
|
||||
scored = [sp for sp in scored if sp[0] >= 85]
|
||||
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
|
||||
if match is None:
|
||||
stats["no_matching_product"] += 1
|
||||
continue
|
||||
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
|
||||
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
|
||||
stats["rated"] += 1
|
||||
if match.get("reviews"):
|
||||
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
|
||||
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
|
||||
except Exception as exc: # noqa: BLE001
|
||||
status, error = "failed", repr(exc)
|
||||
raise
|
||||
finally:
|
||||
client.close()
|
||||
repo.finish_run(run_id, status, stats, error)
|
||||
typer.echo(json.dumps(stats, indent=2))
|
||||
|
||||
|
||||
@app.command()
|
||||
def report() -> None:
|
||||
"""Counts per brand/category and per site."""
|
||||
from app.electronics.db.connection import connect
|
||||
|
||||
with connect() as conn:
|
||||
typer.echo("Sites:")
|
||||
for r in conn.execute("SELECT name, domain, policy, probe_outcome, breaker_until FROM elec.site "
|
||||
"WHERE kind <> 'brand_official' ORDER BY name"):
|
||||
typer.echo(f" {r['name']:22} {r['policy']:9} grade={r['probe_outcome'] or '-'}"
|
||||
f"{' breaker until ' + str(r['breaker_until']) if r['breaker_until'] else ''}")
|
||||
typer.echo("\nProducts by status:")
|
||||
for r in conn.execute("SELECT verification_status, count(*) n FROM elec.product GROUP BY 1"):
|
||||
typer.echo(f" {r['verification_status']:12} {r['n']}")
|
||||
typer.echo("\nVerified catalogue (brand / category):")
|
||||
for r in conn.execute("SELECT * FROM elec.v_brand_summary ORDER BY category, brand"):
|
||||
typer.echo(f" {r['brand']:10} {r['category']:8} products={r['product_count']:3} "
|
||||
f"price ₹{r['min_price']}–₹{r['max_price']} max_platforms={r['max_platforms']}")
|
||||
typer.echo("\nListings by site and source type:")
|
||||
for r in conn.execute("SELECT s.name, l.source_type, count(*) n, count(l.price) priced "
|
||||
"FROM elec.source_listing l JOIN elec.site s ON s.id = l.site_id "
|
||||
"GROUP BY 1, 2 ORDER BY 1, 2"):
|
||||
typer.echo(f" {r['name']:22} {r['source_type']:15} {r['n']:4} (with price: {r['priced']})")
|
||||
|
||||
|
||||
@app.command()
|
||||
def review(approve: Optional[int] = typer.Option(None, help="listing id to approve"),
|
||||
reject: Optional[int] = typer.Option(None, help="listing id to reject")) -> None:
|
||||
"""Show uncertain listing-to-product matches, or approve/reject one."""
|
||||
from app.electronics.db import repository as repo
|
||||
|
||||
if approve or reject:
|
||||
ok = repo.set_review(approve or reject, approve is not None)
|
||||
refreshed = repo.refresh_verification()
|
||||
typer.echo(f"{'updated' if ok else 'nothing pending for that listing'}; products: {refreshed}")
|
||||
return
|
||||
for r in repo.review_queue():
|
||||
typer.echo(f"[{r['listing_id']}] {r['site']}: {r['listing_title']}\n -> {r['product']} "
|
||||
f"({r['method']}, {r['confidence']}) {r['source_url']}")
|
||||
|
||||
|
||||
@app.command("verify-grounding")
|
||||
def verify_grounding(sample: int = 100) -> None:
|
||||
"""Audit: every stored price must appear in the evidence text stored with it."""
|
||||
from app.electronics.db import repository as repo
|
||||
|
||||
bad = 0
|
||||
rows = repo.grounding_sample(sample)
|
||||
for r in rows:
|
||||
digits = re.sub(r"\D", "", r["evidence_text"].replace(".00", ""))
|
||||
price = r["price"]
|
||||
whole = str(int(price)) if price == price.to_integral() else str(price)
|
||||
if whole.replace(".", "") not in digits:
|
||||
bad += 1
|
||||
typer.echo(f"NOT GROUNDED listing {r['id']}: price {price} not in evidence ({r['source_url']})")
|
||||
typer.echo(f"Checked {len(rows)} priced listings; {bad} without evidence.")
|
||||
raise typer.Exit(code=1 if bad else 0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app()
|
||||
Reference in New Issue
Block a user