Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
262 lines
12 KiB
Python
262 lines
12 KiB
Python
"""Command line for the electronics pipeline. Run from backend/:
|
|
|
|
python -m app.electronics.cli migrate
|
|
python -m app.electronics.cli seed-reference
|
|
python -m app.electronics.cli probe [--site croma.com] [--all]
|
|
python -m app.electronics.cli collect --category mobiles --brand samsung --brand xiaomi --limit 15
|
|
python -m app.electronics.cli reviews [--category mobiles]
|
|
python -m app.electronics.cli report
|
|
python -m app.electronics.cli review [--approve ID | --reject ID]
|
|
python -m app.electronics.cli verify-grounding
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
import re
|
|
from decimal import Decimal
|
|
from typing import List, Optional
|
|
|
|
import typer
|
|
|
|
app = typer.Typer(add_completion=False, help="Electronics catalogue: search-first, evidence-backed collection.")
|
|
|
|
|
|
def _setup_logging(verbose: bool) -> None:
|
|
logging.basicConfig(level=logging.DEBUG if verbose else logging.INFO,
|
|
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
|
for noisy in ("httpx", "httpcore", "primp", "ddgs", "urllib3", "sentence_transformers"):
|
|
logging.getLogger(noisy).setLevel(logging.WARNING)
|
|
|
|
|
|
@app.command()
|
|
def migrate() -> None:
|
|
"""Create/upgrade the elec schema in the local electronics_catalog database."""
|
|
from app.electronics.db.migrate import run_migrations
|
|
|
|
applied = run_migrations()
|
|
typer.echo(f"Applied: {', '.join(applied) if applied else 'nothing (up to date)'}")
|
|
|
|
|
|
@app.command("seed-reference")
|
|
def seed_reference() -> None:
|
|
"""Load brands, aliases, categories and sites from reference/*.yaml."""
|
|
from app.electronics.db import repository as repo
|
|
from app.electronics.reference import load_reference
|
|
|
|
typer.echo(json.dumps(repo.seed_reference(load_reference())))
|
|
|
|
|
|
@app.command()
|
|
def probe(site: List[str] = typer.Option([], "--site", help="Domain(s) to probe; default all probe-policy sites"),
|
|
include_official: bool = typer.Option(False, "--official", help="Also probe brand official sites"),
|
|
verbose: bool = False) -> None:
|
|
"""Grade sites A/B/C: may they be scraped, or only searched?"""
|
|
_setup_logging(verbose)
|
|
from app.electronics.db import repository as repo
|
|
from app.electronics.net.polite_client import PoliteClient
|
|
from app.electronics.probe.site_probe import probe_site
|
|
from app.electronics.reference import load_reference
|
|
from app.electronics.search.engine import SearchEngine
|
|
|
|
ref = load_reference()
|
|
if site:
|
|
targets = [s for s in ref.sites.values() if s.domain in site]
|
|
else:
|
|
targets = [s for s in ref.sites.values() if include_official or s.kind != "brand_official"]
|
|
run_id = repo.start_run("probe", {"sites": [s.domain for s in targets]})
|
|
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
|
r.outcome, r.robots_allowed))
|
|
engine = SearchEngine(budget=len(targets) * 3)
|
|
results = {}
|
|
try:
|
|
for s in targets:
|
|
res = probe_site(s, client, engine)
|
|
repo.set_probe_result(s.domain, res["outcome"], res["robots_allowed"], res["evidence"])
|
|
results[s.domain] = res["outcome"]
|
|
typer.echo(f"{s.name:28} {s.domain:24} {res['outcome']} {res['evidence'].get('reason')}")
|
|
finally:
|
|
client.close()
|
|
repo.finish_run(run_id, "done", {"grades": results})
|
|
|
|
|
|
@app.command()
|
|
def collect(category: str = typer.Option(..., help="mobiles | laptops"),
|
|
brand: List[str] = typer.Option([], "--brand", help="Brand slug(s); default all brands of the category"),
|
|
limit: int = typer.Option(15, help="Max models per brand"),
|
|
expand: int = typer.Option(8, help="Models per brand looked up on other platforms"),
|
|
budget: int = typer.Option(200, help="Max search queries this run"),
|
|
no_fetch: bool = typer.Option(False, "--no-fetch", help="Search results only; fetch no pages"),
|
|
no_llm: bool = typer.Option(False, "--no-llm", help="Deterministic spec parsing only"),
|
|
no_embed: bool = typer.Option(False, "--no-embed"),
|
|
reprobe: bool = False,
|
|
verbose: bool = False) -> None:
|
|
"""Discover and collect real listings for allow-listed brands."""
|
|
_setup_logging(verbose)
|
|
from app.electronics.collector import Collector, RunOptions
|
|
from app.electronics.reference import load_reference
|
|
|
|
ref = load_reference()
|
|
if category not in ref.categories:
|
|
raise typer.BadParameter(f"unknown category {category!r}; use one of {list(ref.categories)}")
|
|
brands = brand or [b.slug for b in ref.brands_for(category)]
|
|
unknown = [b for b in brands if b not in ref.brands or category not in ref.brands[b].categories]
|
|
if unknown:
|
|
raise typer.BadParameter(f"not allow-listed for {category}: {unknown}")
|
|
opts = RunOptions(category=category, brands=brands, max_products_per_brand=limit, expand_per_brand=expand,
|
|
search_budget=budget, use_llm=not no_llm, fetch_pages=not no_fetch, reprobe=reprobe)
|
|
stats = Collector(opts, progress=typer.echo).run(embed=not no_embed)
|
|
typer.echo(json.dumps(stats, indent=2, sort_keys=True))
|
|
|
|
|
|
@app.command()
|
|
def prices(limit: int = typer.Option(40, help="Max Google queries (free tier: 100/day)"),
|
|
category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)")) -> None:
|
|
"""Fill missing prices on search-only platforms from Google's structured data (no site fetches)."""
|
|
_setup_logging(False)
|
|
from app.electronics.price_lookup import lookup_prices
|
|
|
|
stats = lookup_prices(limit=limit, category=category, progress=typer.echo)
|
|
typer.echo(json.dumps(stats, indent=2, default=str))
|
|
if stats.get("error"):
|
|
typer.echo("\nGoogle search is not usable yet: " + str(stats["error"]))
|
|
raise typer.Exit(code=1)
|
|
|
|
|
|
@app.command()
|
|
def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
|
|
no_embed: bool = typer.Option(False, "--no-embed")) -> None:
|
|
"""Rebuild products from stored listings with the current matching rules (no network)."""
|
|
_setup_logging(False)
|
|
from app.electronics.match.rematch import rematch as run_rematch
|
|
|
|
stats = run_rematch(category)
|
|
if not no_embed:
|
|
from app.electronics.collector import embed_verified_products
|
|
|
|
try:
|
|
stats["embedded"] = embed_verified_products()
|
|
except Exception as exc: # noqa: BLE001
|
|
typer.echo(f"Embedding skipped: {exc}")
|
|
typer.echo(json.dumps(stats, indent=2))
|
|
|
|
|
|
@app.command()
|
|
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
|
|
limit: int = typer.Option(200, help="Max product pages to re-read"),
|
|
verbose: bool = False) -> None:
|
|
"""Re-read ratings and customer reviews from the product pages already on file.
|
|
|
|
Only pages the collector itself reads (scraped / brand official listings)
|
|
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
|
|
rating or review is stored only when the page's own schema.org data states
|
|
it; nothing is generated.
|
|
"""
|
|
_setup_logging(verbose)
|
|
from rapidfuzz import fuzz
|
|
|
|
from app.electronics.db import repository as repo
|
|
from app.electronics.extract.jsonld import extract_products
|
|
from app.electronics.net.polite_client import PoliteClient
|
|
|
|
rows = repo.listings_for_review_backfill(category)[:limit]
|
|
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
|
|
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
|
|
r.outcome, r.robots_allowed))
|
|
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
|
|
status, error = "done", None
|
|
try:
|
|
for row in rows:
|
|
stats["pages"] += 1
|
|
res = client.get(row["source_url"])
|
|
if not res.ok:
|
|
continue
|
|
stats["pages_ok"] += 1
|
|
products = extract_products(res.text)
|
|
# The same product the listing was stored from: its SKU, else its name.
|
|
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
|
|
if match is None:
|
|
title = (row["title"] or "").lower()
|
|
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
|
|
scored = [sp for sp in scored if sp[0] >= 85]
|
|
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
|
|
if match is None:
|
|
stats["no_matching_product"] += 1
|
|
continue
|
|
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
|
|
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
|
|
stats["rated"] += 1
|
|
if match.get("reviews"):
|
|
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
|
|
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
|
|
except Exception as exc: # noqa: BLE001
|
|
status, error = "failed", repr(exc)
|
|
raise
|
|
finally:
|
|
client.close()
|
|
repo.finish_run(run_id, status, stats, error)
|
|
typer.echo(json.dumps(stats, indent=2))
|
|
|
|
|
|
@app.command()
|
|
def report() -> None:
|
|
"""Counts per brand/category and per site."""
|
|
from app.electronics.db.connection import connect
|
|
|
|
with connect() as conn:
|
|
typer.echo("Sites:")
|
|
for r in conn.execute("SELECT name, domain, policy, probe_outcome, breaker_until FROM elec.site "
|
|
"WHERE kind <> 'brand_official' ORDER BY name"):
|
|
typer.echo(f" {r['name']:22} {r['policy']:9} grade={r['probe_outcome'] or '-'}"
|
|
f"{' breaker until ' + str(r['breaker_until']) if r['breaker_until'] else ''}")
|
|
typer.echo("\nProducts by status:")
|
|
for r in conn.execute("SELECT verification_status, count(*) n FROM elec.product GROUP BY 1"):
|
|
typer.echo(f" {r['verification_status']:12} {r['n']}")
|
|
typer.echo("\nVerified catalogue (brand / category):")
|
|
for r in conn.execute("SELECT * FROM elec.v_brand_summary ORDER BY category, brand"):
|
|
typer.echo(f" {r['brand']:10} {r['category']:8} products={r['product_count']:3} "
|
|
f"price ₹{r['min_price']}–₹{r['max_price']} max_platforms={r['max_platforms']}")
|
|
typer.echo("\nListings by site and source type:")
|
|
for r in conn.execute("SELECT s.name, l.source_type, count(*) n, count(l.price) priced "
|
|
"FROM elec.source_listing l JOIN elec.site s ON s.id = l.site_id "
|
|
"GROUP BY 1, 2 ORDER BY 1, 2"):
|
|
typer.echo(f" {r['name']:22} {r['source_type']:15} {r['n']:4} (with price: {r['priced']})")
|
|
|
|
|
|
@app.command()
|
|
def review(approve: Optional[int] = typer.Option(None, help="listing id to approve"),
|
|
reject: Optional[int] = typer.Option(None, help="listing id to reject")) -> None:
|
|
"""Show uncertain listing-to-product matches, or approve/reject one."""
|
|
from app.electronics.db import repository as repo
|
|
|
|
if approve or reject:
|
|
ok = repo.set_review(approve or reject, approve is not None)
|
|
refreshed = repo.refresh_verification()
|
|
typer.echo(f"{'updated' if ok else 'nothing pending for that listing'}; products: {refreshed}")
|
|
return
|
|
for r in repo.review_queue():
|
|
typer.echo(f"[{r['listing_id']}] {r['site']}: {r['listing_title']}\n -> {r['product']} "
|
|
f"({r['method']}, {r['confidence']}) {r['source_url']}")
|
|
|
|
|
|
@app.command("verify-grounding")
|
|
def verify_grounding(sample: int = 100) -> None:
|
|
"""Audit: every stored price must appear in the evidence text stored with it."""
|
|
from app.electronics.db import repository as repo
|
|
|
|
bad = 0
|
|
rows = repo.grounding_sample(sample)
|
|
for r in rows:
|
|
digits = re.sub(r"\D", "", r["evidence_text"].replace(".00", ""))
|
|
price = r["price"]
|
|
whole = str(int(price)) if price == price.to_integral() else str(price)
|
|
if whole.replace(".", "") not in digits:
|
|
bad += 1
|
|
typer.echo(f"NOT GROUNDED listing {r['id']}: price {price} not in evidence ({r['source_url']})")
|
|
typer.echo(f"Checked {len(rows)} priced listings; {bad} without evidence.")
|
|
raise typer.Exit(code=1 if bad else 0)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
app()
|