"""Command line for the electronics pipeline. Run from backend/: python -m app.electronics.cli migrate python -m app.electronics.cli seed-reference python -m app.electronics.cli probe [--site croma.com] [--all] python -m app.electronics.cli collect --category mobiles --brand samsung --brand xiaomi --limit 15 python -m app.electronics.cli reviews [--category mobiles] python -m app.electronics.cli report python -m app.electronics.cli review [--approve ID | --reject ID] python -m app.electronics.cli verify-grounding """ from __future__ import annotations import json import logging import re from decimal import Decimal from typing import List, Optional import typer app = typer.Typer(add_completion=False, help="Electronics catalogue: search-first, evidence-backed collection.") def _setup_logging(verbose: bool) -> None: logging.basicConfig(level=logging.DEBUG if verbose else logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") for noisy in ("httpx", "httpcore", "primp", "ddgs", "urllib3", "sentence_transformers"): logging.getLogger(noisy).setLevel(logging.WARNING) @app.command() def migrate() -> None: """Create/upgrade the elec schema in the local electronics_catalog database.""" from app.electronics.db.migrate import run_migrations applied = run_migrations() typer.echo(f"Applied: {', '.join(applied) if applied else 'nothing (up to date)'}") @app.command("seed-reference") def seed_reference() -> None: """Load brands, aliases, categories and sites from reference/*.yaml.""" from app.electronics.db import repository as repo from app.electronics.reference import load_reference typer.echo(json.dumps(repo.seed_reference(load_reference()))) @app.command() def probe(site: List[str] = typer.Option([], "--site", help="Domain(s) to probe; default all probe-policy sites"), include_official: bool = typer.Option(False, "--official", help="Also probe brand official sites"), verbose: bool = False) -> None: """Grade sites A/B/C: may they be scraped, or only searched?""" _setup_logging(verbose) from app.electronics.db import repository as repo from app.electronics.net.polite_client import PoliteClient from app.electronics.probe.site_probe import probe_site from app.electronics.reference import load_reference from app.electronics.search.engine import SearchEngine ref = load_reference() if site: targets = [s for s in ref.sites.values() if s.domain in site] else: targets = [s for s in ref.sites.values() if include_official or s.kind != "brand_official"] run_id = repo.start_run("probe", {"sites": [s.domain for s in targets]}) client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes, r.outcome, r.robots_allowed)) engine = SearchEngine(budget=len(targets) * 3) results = {} try: for s in targets: res = probe_site(s, client, engine) repo.set_probe_result(s.domain, res["outcome"], res["robots_allowed"], res["evidence"]) results[s.domain] = res["outcome"] typer.echo(f"{s.name:28} {s.domain:24} {res['outcome']} {res['evidence'].get('reason')}") finally: client.close() repo.finish_run(run_id, "done", {"grades": results}) @app.command() def collect(category: str = typer.Option(..., help="mobiles | laptops"), brand: List[str] = typer.Option([], "--brand", help="Brand slug(s); default all brands of the category"), limit: int = typer.Option(15, help="Max models per brand"), expand: int = typer.Option(8, help="Models per brand looked up on other platforms"), budget: int = typer.Option(200, help="Max search queries this run"), no_fetch: bool = typer.Option(False, "--no-fetch", help="Search results only; fetch no pages"), no_llm: bool = typer.Option(False, "--no-llm", help="Deterministic spec parsing only"), no_embed: bool = typer.Option(False, "--no-embed"), reprobe: bool = False, verbose: bool = False) -> None: """Discover and collect real listings for allow-listed brands.""" _setup_logging(verbose) from app.electronics.collector import Collector, RunOptions from app.electronics.reference import load_reference ref = load_reference() if category not in ref.categories: raise typer.BadParameter(f"unknown category {category!r}; use one of {list(ref.categories)}") brands = brand or [b.slug for b in ref.brands_for(category)] unknown = [b for b in brands if b not in ref.brands or category not in ref.brands[b].categories] if unknown: raise typer.BadParameter(f"not allow-listed for {category}: {unknown}") opts = RunOptions(category=category, brands=brands, max_products_per_brand=limit, expand_per_brand=expand, search_budget=budget, use_llm=not no_llm, fetch_pages=not no_fetch, reprobe=reprobe) stats = Collector(opts, progress=typer.echo).run(embed=not no_embed) typer.echo(json.dumps(stats, indent=2, sort_keys=True)) @app.command() def prices(limit: int = typer.Option(40, help="Max Google queries (free tier: 100/day)"), category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)")) -> None: """Fill missing prices on search-only platforms from Google's structured data (no site fetches).""" _setup_logging(False) from app.electronics.price_lookup import lookup_prices stats = lookup_prices(limit=limit, category=category, progress=typer.echo) typer.echo(json.dumps(stats, indent=2, default=str)) if stats.get("error"): typer.echo("\nGoogle search is not usable yet: " + str(stats["error"])) raise typer.Exit(code=1) @app.command() def rematch(category: str = typer.Option(..., help="mobiles | laptops"), no_embed: bool = typer.Option(False, "--no-embed")) -> None: """Rebuild products from stored listings with the current matching rules (no network).""" _setup_logging(False) from app.electronics.match.rematch import rematch as run_rematch stats = run_rematch(category) if not no_embed: from app.electronics.collector import embed_verified_products try: stats["embedded"] = embed_verified_products() except Exception as exc: # noqa: BLE001 typer.echo(f"Embedding skipped: {exc}") typer.echo(json.dumps(stats, indent=2)) @app.command() def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"), limit: int = typer.Option(500, help="Max product pages to re-read"), budget: int = typer.Option(60, help="Max web searches for finding brand-store pages"), no_discover: bool = typer.Option(False, "--no-discover", help="Skip finding brand-store pages"), verbose: bool = False) -> None: """Refresh real customer ratings and reviews from free, robots-allowed sources. Finds each verified product's page on brand stores that publish reviews (brands.yaml `reviews_site`), then re-reads every readable product page for its rating, star breakdown and reviews, plus Vijay Sales' public review feed. A rating or review is stored only when its source states it. """ _setup_logging(verbose) from app.electronics.review_refresh import refresh_reviews stats = refresh_reviews(category=category, limit=limit, budget=budget, discover=not no_discover, progress=typer.echo) typer.echo(json.dumps(stats, indent=2)) @app.command() def report() -> None: """Counts per brand/category and per site.""" from app.electronics.db.connection import connect with connect() as conn: typer.echo("Sites:") for r in conn.execute("SELECT name, domain, policy, probe_outcome, breaker_until FROM elec.site " "WHERE kind <> 'brand_official' ORDER BY name"): typer.echo(f" {r['name']:22} {r['policy']:9} grade={r['probe_outcome'] or '-'}" f"{' breaker until ' + str(r['breaker_until']) if r['breaker_until'] else ''}") typer.echo("\nProducts by status:") for r in conn.execute("SELECT verification_status, count(*) n FROM elec.product GROUP BY 1"): typer.echo(f" {r['verification_status']:12} {r['n']}") typer.echo("\nVerified catalogue (brand / category):") for r in conn.execute("SELECT * FROM elec.v_brand_summary ORDER BY category, brand"): typer.echo(f" {r['brand']:10} {r['category']:8} products={r['product_count']:3} " f"price ₹{r['min_price']}–₹{r['max_price']} max_platforms={r['max_platforms']}") typer.echo("\nListings by site and source type:") for r in conn.execute("SELECT s.name, l.source_type, count(*) n, count(l.price) priced " "FROM elec.source_listing l JOIN elec.site s ON s.id = l.site_id " "GROUP BY 1, 2 ORDER BY 1, 2"): typer.echo(f" {r['name']:22} {r['source_type']:15} {r['n']:4} (with price: {r['priced']})") @app.command() def review(approve: Optional[int] = typer.Option(None, help="listing id to approve"), reject: Optional[int] = typer.Option(None, help="listing id to reject")) -> None: """Show uncertain listing-to-product matches, or approve/reject one.""" from app.electronics.db import repository as repo if approve or reject: ok = repo.set_review(approve or reject, approve is not None) refreshed = repo.refresh_verification() typer.echo(f"{'updated' if ok else 'nothing pending for that listing'}; products: {refreshed}") return for r in repo.review_queue(): typer.echo(f"[{r['listing_id']}] {r['site']}: {r['listing_title']}\n -> {r['product']} " f"({r['method']}, {r['confidence']}) {r['source_url']}") @app.command("verify-grounding") def verify_grounding(sample: int = 100) -> None: """Audit: every stored price must appear in the evidence text stored with it.""" from app.electronics.db import repository as repo bad = 0 rows = repo.grounding_sample(sample) for r in rows: digits = re.sub(r"\D", "", r["evidence_text"].replace(".00", "")) price = r["price"] whole = str(int(price)) if price == price.to_integral() else str(price) if whole.replace(".", "") not in digits: bad += 1 typer.echo(f"NOT GROUNDED listing {r['id']}: price {price} not in evidence ({r['source_url']})") typer.echo(f"Checked {len(rows)} priced listings; {bad} without evidence.") raise typer.Exit(code=1 if bad else 0) if __name__ == "__main__": app()