Files
loyaly-catalogue/backend/app/electronics/cli.py
2026-10-05 11:21:39 +05:30

226 lines
11 KiB
Python

"""Command line for the electronics pipeline. Run from backend/:
python -m app.electronics.cli migrate
python -m app.electronics.cli seed-reference
python -m app.electronics.cli probe [--site croma.com] [--all]
python -m app.electronics.cli collect --category mobiles --brand samsung --brand xiaomi --limit 15
python -m app.electronics.cli reviews [--category mobiles]
python -m app.electronics.cli report
python -m app.electronics.cli review [--approve ID | --reject ID]
python -m app.electronics.cli verify-grounding
"""
from __future__ import annotations
import json
import logging
import re
from decimal import Decimal
from typing import List, Optional
import typer
app = typer.Typer(add_completion=False, help="Electronics catalogue: search-first, evidence-backed collection.")
def _setup_logging(verbose: bool) -> None:
logging.basicConfig(level=logging.DEBUG if verbose else logging.INFO,
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
for noisy in ("httpx", "httpcore", "primp", "ddgs", "urllib3", "sentence_transformers"):
logging.getLogger(noisy).setLevel(logging.WARNING)
@app.command()
def migrate() -> None:
"""Create/upgrade the elec schema in the local electronics_catalog database."""
from app.electronics.db.migrate import run_migrations
applied = run_migrations()
typer.echo(f"Applied: {', '.join(applied) if applied else 'nothing (up to date)'}")
@app.command("seed-reference")
def seed_reference() -> None:
"""Load brands, aliases, categories and sites from reference/*.yaml."""
from app.electronics.db import repository as repo
from app.electronics.reference import load_reference
typer.echo(json.dumps(repo.seed_reference(load_reference())))
@app.command()
def probe(site: List[str] = typer.Option([], "--site", help="Domain(s) to probe; default all probe-policy sites"),
include_official: bool = typer.Option(False, "--official", help="Also probe brand official sites"),
verbose: bool = False) -> None:
"""Grade sites A/B/C: may they be scraped, or only searched?"""
_setup_logging(verbose)
from app.electronics.db import repository as repo
from app.electronics.net.polite_client import PoliteClient
from app.electronics.probe.site_probe import probe_site
from app.electronics.reference import load_reference
from app.electronics.search.engine import SearchEngine
ref = load_reference()
if site:
targets = [s for s in ref.sites.values() if s.domain in site]
else:
targets = [s for s in ref.sites.values() if include_official or s.kind != "brand_official"]
run_id = repo.start_run("probe", {"sites": [s.domain for s in targets]})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
engine = SearchEngine(budget=len(targets) * 3)
results = {}
try:
for s in targets:
res = probe_site(s, client, engine)
repo.set_probe_result(s.domain, res["outcome"], res["robots_allowed"], res["evidence"])
results[s.domain] = res["outcome"]
typer.echo(f"{s.name:28} {s.domain:24} {res['outcome']} {res['evidence'].get('reason')}")
finally:
client.close()
repo.finish_run(run_id, "done", {"grades": results})
@app.command()
def collect(category: str = typer.Option(..., help="mobiles | laptops"),
brand: List[str] = typer.Option([], "--brand", help="Brand slug(s); default all brands of the category"),
limit: int = typer.Option(15, help="Max models per brand"),
expand: int = typer.Option(8, help="Models per brand looked up on other platforms"),
budget: int = typer.Option(200, help="Max search queries this run"),
no_fetch: bool = typer.Option(False, "--no-fetch", help="Search results only; fetch no pages"),
no_llm: bool = typer.Option(False, "--no-llm", help="Deterministic spec parsing only"),
no_embed: bool = typer.Option(False, "--no-embed"),
reprobe: bool = False,
verbose: bool = False) -> None:
"""Discover and collect real listings for allow-listed brands."""
_setup_logging(verbose)
from app.electronics.collector import Collector, RunOptions
from app.electronics.reference import load_reference
ref = load_reference()
if category not in ref.categories:
raise typer.BadParameter(f"unknown category {category!r}; use one of {list(ref.categories)}")
brands = brand or [b.slug for b in ref.brands_for(category)]
unknown = [b for b in brands if b not in ref.brands or category not in ref.brands[b].categories]
if unknown:
raise typer.BadParameter(f"not allow-listed for {category}: {unknown}")
opts = RunOptions(category=category, brands=brands, max_products_per_brand=limit, expand_per_brand=expand,
search_budget=budget, use_llm=not no_llm, fetch_pages=not no_fetch, reprobe=reprobe)
stats = Collector(opts, progress=typer.echo).run(embed=not no_embed)
typer.echo(json.dumps(stats, indent=2, sort_keys=True))
@app.command()
def prices(limit: int = typer.Option(40, help="Max Google queries (free tier: 100/day)"),
category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)")) -> None:
"""Fill missing prices on search-only platforms from Google's structured data (no site fetches)."""
_setup_logging(False)
from app.electronics.price_lookup import lookup_prices
stats = lookup_prices(limit=limit, category=category, progress=typer.echo)
typer.echo(json.dumps(stats, indent=2, default=str))
if stats.get("error"):
typer.echo("\nGoogle search is not usable yet: " + str(stats["error"]))
raise typer.Exit(code=1)
@app.command()
def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
no_embed: bool = typer.Option(False, "--no-embed")) -> None:
"""Rebuild products from stored listings with the current matching rules (no network)."""
_setup_logging(False)
from app.electronics.match.rematch import rematch as run_rematch
stats = run_rematch(category)
if not no_embed:
from app.electronics.collector import embed_verified_products
try:
stats["embedded"] = embed_verified_products()
except Exception as exc: # noqa: BLE001
typer.echo(f"Embedding skipped: {exc}")
typer.echo(json.dumps(stats, indent=2))
@app.command()
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
limit: int = typer.Option(500, help="Max product pages to re-read"),
budget: int = typer.Option(60, help="Max web searches for finding brand-store pages"),
no_discover: bool = typer.Option(False, "--no-discover", help="Skip finding brand-store pages"),
verbose: bool = False) -> None:
"""Refresh real customer ratings and reviews from free, robots-allowed sources.
Finds each verified product's page on brand stores that publish reviews
(brands.yaml `reviews_site`), then re-reads every readable product page for
its rating, star breakdown and reviews, plus Vijay Sales' public review
feed. A rating or review is stored only when its source states it.
"""
_setup_logging(verbose)
from app.electronics.review_refresh import refresh_reviews
stats = refresh_reviews(category=category, limit=limit, budget=budget, discover=not no_discover,
progress=typer.echo)
typer.echo(json.dumps(stats, indent=2))
@app.command()
def report() -> None:
"""Counts per brand/category and per site."""
from app.electronics.db.connection import connect
with connect() as conn:
typer.echo("Sites:")
for r in conn.execute("SELECT name, domain, policy, probe_outcome, breaker_until FROM elec.site "
"WHERE kind <> 'brand_official' ORDER BY name"):
typer.echo(f" {r['name']:22} {r['policy']:9} grade={r['probe_outcome'] or '-'}"
f"{' breaker until ' + str(r['breaker_until']) if r['breaker_until'] else ''}")
typer.echo("\nProducts by status:")
for r in conn.execute("SELECT verification_status, count(*) n FROM elec.product GROUP BY 1"):
typer.echo(f" {r['verification_status']:12} {r['n']}")
typer.echo("\nVerified catalogue (brand / category):")
for r in conn.execute("SELECT * FROM elec.v_brand_summary ORDER BY category, brand"):
typer.echo(f" {r['brand']:10} {r['category']:8} products={r['product_count']:3} "
f"price ₹{r['min_price']}–₹{r['max_price']} max_platforms={r['max_platforms']}")
typer.echo("\nListings by site and source type:")
for r in conn.execute("SELECT s.name, l.source_type, count(*) n, count(l.price) priced "
"FROM elec.source_listing l JOIN elec.site s ON s.id = l.site_id "
"GROUP BY 1, 2 ORDER BY 1, 2"):
typer.echo(f" {r['name']:22} {r['source_type']:15} {r['n']:4} (with price: {r['priced']})")
@app.command()
def review(approve: Optional[int] = typer.Option(None, help="listing id to approve"),
reject: Optional[int] = typer.Option(None, help="listing id to reject")) -> None:
"""Show uncertain listing-to-product matches, or approve/reject one."""
from app.electronics.db import repository as repo
if approve or reject:
ok = repo.set_review(approve or reject, approve is not None)
refreshed = repo.refresh_verification()
typer.echo(f"{'updated' if ok else 'nothing pending for that listing'}; products: {refreshed}")
return
for r in repo.review_queue():
typer.echo(f"[{r['listing_id']}] {r['site']}: {r['listing_title']}\n -> {r['product']} "
f"({r['method']}, {r['confidence']}) {r['source_url']}")
@app.command("verify-grounding")
def verify_grounding(sample: int = 100) -> None:
"""Audit: every stored price must appear in the evidence text stored with it."""
from app.electronics.db import repository as repo
bad = 0
rows = repo.grounding_sample(sample)
for r in rows:
digits = re.sub(r"\D", "", r["evidence_text"].replace(".00", ""))
price = r["price"]
whole = str(int(price)) if price == price.to_integral() else str(price)
if whole.replace(".", "") not in digits:
bad += 1
typer.echo(f"NOT GROUNDED listing {r['id']}: price {price} not in evidence ({r['source_url']})")
typer.echo(f"Checked {len(rows)} priced listings; {bad} without evidence.")
raise typer.Exit(code=1 if bad else 0)
if __name__ == "__main__":
app()