Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

@@ -0,0 +1,99 @@
"""Decide, per site, whether it may be scraped or only searched.
A robots.txt allows product pages, HTTP 200 without a bot check, and the
page carries a schema.org Product with an INR offer -> scrape
B fetchable, product name/specs readable from the HTML, but no
structured price -> scrape specs/images,
price from search
C serp_only policy, robots.txt disallows, blocked / CAPTCHA, or the page
has no product data without JavaScript -> web search only
The probe looks at 2-3 real product URLs for the site, found through web
search, so it grades the pages the collector would actually fetch.
"""
from __future__ import annotations
import logging
from typing import Dict, List, Optional
from app.electronics.extract.html_fallback import embedded_state, extract_page
from app.electronics.extract.jsonld import extract_products
from app.electronics.net.polite_client import PoliteClient
from app.electronics.reference import SiteRef, load_reference, site_for_url
from app.electronics.search.engine import SearchEngine
logger = logging.getLogger(__name__)
def sample_product_urls(site: SiteRef, engine: SearchEngine, limit: int = 3) -> List[str]:
ref = load_reference()
urls: List[str] = []
if site.kind == "brand_official":
brand = ref.brands[site.brand_slug]
terms = [ref.categories[c].search_terms[0] for c in brand.categories]
queries = [f"site:{site.domain} {brand.name} {t}" for t in terms]
else:
queries = [f"site:{site.domain} samsung galaxy 5g", f"site:{site.domain} lenovo laptop"]
rx = site.product_url_re
for q in queries:
for hit in engine.text(q, max_results=15) or []:
s = site_for_url(hit.url)
if not s or s.domain != site.domain:
continue
if rx is not None and not rx.search(hit.url):
continue
if hit.url not in urls:
urls.append(hit.url)
if len(urls) >= limit:
return urls
return urls
def grade_page(html: str) -> Dict[str, object]:
products = extract_products(html)
priced = [p for p in products if p.get("price") is not None and (p.get("currency") in (None, "INR"))]
page = extract_page(html)
return {
"jsonld_products": len(products),
"jsonld_priced": len(priced),
"meta_price": page.get("price") is not None,
"has_title": bool(page.get("name")),
"spec_rows": len(page.get("properties") or {}),
"embedded_state": embedded_state(html) is not None,
}
def probe_site(site: SiteRef, client: PoliteClient, engine: SearchEngine) -> Dict[str, object]:
"""Returns {"outcome", "robots_allowed", "evidence"}; never raises."""
if site.policy == "serp_only":
return {"outcome": "C", "robots_allowed": None,
"evidence": {"reason": "policy serp_only: this site is never fetched directly"}}
urls = sample_product_urls(site, engine)
if not urls:
return {"outcome": "C", "robots_allowed": None,
"evidence": {"reason": "no product URLs found through web search"}}
pages: List[dict] = []
robots_any: Optional[bool] = None
for url in urls:
res = client.get(url)
entry = {"url": url, "status": res.status, "outcome": res.outcome}
robots_any = res.robots_allowed if robots_any is None else (robots_any or bool(res.robots_allowed))
if res.ok:
entry.update(grade_page(res.text))
pages.append(entry)
if res.outcome in ("captcha", "blocked", "breaker_open"):
break
ok_pages = [p for p in pages if p["outcome"] == "ok"]
if any(p["outcome"] in ("captcha", "blocked") for p in pages):
outcome, reason = "C", "blocked or bot check - not fetched again until the breaker cools down"
elif all(p["outcome"] == "robots_disallowed" for p in pages):
outcome, reason = "C", "robots.txt disallows product pages"
elif not ok_pages:
outcome, reason = "C", "product pages could not be fetched"
elif any(p.get("jsonld_priced") for p in ok_pages):
outcome, reason = "A", "schema.org Product with an INR offer"
elif any(p.get("has_title") and (p.get("spec_rows") or p.get("meta_price") or p.get("jsonld_products")) for p in ok_pages):
outcome, reason = "B", "product details readable from HTML; no structured price"
else:
outcome, reason = "C", "no product data without JavaScript"
return {"outcome": outcome, "robots_allowed": robots_any, "evidence": {"reason": reason, "pages": pages}}