Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
0
backend/app/electronics/probe/__init__.py
Normal file
0
backend/app/electronics/probe/__init__.py
Normal file
99
backend/app/electronics/probe/site_probe.py
Normal file
99
backend/app/electronics/probe/site_probe.py
Normal file
@@ -0,0 +1,99 @@
|
||||
"""Decide, per site, whether it may be scraped or only searched.
|
||||
|
||||
A robots.txt allows product pages, HTTP 200 without a bot check, and the
|
||||
page carries a schema.org Product with an INR offer -> scrape
|
||||
B fetchable, product name/specs readable from the HTML, but no
|
||||
structured price -> scrape specs/images,
|
||||
price from search
|
||||
C serp_only policy, robots.txt disallows, blocked / CAPTCHA, or the page
|
||||
has no product data without JavaScript -> web search only
|
||||
|
||||
The probe looks at 2-3 real product URLs for the site, found through web
|
||||
search, so it grades the pages the collector would actually fetch.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from app.electronics.extract.html_fallback import embedded_state, extract_page
|
||||
from app.electronics.extract.jsonld import extract_products
|
||||
from app.electronics.net.polite_client import PoliteClient
|
||||
from app.electronics.reference import SiteRef, load_reference, site_for_url
|
||||
from app.electronics.search.engine import SearchEngine
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def sample_product_urls(site: SiteRef, engine: SearchEngine, limit: int = 3) -> List[str]:
|
||||
ref = load_reference()
|
||||
urls: List[str] = []
|
||||
if site.kind == "brand_official":
|
||||
brand = ref.brands[site.brand_slug]
|
||||
terms = [ref.categories[c].search_terms[0] for c in brand.categories]
|
||||
queries = [f"site:{site.domain} {brand.name} {t}" for t in terms]
|
||||
else:
|
||||
queries = [f"site:{site.domain} samsung galaxy 5g", f"site:{site.domain} lenovo laptop"]
|
||||
rx = site.product_url_re
|
||||
for q in queries:
|
||||
for hit in engine.text(q, max_results=15) or []:
|
||||
s = site_for_url(hit.url)
|
||||
if not s or s.domain != site.domain:
|
||||
continue
|
||||
if rx is not None and not rx.search(hit.url):
|
||||
continue
|
||||
if hit.url not in urls:
|
||||
urls.append(hit.url)
|
||||
if len(urls) >= limit:
|
||||
return urls
|
||||
return urls
|
||||
|
||||
|
||||
def grade_page(html: str) -> Dict[str, object]:
|
||||
products = extract_products(html)
|
||||
priced = [p for p in products if p.get("price") is not None and (p.get("currency") in (None, "INR"))]
|
||||
page = extract_page(html)
|
||||
return {
|
||||
"jsonld_products": len(products),
|
||||
"jsonld_priced": len(priced),
|
||||
"meta_price": page.get("price") is not None,
|
||||
"has_title": bool(page.get("name")),
|
||||
"spec_rows": len(page.get("properties") or {}),
|
||||
"embedded_state": embedded_state(html) is not None,
|
||||
}
|
||||
|
||||
|
||||
def probe_site(site: SiteRef, client: PoliteClient, engine: SearchEngine) -> Dict[str, object]:
|
||||
"""Returns {"outcome", "robots_allowed", "evidence"}; never raises."""
|
||||
if site.policy == "serp_only":
|
||||
return {"outcome": "C", "robots_allowed": None,
|
||||
"evidence": {"reason": "policy serp_only: this site is never fetched directly"}}
|
||||
urls = sample_product_urls(site, engine)
|
||||
if not urls:
|
||||
return {"outcome": "C", "robots_allowed": None,
|
||||
"evidence": {"reason": "no product URLs found through web search"}}
|
||||
pages: List[dict] = []
|
||||
robots_any: Optional[bool] = None
|
||||
for url in urls:
|
||||
res = client.get(url)
|
||||
entry = {"url": url, "status": res.status, "outcome": res.outcome}
|
||||
robots_any = res.robots_allowed if robots_any is None else (robots_any or bool(res.robots_allowed))
|
||||
if res.ok:
|
||||
entry.update(grade_page(res.text))
|
||||
pages.append(entry)
|
||||
if res.outcome in ("captcha", "blocked", "breaker_open"):
|
||||
break
|
||||
ok_pages = [p for p in pages if p["outcome"] == "ok"]
|
||||
if any(p["outcome"] in ("captcha", "blocked") for p in pages):
|
||||
outcome, reason = "C", "blocked or bot check - not fetched again until the breaker cools down"
|
||||
elif all(p["outcome"] == "robots_disallowed" for p in pages):
|
||||
outcome, reason = "C", "robots.txt disallows product pages"
|
||||
elif not ok_pages:
|
||||
outcome, reason = "C", "product pages could not be fetched"
|
||||
elif any(p.get("jsonld_priced") for p in ok_pages):
|
||||
outcome, reason = "A", "schema.org Product with an INR offer"
|
||||
elif any(p.get("has_title") and (p.get("spec_rows") or p.get("meta_price") or p.get("jsonld_products")) for p in ok_pages):
|
||||
outcome, reason = "B", "product details readable from HTML; no structured price"
|
||||
else:
|
||||
outcome, reason = "C", "no product data without JavaScript"
|
||||
return {"outcome": outcome, "robots_allowed": robots_any, "evidence": {"reason": reason, "pages": pages}}
|
||||
Reference in New Issue
Block a user