Electronics Catalog: API, MCP server, frontend and deployment

Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
sriram
2026-10-01 12:17:42 +05:30
commit c7e4d59188
115 changed files with 14329 additions and 0 deletions

View File

View File

@@ -0,0 +1,261 @@
"""Command line for the electronics pipeline. Run from backend/:
python -m app.electronics.cli migrate
python -m app.electronics.cli seed-reference
python -m app.electronics.cli probe [--site croma.com] [--all]
python -m app.electronics.cli collect --category mobiles --brand samsung --brand xiaomi --limit 15
python -m app.electronics.cli reviews [--category mobiles]
python -m app.electronics.cli report
python -m app.electronics.cli review [--approve ID | --reject ID]
python -m app.electronics.cli verify-grounding
"""
from __future__ import annotations
import json
import logging
import re
from decimal import Decimal
from typing import List, Optional
import typer
app = typer.Typer(add_completion=False, help="Electronics catalogue: search-first, evidence-backed collection.")
def _setup_logging(verbose: bool) -> None:
logging.basicConfig(level=logging.DEBUG if verbose else logging.INFO,
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
for noisy in ("httpx", "httpcore", "primp", "ddgs", "urllib3", "sentence_transformers"):
logging.getLogger(noisy).setLevel(logging.WARNING)
@app.command()
def migrate() -> None:
"""Create/upgrade the elec schema in the local electronics_catalog database."""
from app.electronics.db.migrate import run_migrations
applied = run_migrations()
typer.echo(f"Applied: {', '.join(applied) if applied else 'nothing (up to date)'}")
@app.command("seed-reference")
def seed_reference() -> None:
"""Load brands, aliases, categories and sites from reference/*.yaml."""
from app.electronics.db import repository as repo
from app.electronics.reference import load_reference
typer.echo(json.dumps(repo.seed_reference(load_reference())))
@app.command()
def probe(site: List[str] = typer.Option([], "--site", help="Domain(s) to probe; default all probe-policy sites"),
include_official: bool = typer.Option(False, "--official", help="Also probe brand official sites"),
verbose: bool = False) -> None:
"""Grade sites A/B/C: may they be scraped, or only searched?"""
_setup_logging(verbose)
from app.electronics.db import repository as repo
from app.electronics.net.polite_client import PoliteClient
from app.electronics.probe.site_probe import probe_site
from app.electronics.reference import load_reference
from app.electronics.search.engine import SearchEngine
ref = load_reference()
if site:
targets = [s for s in ref.sites.values() if s.domain in site]
else:
targets = [s for s in ref.sites.values() if include_official or s.kind != "brand_official"]
run_id = repo.start_run("probe", {"sites": [s.domain for s in targets]})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
engine = SearchEngine(budget=len(targets) * 3)
results = {}
try:
for s in targets:
res = probe_site(s, client, engine)
repo.set_probe_result(s.domain, res["outcome"], res["robots_allowed"], res["evidence"])
results[s.domain] = res["outcome"]
typer.echo(f"{s.name:28} {s.domain:24} {res['outcome']} {res['evidence'].get('reason')}")
finally:
client.close()
repo.finish_run(run_id, "done", {"grades": results})
@app.command()
def collect(category: str = typer.Option(..., help="mobiles | laptops"),
brand: List[str] = typer.Option([], "--brand", help="Brand slug(s); default all brands of the category"),
limit: int = typer.Option(15, help="Max models per brand"),
expand: int = typer.Option(8, help="Models per brand looked up on other platforms"),
budget: int = typer.Option(200, help="Max search queries this run"),
no_fetch: bool = typer.Option(False, "--no-fetch", help="Search results only; fetch no pages"),
no_llm: bool = typer.Option(False, "--no-llm", help="Deterministic spec parsing only"),
no_embed: bool = typer.Option(False, "--no-embed"),
reprobe: bool = False,
verbose: bool = False) -> None:
"""Discover and collect real listings for allow-listed brands."""
_setup_logging(verbose)
from app.electronics.collector import Collector, RunOptions
from app.electronics.reference import load_reference
ref = load_reference()
if category not in ref.categories:
raise typer.BadParameter(f"unknown category {category!r}; use one of {list(ref.categories)}")
brands = brand or [b.slug for b in ref.brands_for(category)]
unknown = [b for b in brands if b not in ref.brands or category not in ref.brands[b].categories]
if unknown:
raise typer.BadParameter(f"not allow-listed for {category}: {unknown}")
opts = RunOptions(category=category, brands=brands, max_products_per_brand=limit, expand_per_brand=expand,
search_budget=budget, use_llm=not no_llm, fetch_pages=not no_fetch, reprobe=reprobe)
stats = Collector(opts, progress=typer.echo).run(embed=not no_embed)
typer.echo(json.dumps(stats, indent=2, sort_keys=True))
@app.command()
def prices(limit: int = typer.Option(40, help="Max Google queries (free tier: 100/day)"),
category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)")) -> None:
"""Fill missing prices on search-only platforms from Google's structured data (no site fetches)."""
_setup_logging(False)
from app.electronics.price_lookup import lookup_prices
stats = lookup_prices(limit=limit, category=category, progress=typer.echo)
typer.echo(json.dumps(stats, indent=2, default=str))
if stats.get("error"):
typer.echo("\nGoogle search is not usable yet: " + str(stats["error"]))
raise typer.Exit(code=1)
@app.command()
def rematch(category: str = typer.Option(..., help="mobiles | laptops"),
no_embed: bool = typer.Option(False, "--no-embed")) -> None:
"""Rebuild products from stored listings with the current matching rules (no network)."""
_setup_logging(False)
from app.electronics.match.rematch import rematch as run_rematch
stats = run_rematch(category)
if not no_embed:
from app.electronics.collector import embed_verified_products
try:
stats["embedded"] = embed_verified_products()
except Exception as exc: # noqa: BLE001
typer.echo(f"Embedding skipped: {exc}")
typer.echo(json.dumps(stats, indent=2))
@app.command()
def reviews(category: Optional[str] = typer.Option(None, help="mobiles | laptops (default both)"),
limit: int = typer.Option(200, help="Max product pages to re-read"),
verbose: bool = False) -> None:
"""Re-read ratings and customer reviews from the product pages already on file.
Only pages the collector itself reads (scraped / brand official listings)
are fetched, politely (robots.txt, per-site pacing, circuit breaker). A
rating or review is stored only when the page's own schema.org data states
it; nothing is generated.
"""
_setup_logging(verbose)
from rapidfuzz import fuzz
from app.electronics.db import repository as repo
from app.electronics.extract.jsonld import extract_products
from app.electronics.net.polite_client import PoliteClient
rows = repo.listings_for_review_backfill(category)[:limit]
run_id = repo.start_run("reviews", {"category": category, "pages": len(rows)})
client = PoliteClient(on_fetch=lambda r, host: repo.log_fetch(run_id, r.url, host, r.status, r.bytes,
r.outcome, r.robots_allowed))
stats = {"pages": 0, "pages_ok": 0, "rated": 0, "reviews_stored": 0, "no_matching_product": 0}
status, error = "done", None
try:
for row in rows:
stats["pages"] += 1
res = client.get(row["source_url"])
if not res.ok:
continue
stats["pages_ok"] += 1
products = extract_products(res.text)
# The same product the listing was stored from: its SKU, else its name.
match = next((p for p in products if p.get("sku") and p["sku"] == row["source_sku"]), None)
if match is None:
title = (row["title"] or "").lower()
scored = [(fuzz.token_set_ratio(p["name"].lower(), title), p) for p in products]
scored = [sp for sp in scored if sp[0] >= 85]
match = max(scored, key=lambda sp: sp[0])[1] if scored else None
if match is None:
stats["no_matching_product"] += 1
continue
if match.get("rating") is not None and Decimal(0) < match["rating"] <= Decimal(5):
repo.update_listing_rating(row["listing_id"], match["rating"], match.get("review_count"))
stats["rated"] += 1
if match.get("reviews"):
stats["reviews_stored"] += repo.save_reviews(row["listing_id"], match["reviews"])
typer.echo(f" {row['domain']:22} rating={match.get('rating')} reviews={len(match.get('reviews') or [])}")
except Exception as exc: # noqa: BLE001
status, error = "failed", repr(exc)
raise
finally:
client.close()
repo.finish_run(run_id, status, stats, error)
typer.echo(json.dumps(stats, indent=2))
@app.command()
def report() -> None:
"""Counts per brand/category and per site."""
from app.electronics.db.connection import connect
with connect() as conn:
typer.echo("Sites:")
for r in conn.execute("SELECT name, domain, policy, probe_outcome, breaker_until FROM elec.site "
"WHERE kind <> 'brand_official' ORDER BY name"):
typer.echo(f" {r['name']:22} {r['policy']:9} grade={r['probe_outcome'] or '-'}"
f"{' breaker until ' + str(r['breaker_until']) if r['breaker_until'] else ''}")
typer.echo("\nProducts by status:")
for r in conn.execute("SELECT verification_status, count(*) n FROM elec.product GROUP BY 1"):
typer.echo(f" {r['verification_status']:12} {r['n']}")
typer.echo("\nVerified catalogue (brand / category):")
for r in conn.execute("SELECT * FROM elec.v_brand_summary ORDER BY category, brand"):
typer.echo(f" {r['brand']:10} {r['category']:8} products={r['product_count']:3} "
f"price ₹{r['min_price']}–₹{r['max_price']} max_platforms={r['max_platforms']}")
typer.echo("\nListings by site and source type:")
for r in conn.execute("SELECT s.name, l.source_type, count(*) n, count(l.price) priced "
"FROM elec.source_listing l JOIN elec.site s ON s.id = l.site_id "
"GROUP BY 1, 2 ORDER BY 1, 2"):
typer.echo(f" {r['name']:22} {r['source_type']:15} {r['n']:4} (with price: {r['priced']})")
@app.command()
def review(approve: Optional[int] = typer.Option(None, help="listing id to approve"),
reject: Optional[int] = typer.Option(None, help="listing id to reject")) -> None:
"""Show uncertain listing-to-product matches, or approve/reject one."""
from app.electronics.db import repository as repo
if approve or reject:
ok = repo.set_review(approve or reject, approve is not None)
refreshed = repo.refresh_verification()
typer.echo(f"{'updated' if ok else 'nothing pending for that listing'}; products: {refreshed}")
return
for r in repo.review_queue():
typer.echo(f"[{r['listing_id']}] {r['site']}: {r['listing_title']}\n -> {r['product']} "
f"({r['method']}, {r['confidence']}) {r['source_url']}")
@app.command("verify-grounding")
def verify_grounding(sample: int = 100) -> None:
"""Audit: every stored price must appear in the evidence text stored with it."""
from app.electronics.db import repository as repo
bad = 0
rows = repo.grounding_sample(sample)
for r in rows:
digits = re.sub(r"\D", "", r["evidence_text"].replace(".00", ""))
price = r["price"]
whole = str(int(price)) if price == price.to_integral() else str(price)
if whole.replace(".", "") not in digits:
bad += 1
typer.echo(f"NOT GROUNDED listing {r['id']}: price {price} not in evidence ({r['source_url']})")
typer.echo(f"Checked {len(rows)} priced listings; {bad} without evidence.")
raise typer.Exit(code=1 if bad else 0)
if __name__ == "__main__":
app()

View File

@@ -0,0 +1,581 @@
"""Search-first collection of real product listings.
For one category and a set of allow-listed brands:
1. DISCOVER web search `site:<platform> <brand> <category term>` on every
registered platform (marketplaces, national chains, Tamil Nadu
chains, the brand's own site). Only URLs that are single product
pages on a registered platform are kept.
2. EXPAND for each model found, search `<brand> <model> price` to find the
same model on other platforms.
3. COLLECT per URL, by the platform's probe grade:
A/B and breaker closed -> fetch the page politely and read
JSON-LD / meta / spec tables
C (or fetch refused) -> use the search result itself: its
title, snippet price and stock text
4. MATCH link the listing to one canonical variant (match.matcher)
5. ENRICH specs (deterministic, LLM gap-fill grounded in page text) and
images (only from the product's own listings, validated live)
6. VERIFY products with listings on ≥2 sites (≥1 a retailer) become
verified and visible.
Nothing in this module invents a product, price or image: every value is read
from a page or a search result, and stored with that URL and text.
"""
from __future__ import annotations
import hashlib
import json
import logging
import re
from dataclasses import dataclass, field
from decimal import Decimal
from typing import Callable, Dict, List, Optional, Tuple
from urllib.parse import urlparse
from rapidfuzz import fuzz
from app.electronics.db import repository as repo
from app.electronics.extract.html_fallback import extract_page, spec_tables, visible_text
from app.electronics.extract.jsonld import extract_products
from app.electronics.extract.serp_parser import clean_result_title, read_price, read_rating, read_stock
from app.electronics.match.matcher import decide
from app.electronics.models import Listing
from app.electronics.net.breaker import CircuitBreaker
from app.electronics.net.polite_client import PoliteClient
from app.electronics.normalise.brand_alias import looks_like_device_title
from app.electronics.normalise.llm_fill import fill_missing
from app.electronics.normalise.spec_normaliser import normalise_specs
from app.electronics.normalise.title_parser import ParsedTitle, fill_from_context, parse_title, variant_key
from app.electronics.probe.site_probe import probe_site
from app.electronics.reference import BrandRef, SiteRef, load_reference, site_for_url
from app.electronics.search.engine import SearchEngine
from app.electronics.search.providers import SearchHit
from app.infrastructure.settings import ELEC_PROBE_TTL_DAYS, MIN_IMAGE_BYTES
from bs4 import BeautifulSoup
logger = logging.getLogger(__name__)
# Titles that belong to another category even when the brand matches.
_OFF_CATEGORY = {
"mobiles": re.compile(r"\b(?:tab|tablet|pad|watch|buds|earbuds|laptop|book|monitor|tv|television|band)\b", re.I),
"laptops": re.compile(r"\b(?:tablet|tab|monitor|mouse|keyboard|phone|smartphone|printer|desktop|all[- ]in[- ]one)\b", re.I),
}
_LISTING_PAGE = re.compile(r"/(?:search|s|c|category|categories|brand|brands|compare|offers?|deals?)(?:/|\?|$)", re.I)
@dataclass
class RunOptions:
category: str
brands: List[str] # brand slugs
max_products_per_brand: int = 15
expand_per_brand: int = 8 # models to look up on other platforms
search_budget: int = 200
use_llm: bool = True
fetch_pages: bool = True
find_images: bool = True
reprobe: bool = False
@dataclass
class RunStats:
counts: Dict[str, int] = field(default_factory=dict)
def inc(self, key: str, n: int = 1) -> None:
self.counts[key] = self.counts.get(key, 0) + n
def source_sku(site: SiteRef, url: str) -> str:
rx = site.product_url_re
if rx is not None:
m = rx.search(url)
if m and m.groups() and m.group(1):
return m.group(1)
p = urlparse(url)
return (p.netloc.lower().removeprefix("www.") + p.path.rstrip("/").lower())[:300]
_INDIA_PATH = re.compile(r"^/(?:in|in-en|en-in|en_in|in_en)(?:/|$)", re.I)
def is_product_url(site: SiteRef, url: str) -> bool:
p = urlparse(url)
if _LISTING_PAGE.search(p.path):
return False
if site.kind == "brand_official":
# Only the brand's Indian storefront: www.samsung.com/in/..., not
# us.samsung.com or news.samsung.com. Domains that are Indian already
# (oneplus.in, motorola.co.in) qualify as they are.
host = (p.hostname or "").lower()
if host not in (site.domain, "www." + site.domain, "in." + site.domain):
return False
if not site.domain.endswith((".in", ".co.in")) and not _INDIA_PATH.search(p.path) and not host.startswith("in."):
return False
rx = site.product_url_re
if rx is not None:
return bool(rx.search(url))
return p.path.count("/") >= 2 # brand sites: at least /section/product
def embed_verified_products() -> int:
"""MiniLM vectors for verified products that do not have one yet."""
rows = repo.products_without_embedding()
if not rows:
return 0
from app.services.embeddings_service import embed_texts
texts = [
f"{r['brand']} {r['display_name']} {r['category']} "
+ " ".join(f"{k} {v}" for k, v in (r["canonical_specs"] or {}).items())
for r in rows
]
for r, vec in zip(rows, embed_texts(texts)):
repo.set_embedding(r["id"], vec)
return len(rows)
class Collector:
def __init__(self, options: RunOptions, *, progress: Optional[Callable[[str], None]] = None) -> None:
self.opt = options
self.ref = load_reference()
self.stats = RunStats()
self.progress = progress or (lambda msg: logger.info(msg))
self.run_id: Optional[int] = None
self.ids = repo.id_maps()
self.site_rows = {r["domain"]: r for r in repo.sites()}
self.breaker = CircuitBreaker(on_trip=self._on_trip)
for r in self.site_rows.values():
if r.get("breaker_until"):
self.breaker.preload(r["domain"], r["breaker_until"].timestamp(), r.get("breaker_reason") or "")
self.client = PoliteClient(breaker=self.breaker, on_fetch=self._on_fetch)
self.engine = SearchEngine(budget=options.search_budget)
self._touched_products: Dict[int, List[Tuple[int, Listing]]] = {}
# -- callbacks -----------------------------------------------------------
def _on_trip(self, host: str, reason: str, until: float) -> None:
self.stats.inc("breaker_trips")
self.progress(f"Circuit breaker opened for {host}: {reason}. Falling back to web search for it.")
repo.trip_breaker(host, reason, until)
def _on_fetch(self, result, host: str) -> None:
self.stats.inc(f"fetch_{result.outcome}")
repo.log_fetch(self.run_id, result.url, host, result.status, result.bytes, result.outcome, result.robots_allowed)
# -- grading -------------------------------------------------------------
def _grade(self, site: SiteRef) -> str:
if site.policy == "serp_only":
return "C"
host = site.domain
if self.breaker.is_open(host) or self.breaker.is_open("www." + host):
return "C"
return (self.site_rows.get(site.domain) or {}).get("probe_outcome") or "C"
def ensure_probes(self, sites: List[SiteRef]) -> None:
from datetime import datetime, timedelta, timezone
stale_before = datetime.now(timezone.utc) - timedelta(days=ELEC_PROBE_TTL_DAYS)
for site in sites:
row = self.site_rows.get(site.domain) or {}
if not self.opt.reprobe and row.get("probed_at") and row["probed_at"] > stale_before:
continue
self.progress(f"Probing {site.name} ({site.domain})")
result = probe_site(site, self.client, self.engine)
repo.set_probe_result(site.domain, result["outcome"], result["robots_allowed"], result["evidence"])
row.update(probe_outcome=result["outcome"])
self.site_rows[site.domain] = row
self.stats.inc(f"probe_{result['outcome']}")
self.progress(f" -> grade {result['outcome']}: {result['evidence'].get('reason')}")
# -- discovery -----------------------------------------------------------
def _accept_hit(self, hit: SearchHit, brand: BrandRef) -> Optional[Tuple[SiteRef, ParsedTitle]]:
site = site_for_url(hit.url)
if site is None or not self._site_allowed_for(site, brand):
return None
if not is_product_url(site, hit.url):
return None
title = clean_result_title(hit.title)
if not looks_like_device_title(title) or _OFF_CATEGORY[self.opt.category].search(title):
return None
parsed = parse_title(title, self.opt.category, expected_brand=brand.slug)
if parsed.brand is None or not parsed.model_norm:
return None
fill_from_context(parsed, self.opt.category, snippet=hit.snippet or "")
return site, parsed
def _site_allowed_for(self, site: SiteRef, brand: BrandRef) -> bool:
if site.kind == "brand_official":
return site.brand_slug == brand.slug
return True
def _platforms_for(self, brand: BrandRef) -> List[SiteRef]:
return [s for s in self.ref.sites.values()
if s.kind != "brand_official" or s.brand_slug == brand.slug]
def _search_names(self, brand: BrandRef) -> List[str]:
"""The brand, plus the sub-brands phones are actually sold under
("Redmi", "POCO", "iQOO") - a search for "Xiaomi" alone misses most
Redmi listings."""
names = [brand.name]
if self.opt.category == "mobiles":
names += [s.upper() if len(s) <= 4 else s.title() for s in brand.sub_brands
if s not in ("mi", "iphone", "pixel", "narzo")][:2]
return names
def _collect_hits(self, hits: Optional[List[SearchHit]], brand: BrandRef, query: str,
found: Dict, models: Dict[str, ParsedTitle]) -> int:
new = 0
for hit in hits or []:
accepted = self._accept_hit(hit, brand)
if not accepted:
continue
site_ref, parsed = accepted
key = (site_ref.domain, source_sku(site_ref, hit.url))
if key not in found:
found[key] = (hit, site_ref, parsed, query)
new += 1
models.setdefault(parsed.model_norm, parsed)
return new
def discover(self, brand: BrandRef) -> Dict[Tuple[str, str], Tuple[SearchHit, SiteRef, ParsedTitle, str]]:
category = self.ref.categories[self.opt.category]
found: Dict[Tuple[str, str], Tuple[SearchHit, SiteRef, ParsedTitle, str]] = {}
models: Dict[str, ParsedTitle] = {}
per_site_target = max(4, self.opt.max_products_per_brand)
for site in self._platforms_for(brand):
got = 0
for name in self._search_names(brand):
for terms in category.query_terms or category.search_terms:
query = f"site:{site.domain} {name} {terms}"
hits = self.engine.text(query, max_results=20)
if hits is None:
self.stats.inc("search_unavailable")
continue
got += self._collect_hits(hits, brand, query, found, models)
if got >= per_site_target:
break
if got >= per_site_target:
break
self.progress(f"{brand.name}: {len(found)} listing URLs, {len(models)} models from platform searches")
# Cross-platform: look each variant up by name, to find the same product
# on platforms the site: searches missed. Phones are grouped by model
# line; laptops by full configuration (line + CPU + RAM + storage),
# because one laptop line is sold in dozens of configurations and only
# the exact one confirms a product.
for p in self._expansion_targets(found):
query = f"{brand.name} {p.model or p.model_norm} {self._variant_terms(p)} price"
self._collect_hits(self.engine.text(re.sub(r"\s+", " ", query), max_results=20), brand, query, found, models)
self.progress(f"{brand.name}: {len(found)} listing URLs after cross-platform search")
return found
def _expansion_targets(self, found: Dict) -> List[ParsedTitle]:
"""Which variants to look up on other platforms, most useful first:
variants seen on the most sites, then ones whose page we can read with
a price (grade A/B platforms), since one more site verifies those."""
groups: Dict[str, Dict] = {}
for (domain, _), (_, site, parsed, _) in found.items():
if self.opt.category == "laptops":
key = variant_key(parsed, "laptops")
if key is None:
continue
else:
key = parsed.model_norm
g = groups.setdefault(key, {"parsed": parsed, "sites": set(), "readable": False})
g["sites"].add(domain)
g["readable"] = g["readable"] or self._grade(site) in ("A", "B")
ranked = sorted(groups.values(), key=lambda g: (-len(g["sites"]), not g["readable"]))
return [g["parsed"] for g in ranked[: self.opt.expand_per_brand]]
@staticmethod
def _variant_terms(p: ParsedTitle) -> str:
parts = []
if p.processor:
# "ryzen 5 7530u" -> "Ryzen 5 7530U", "i5-1334u" -> "i5-1334U"
parts.append(" ".join(t.upper() if any(ch.isdigit() for ch in t) and len(t) > 2 else t.title()
for t in p.processor.split()))
if p.ram_gb:
parts.append(f"{format(p.ram_gb.normalize(), 'f')}GB RAM")
if p.storage_gb:
parts.append(f"{format(p.storage_gb.normalize(), 'f')}GB")
return " ".join(parts)
# -- listing construction --------------------------------------------------
def _base_listing(self, site: SiteRef, url: str, parsed: ParsedTitle, title: str, source_type: str,
evidence: str, confidence: float, parser: str, query: str) -> Listing:
l = Listing(
site_domain=site.domain, source_sku=source_sku(site, url), source_url=url, source_type=source_type,
brand_slug=parsed.brand.brand_slug, category=self.opt.category, title=title,
evidence_text=evidence, confidence=confidence, parser=parser, family=parsed.brand.family,
model=parsed.model, model_number=parsed.mpn, ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb,
colour=parsed.colour, search_query=query,
)
l.model_norm = parsed.model_norm
l.processor = parsed.processor
l.variant_key = variant_key(parsed, self.opt.category)
return l
def listing_from_search(self, hit: SearchHit, site: SiteRef, parsed: ParsedTitle, query: str) -> Listing:
title = clean_result_title(hit.title)
evidence = f"{hit.title} — {hit.snippet}".strip(" —")
reading = read_price(f"{hit.title} {hit.snippet}")
price, mrp = reading.price, reading.mrp
# A snippet that names a different RAM/storage than the title is about
# another variant; its price cannot be trusted for this one.
snippet_variant = parse_title(hit.snippet or "", self.opt.category)
for a, b in ((snippet_variant.storage_gb, parsed.storage_gb), (snippet_variant.ram_gb, parsed.ram_gb)):
if a is not None and b is not None and a != b:
price = mrp = None
self.stats.inc("snippet_price_variant_conflict")
# Truncated titles ("... - (16 GB ...") lose the variant; the snippet
# of the same result usually states it.
fill_from_context(parsed, self.opt.category, snippet=hit.snippet or "")
parser, confidence = f"serp:{hit.provider}", (0.55 if price is not None else 0.45)
in_stock = read_stock(hit.snippet or "")
availability = None if in_stock is None else ("InStock" if in_stock else "OutOfStock")
# Structured offer the search engine read from the page itself (Google
# pagemap). Better than snippet text, and still no request to the site.
if hit.offer and price is None:
try:
offered = Decimal(str(hit.offer["price"]).replace(",", ""))
except Exception: # noqa: BLE001
offered = None
if offered is not None and Decimal(500) <= offered <= Decimal(1000000):
price, mrp = offered, None
evidence = f"{evidence} || search-engine offer data: {json.dumps(hit.offer['raw'], default=str)[:600]}"
parser, confidence = f"serp:{hit.provider}:pagemap", 0.65
av = (hit.offer.get("availability") or "").lower().replace(" ", "")
if "instock" in av:
in_stock, availability = True, "InStock"
elif "outofstock" in av or "soldout" in av:
in_stock, availability = False, "OutOfStock"
# Rating: the engine's structured data first (read from the page
# itself), else an explicit "x out of 5" in this result's own text.
rating, review_count = None, None
if hit.rating and hit.rating.get("rating") is not None:
rating = Decimal(str(hit.rating["rating"]))
review_count = hit.rating.get("review_count")
evidence = f"{evidence} || search-engine rating data: {json.dumps(hit.rating.get('raw'), default=str)[:300]}"
else:
stated = read_rating(f"{hit.title} {hit.snippet}")
if stated.rating is not None:
rating, review_count = stated.rating, stated.review_count
listing = self._base_listing(site, hit.url, parsed, title, "search_snippet", evidence,
confidence, parser, query)
listing.price, listing.mrp = price, mrp
listing.in_stock, listing.availability = in_stock, availability
listing.rating, listing.review_count = rating, review_count
return listing
def listing_from_page(self, hit: SearchHit, site: SiteRef, parsed_hit: ParsedTitle, query: str,
html: str, final_url: str) -> Optional[Listing]:
products = extract_products(html)
page = extract_page(html)
name = None
product = None
for p in products:
pp = parse_title(p["name"], self.opt.category, expected_brand=parsed_hit.brand.brand_slug)
if pp.brand and pp.model_norm and fuzz.token_set_ratio(pp.model_norm, parsed_hit.model_norm) >= 85:
product, name = p, p["name"]
break
if name is None:
name = page.get("name")
if not name:
return None
parsed = parse_title(name, self.opt.category, expected_brand=parsed_hit.brand.brand_slug)
if parsed.brand is None or not parsed.model_norm:
return None
if fuzz.token_set_ratio(parsed.model_norm, parsed_hit.model_norm) < 85:
# The URL did not lead to the product the search result named.
self.stats.inc("page_title_mismatch")
return None
# Fill variant fields the page name leaves out from the search title
# of the same URL (both are statements by the same site).
for attr in ("ram_gb", "storage_gb", "colour", "mpn", "processor"):
if getattr(parsed, attr) is None and getattr(parsed_hit, attr) is not None:
setattr(parsed, attr, getattr(parsed_hit, attr))
grade = self._grade(site)
source_type = "brand_official" if site.kind == "brand_official" else "scraped_page"
if product is not None:
price = product.get("price")
currency = product.get("currency")
evidence = product["evidence"]
parser = "jsonld"
else:
price, currency, evidence, parser = page.get("price"), page.get("currency"), page.get("evidence") or "", "html_meta"
if currency not in (None, "INR"):
price = None
if currency is None and site.kind == "brand_official":
price = None # a brand's global site may not be quoting rupees
if price is not None and not (Decimal(500) <= price <= Decimal(1000000)):
price = None
evidence = evidence or f"{name} ({final_url})"
listing = self._base_listing(
site, final_url if final_url.startswith("http") else hit.url, parsed, name, source_type,
evidence, 0.9 if (price is not None and parser == "jsonld") else 0.75, f"{parser}:grade{grade}", query,
)
# The site's own SKU when the page states it; otherwise its canonical URL.
listing.source_sku = ((product or {}).get("sku") or source_sku(site, final_url or hit.url))[:300]
listing.price = price
listing.in_stock = (product or page).get("in_stock")
listing.availability = (product or page).get("availability")
listing.image_urls = list(dict.fromkeys((product or {}).get("images", []) + page.get("images", [])))[:8]
if product:
listing.gtin = product.get("gtin")
listing.model_number = listing.model_number or product.get("mpn")
listing.rating = product.get("rating")
listing.review_count = product.get("review_count")
listing.reviews = list(product.get("reviews") or [])
listing.colour = listing.colour or product.get("color")
raw_specs = dict((product or {}).get("properties") or {})
raw_specs.update({k: v for k, v in spec_tables(BeautifulSoup(html, "lxml")).items() if k not in raw_specs})
listing.specs_raw = dict(list(raw_specs.items())[:150])
listing.specs, listing.spec_sources = normalise_specs(self.opt.category, raw_specs)
if self.opt.use_llm:
wanted = [k for k in self.ref.spec_keys.get(self.opt.category, {}) if k not in listing.specs]
if wanted:
text = "\n".join(f"{k}: {v}" for k, v in raw_specs.items()) or visible_text(html, 3500)
extra, extra_src = fill_missing(self.opt.category, text, wanted)
listing.specs.update(extra)
listing.spec_sources.update(extra_src)
self.stats.inc("llm_specs_kept", len(extra))
if self.opt.category == "laptops":
# "13th Gen Intel Core i7/ 16GB RAM" in a title names no CPU model;
# the page's own spec table usually does.
spec_texts = tuple(str(v) for k, v in raw_specs.items() if "processor" in k.lower() or "cpu" in k.lower())
spec_texts += (str(listing.specs.get("processor") or ""),)
before = parsed.processor
fill_from_context(parsed, self.opt.category, spec_texts=spec_texts)
if parsed.processor != before:
listing.processor = parsed.processor
listing.variant_key = variant_key(parsed, self.opt.category)
listing.content_hash = hashlib.sha1(html.encode("utf-8", "ignore")).hexdigest()
return listing
# -- persistence -----------------------------------------------------------
def store(self, listing: Listing) -> Optional[int]:
try:
listing_id = repo.upsert_listing(listing, self.ids, self.run_id)
except ValueError as exc:
self.stats.inc("rejected_listing")
logger.info("Listing rejected (%s): %s", exc, listing.source_url)
return None
self.stats.inc(f"listing_{listing.source_type}")
if listing.reviews:
self.stats.inc("reviews_stored", repo.save_reviews(listing_id, listing.reviews))
if listing.price is not None:
self.stats.inc("listing_with_price")
decision = decide(listing, repo.product_candidates(listing.brand_slug, listing.category))
if decision is None:
self.stats.inc("listing_unmatched_no_variant")
return listing_id
product_id = decision.product_id or repo.create_product(listing, self.ids)
if decision.product_id is None:
self.stats.inc("product_created")
repo.map_listing(listing_id, product_id, decision.method, decision.confidence, decision.review_status)
if decision.review_status == "pending":
self.stats.inc("match_pending_review")
else:
repo.merge_product_specs(product_id, listing.specs, listing.spec_sources, listing.source_url)
self._touched_products.setdefault(product_id, []).append((listing_id, listing))
return listing_id
# -- images ----------------------------------------------------------------
def attach_images(self) -> None:
rank_for = {"brand_official": 10, "scraped_page": 20, "search_snippet": 50}
for product_id, entries in self._touched_products.items():
if repo.product_image_count(product_id) >= 3:
continue
added = 0
for listing_id, listing in sorted(entries, key=lambda e: rank_for[e[1].source_type]):
for url in listing.image_urls:
if added >= 3:
break
if self.client.check_image(url, MIN_IMAGE_BYTES):
repo.add_image(product_id, url, listing_id, listing.source_type, rank_for[listing.source_type])
added += 1
self.stats.inc("images_from_pages")
if added or not self.opt.find_images:
continue
added = self._images_from_search(product_id, entries)
self.stats.inc("images_from_search", added)
def _images_from_search(self, product_id: int, entries: List[Tuple[int, Listing]]) -> int:
"""Image search results are used only when the page an image sits on is
one of THIS product's own listings (same site, same product), and the
image result's title names the model."""
_, listing = entries[0]
brand = self.ref.brands[listing.brand_slug]
listing_by_site = {l.site_domain: lid for lid, l in entries}
hits = self.engine.images(f"{brand.name} {listing.model or listing.model_norm}", max_results=15) or []
added = 0
for hit in hits:
site = site_for_url(hit.url)
if site is None or site.domain not in listing_by_site:
continue
parsed = parse_title(clean_result_title(hit.title), listing.category, expected_brand=brand.slug)
if not parsed.model_norm or fuzz.token_set_ratio(parsed.model_norm, listing.model_norm or "") < 90:
continue
if self.client.check_image(hit.image_url, MIN_IMAGE_BYTES):
repo.add_image(product_id, hit.image_url, listing_by_site[site.domain], "search_image", 60)
added += 1
if added >= 2:
break
return added
# -- embeddings ------------------------------------------------------------
def embed(self) -> int:
return embed_verified_products()
# -- the run ---------------------------------------------------------------
def run(self, *, embed: bool = True) -> Dict[str, int]:
self.run_id = repo.start_run("collect", {
"category": self.opt.category, "brands": self.opt.brands,
"max_products_per_brand": self.opt.max_products_per_brand,
"search_budget": self.opt.search_budget,
})
status, error = "done", None
try:
brands = [self.ref.brands[b] for b in self.opt.brands]
probe_targets = {s.domain: s for b in brands for s in self._platforms_for(b) if s.policy == "probe"}
if self.opt.fetch_pages:
self.ensure_probes(list(probe_targets.values()))
for brand in brands:
found = self.discover(brand)
# Keep the most common models first, up to the per-brand limit.
by_model: Dict[str, int] = {}
for (_, _), (_, _, parsed, _) in found.items():
by_model[parsed.model_norm] = by_model.get(parsed.model_norm, 0) + 1
keep = set(sorted(by_model, key=lambda m: -by_model[m])[: self.opt.max_products_per_brand])
for (domain, _sku), (hit, site, parsed, query) in found.items():
if parsed.model_norm not in keep:
continue
listing = None
if self.opt.fetch_pages and self._grade(site) in ("A", "B"):
res = self.client.get(hit.url)
if res.ok:
listing = self.listing_from_page(hit, site, parsed, query, res.text, res.final_url)
if listing is None:
self.stats.inc("page_unusable_fell_back_to_search")
if listing is None:
listing = self.listing_from_search(hit, site, parsed, query)
self.store(listing)
self.progress(f"{brand.name}: stored listings; {self.stats.counts}")
self.attach_images()
verification = repo.refresh_verification()
self.stats.counts.update({f"products_{k}": v for k, v in verification.items()})
if embed:
try:
self.stats.inc("embedded", self.embed())
except Exception as exc: # noqa: BLE001 - embeddings are optional
logger.warning("Embedding step skipped: %s", exc)
self.stats.counts.update({f"search_{k}": v for k, v in self.engine.stats.items()})
except Exception as exc:
status, error = "failed", repr(exc)
logger.exception("Collection run failed")
raise
finally:
repo.finish_run(self.run_id, status, self.stats.counts, error)
self.client.close()
return self.stats.counts

View File

View File

@@ -0,0 +1,68 @@
"""Connections to the local electronics database.
settings.py has already refused to load unless DB_HOST is local and DB_NAME is
electronics_catalog, so nothing here can reach another database.
"""
from __future__ import annotations
import logging
from contextlib import contextmanager
from typing import Iterator
import psycopg
from psycopg.rows import dict_row
from app.infrastructure.settings import (
DB_CONNECT_TIMEOUT_SECONDS,
DB_HOST,
DB_NAME,
DB_PASSWORD,
DB_PORT,
DB_USER,
)
logger = logging.getLogger(__name__)
def connect(*, autocommit: bool = False) -> psycopg.Connection:
conn = psycopg.connect(
host=DB_HOST,
port=DB_PORT,
dbname=DB_NAME,
user=DB_USER,
password=DB_PASSWORD,
connect_timeout=DB_CONNECT_TIMEOUT_SECONDS,
autocommit=autocommit,
row_factory=dict_row,
)
try:
from pgvector.psycopg import register_vector
register_vector(conn)
except Exception: # noqa: BLE001 - the extension is created by migration 0001
pass
return conn
@contextmanager
def transaction() -> Iterator[psycopg.Connection]:
"""A connection whose work is committed on success, rolled back on error."""
conn = connect()
try:
yield conn
conn.commit()
except Exception:
conn.rollback()
raise
finally:
conn.close()
def check_connection() -> bool:
try:
with connect(autocommit=True) as conn:
conn.execute("SELECT 1")
return True
except Exception as exc: # noqa: BLE001 - a health probe reports, never raises
logger.debug("Database unreachable: %s", exc)
return False

View File

@@ -0,0 +1,63 @@
"""Tiny migration runner for the numbered SQL files in ./migrations.
Each file runs once, in its own transaction, and is recorded in
elec.schema_migrations with a checksum. Editing an applied file is refused
rather than silently ignored - add a new numbered file instead.
"""
from __future__ import annotations
import hashlib
import logging
from pathlib import Path
from typing import List
from app.electronics.db.connection import connect
logger = logging.getLogger(__name__)
MIGRATIONS_DIR = Path(__file__).resolve().parent / "migrations"
_BOOTSTRAP = """
CREATE SCHEMA IF NOT EXISTS elec;
CREATE TABLE IF NOT EXISTS elec.schema_migrations (
version TEXT PRIMARY KEY,
checksum TEXT NOT NULL,
applied_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
"""
def _files() -> List[Path]:
return sorted(MIGRATIONS_DIR.glob("[0-9][0-9][0-9][0-9]_*.sql"))
def run_migrations() -> List[str]:
"""Apply pending migrations. Returns the versions applied by this call."""
applied_now: List[str] = []
with connect() as conn:
conn.execute(_BOOTSTRAP)
conn.commit()
done = {
r["version"]: r["checksum"]
for r in conn.execute("SELECT version, checksum FROM elec.schema_migrations")
}
for path in _files():
sql = path.read_text(encoding="utf-8")
checksum = hashlib.sha256(sql.encode("utf-8")).hexdigest()
version = path.stem
if version in done:
if done[version] != checksum:
raise RuntimeError(
f"Migration {version} was edited after it was applied. "
f"Revert the edit and add a new numbered migration instead."
)
continue
logger.info("Applying migration %s", version)
with conn.transaction():
conn.execute(sql)
conn.execute(
"INSERT INTO elec.schema_migrations (version, checksum) VALUES (%s, %s)",
(version, checksum),
)
applied_now.append(version)
return applied_now

View File

@@ -0,0 +1,4 @@
-- Extensions and the dedicated schema. Everything this project owns lives in
-- schema `elec` of database `electronics_catalog`.
CREATE EXTENSION IF NOT EXISTS vector;
CREATE SCHEMA IF NOT EXISTS elec;

View File

@@ -0,0 +1,57 @@
-- Reference data: brands, categories, retail sites. Seeded from
-- app/electronics/reference/*.yaml by `elec seed-reference`.
CREATE TABLE elec.brand (
id SERIAL PRIMARY KEY,
name TEXT NOT NULL UNIQUE,
slug TEXT NOT NULL UNIQUE,
parent_brand_id INT REFERENCES elec.brand(id),
is_popular BOOLEAN NOT NULL DEFAULT TRUE,
official_domains TEXT[] NOT NULL DEFAULT '{}',
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
-- Every spelling that resolves to a brand. Sub-brands (Redmi, iQOO, Pixel)
-- resolve to their parent and are remembered as the product family.
CREATE TABLE elec.brand_alias (
alias TEXT PRIMARY KEY CHECK (alias = lower(alias)),
brand_id INT NOT NULL REFERENCES elec.brand(id) ON DELETE CASCADE,
is_sub_brand BOOLEAN NOT NULL DEFAULT FALSE
);
CREATE TABLE elec.category (
id SERIAL PRIMARY KEY,
slug TEXT NOT NULL UNIQUE,
name TEXT NOT NULL UNIQUE
);
CREATE TABLE elec.brand_category (
brand_id INT NOT NULL REFERENCES elec.brand(id) ON DELETE CASCADE,
category_id INT NOT NULL REFERENCES elec.category(id) ON DELETE CASCADE,
PRIMARY KEY (brand_id, category_id)
);
-- A retail platform or a brand's own site, with the outcome of its probe.
-- probe_outcome A = fetchable with structured product data (scraped)
-- B = fetchable, product data from page HTML/state (scraped)
-- C = not fetched: serp_only policy, robots.txt disallow,
-- block/CAPTCHA, or unreachable -> web search only
CREATE TABLE elec.site (
id SERIAL PRIMARY KEY,
domain TEXT NOT NULL UNIQUE,
name TEXT NOT NULL,
kind TEXT NOT NULL CHECK (kind IN ('marketplace','national_chain','tn_regional','brand_official')),
region TEXT NOT NULL CHECK (region IN ('national','TN')),
policy TEXT NOT NULL CHECK (policy IN ('probe','serp_only')),
brand_id INT REFERENCES elec.brand(id),
product_url TEXT,
pincode_param TEXT,
enabled BOOLEAN NOT NULL DEFAULT TRUE,
probe_outcome CHAR(1) CHECK (probe_outcome IN ('A','B','C')),
robots_allowed BOOLEAN,
probe_evidence JSONB NOT NULL DEFAULT '{}'::jsonb,
probed_at TIMESTAMPTZ,
breaker_until TIMESTAMPTZ,
breaker_reason TEXT,
CHECK (kind <> 'brand_official' OR brand_id IS NOT NULL)
);

View File

@@ -0,0 +1,160 @@
-- Runs, fetch audit trail, search cache, listings, prices, canonical products.
CREATE TABLE elec.crawl_run (
id BIGSERIAL PRIMARY KEY,
kind TEXT NOT NULL,
params JSONB NOT NULL DEFAULT '{}'::jsonb,
status TEXT NOT NULL DEFAULT 'running' CHECK (status IN ('running','done','failed')),
stats JSONB NOT NULL DEFAULT '{}'::jsonb,
error TEXT,
started_at TIMESTAMPTZ NOT NULL DEFAULT now(),
ended_at TIMESTAMPTZ
);
-- Every HTTP request made to a retail or brand site. Evidence that the
-- crawler obeyed robots.txt and its rate limits.
CREATE TABLE elec.fetch_log (
id BIGSERIAL PRIMARY KEY,
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL,
url TEXT NOT NULL,
host TEXT NOT NULL,
status INT,
bytes INT,
outcome TEXT NOT NULL,
robots_allowed BOOLEAN,
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
CREATE INDEX fetch_log_host_time ON elec.fetch_log (host, fetched_at DESC);
CREATE TABLE elec.search_cache (
provider TEXT NOT NULL,
kind TEXT NOT NULL CHECK (kind IN ('text','images')),
query TEXT NOT NULL,
results JSONB NOT NULL,
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now(),
PRIMARY KEY (provider, kind, query)
);
-- One canonical product = one real-world variant (model + RAM + storage).
-- verification_status becomes 'verified' only when the product has a brand
-- official page, or listings on at least two different sites.
CREATE TABLE elec.product (
id BIGSERIAL PRIMARY KEY,
brand_id INT NOT NULL REFERENCES elec.brand(id),
category_id INT NOT NULL REFERENCES elec.category(id),
family TEXT,
model TEXT NOT NULL,
model_norm TEXT NOT NULL,
variant_key TEXT NOT NULL UNIQUE,
display_name TEXT NOT NULL,
ram_gb NUMERIC(6,1),
storage_gb NUMERIC(7,1),
processor TEXT,
mpn TEXT,
gtin TEXT,
canonical_specs JSONB NOT NULL DEFAULT '{}'::jsonb,
spec_sources JSONB NOT NULL DEFAULT '{}'::jsonb,
verification_status TEXT NOT NULL DEFAULT 'unverified'
CHECK (verification_status IN ('verified','unverified','rejected')),
evidence_count INT NOT NULL DEFAULT 0,
embedding vector(384),
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
);
CREATE INDEX product_brand_cat ON elec.product (brand_id, category_id);
CREATE INDEX product_specs_gin ON elec.product USING GIN (canonical_specs);
CREATE INDEX product_embedding_hnsw ON elec.product USING hnsw (embedding vector_cosine_ops);
-- The latest state of one product page on one site, or of one search result
-- that points at such a page. Nothing is stored without the URL it came from
-- and the text that the values were read from.
CREATE TABLE elec.source_listing (
id BIGSERIAL PRIMARY KEY,
site_id INT NOT NULL REFERENCES elec.site(id),
source_sku TEXT NOT NULL,
source_url TEXT NOT NULL CHECK (source_url ~ '^https?://'),
source_type TEXT NOT NULL CHECK (source_type IN ('scraped_page','search_snippet','brand_official')),
brand_id INT NOT NULL REFERENCES elec.brand(id),
category_id INT NOT NULL REFERENCES elec.category(id),
family TEXT,
title TEXT NOT NULL,
model TEXT,
model_number TEXT,
ram_gb NUMERIC(6,1),
storage_gb NUMERIC(7,1),
colour TEXT,
price NUMERIC(12,2) CHECK (price IS NULL OR price BETWEEN 500 AND 1000000),
mrp NUMERIC(12,2) CHECK (mrp IS NULL OR mrp BETWEEN 500 AND 1000000),
currency TEXT NOT NULL DEFAULT 'INR' CHECK (currency = 'INR'),
availability TEXT,
in_stock BOOLEAN,
pincode TEXT,
pincode_applied BOOLEAN NOT NULL DEFAULT FALSE,
rating NUMERIC(3,2) CHECK (rating IS NULL OR rating BETWEEN 0 AND 5),
review_count INT,
gtin TEXT,
image_urls TEXT[] NOT NULL DEFAULT '{}',
specs_raw JSONB NOT NULL DEFAULT '{}'::jsonb,
specs JSONB NOT NULL DEFAULT '{}'::jsonb,
evidence_text TEXT NOT NULL CHECK (length(evidence_text) > 0),
search_query TEXT,
confidence NUMERIC(3,2) NOT NULL CHECK (confidence BETWEEN 0 AND 1),
parser TEXT NOT NULL,
content_hash TEXT,
first_seen_at TIMESTAMPTZ NOT NULL DEFAULT now(),
last_seen_at TIMESTAMPTZ NOT NULL DEFAULT now(),
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL,
UNIQUE (site_id, source_sku),
CHECK (pincode_applied = FALSE OR pincode IS NOT NULL)
);
CREATE INDEX listing_brand_cat ON elec.source_listing (brand_id, category_id);
-- Append-only price observations. UPDATE is refused by a trigger.
CREATE TABLE elec.price_history (
id BIGSERIAL PRIMARY KEY,
listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
price NUMERIC(12,2) CHECK (price IS NULL OR price BETWEEN 500 AND 1000000),
mrp NUMERIC(12,2),
availability TEXT,
in_stock BOOLEAN,
source_type TEXT NOT NULL,
pincode TEXT,
pincode_applied BOOLEAN NOT NULL DEFAULT FALSE,
evidence_text TEXT NOT NULL CHECK (length(evidence_text) > 0),
observed_at TIMESTAMPTZ NOT NULL DEFAULT now(),
crawl_run_id BIGINT REFERENCES elec.crawl_run(id) ON DELETE SET NULL
);
CREATE INDEX price_history_listing_time ON elec.price_history (listing_id, observed_at DESC);
CREATE FUNCTION elec.refuse_update() RETURNS trigger LANGUAGE plpgsql AS $$
BEGIN
RAISE EXCEPTION 'elec.price_history is append-only';
END $$;
CREATE TRIGGER price_history_append_only BEFORE UPDATE ON elec.price_history
FOR EACH ROW EXECUTE FUNCTION elec.refuse_update();
-- Which canonical product a listing belongs to, and how sure we are.
-- Only 'auto' and 'approved' links count as evidence or appear in views.
CREATE TABLE elec.product_listing_map (
listing_id BIGINT PRIMARY KEY REFERENCES elec.source_listing(id) ON DELETE CASCADE,
product_id BIGINT NOT NULL REFERENCES elec.product(id) ON DELETE CASCADE,
method TEXT NOT NULL CHECK (method IN ('gtin','mpn','variant_key','fuzzy','manual')),
confidence NUMERIC(3,2) NOT NULL CHECK (confidence BETWEEN 0 AND 1),
review_status TEXT NOT NULL CHECK (review_status IN ('auto','pending','approved','rejected')),
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
reviewed_at TIMESTAMPTZ
);
CREATE INDEX map_product ON elec.product_listing_map (product_id);
-- Images are URLs only (never downloaded), each tied to the listing it was
-- found on and checked live.
CREATE TABLE elec.product_image (
id BIGSERIAL PRIMARY KEY,
product_id BIGINT NOT NULL REFERENCES elec.product(id) ON DELETE CASCADE,
url TEXT NOT NULL CHECK (url ~ '^https?://'),
source_listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
source_type TEXT NOT NULL,
rank INT NOT NULL DEFAULT 100,
validated_at TIMESTAMPTZ NOT NULL DEFAULT now(),
UNIQUE (product_id, url)
);

View File

@@ -0,0 +1,71 @@
-- Read-side views. Public views only ever show VERIFIED products and links
-- that are 'auto' or 'approved'.
CREATE VIEW elec.v_product_availability AS
SELECT p.id AS product_id,
b.name AS brand,
c.slug AS category,
p.display_name,
s.id AS site_id,
s.name AS site,
s.domain,
s.kind AS site_kind,
s.region AS site_region,
l.id AS listing_id,
l.source_url,
l.source_type,
l.title AS listing_title,
l.colour,
l.price,
l.mrp,
l.in_stock,
l.availability,
l.pincode,
l.pincode_applied,
l.confidence,
l.last_seen_at AS observed_at
FROM elec.product p
JOIN elec.brand b ON b.id = p.brand_id
JOIN elec.category c ON c.id = p.category_id
JOIN elec.product_listing_map m ON m.product_id = p.id AND m.review_status IN ('auto','approved')
JOIN elec.source_listing l ON l.id = m.listing_id
JOIN elec.site s ON s.id = l.site_id
WHERE p.verification_status = 'verified';
-- Cheapest known price per product. Scraped prices are preferred over search
-- snippet prices; a listing known to be out of stock is skipped.
CREATE VIEW elec.v_best_price AS
SELECT DISTINCT ON (product_id)
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
FROM elec.v_product_availability
WHERE price IS NOT NULL AND in_stock IS DISTINCT FROM FALSE
ORDER BY product_id, (source_type = 'search_snippet'), price, observed_at DESC;
CREATE VIEW elec.v_brand_catalog AS
SELECT p.id AS product_id, b.name AS brand, b.slug AS brand_slug, c.slug AS category,
p.family, p.display_name, p.model, p.ram_gb, p.storage_gb, p.processor,
p.canonical_specs,
bp.price AS best_price,
bp.site AS best_price_site,
bp.source_type AS best_price_source_type,
(SELECT count(DISTINCT a.site_id) FROM elec.v_product_availability a
WHERE a.product_id = p.id) AS platform_count,
(SELECT coalesce(bool_or(a.site_region = 'TN'), FALSE) FROM elec.v_product_availability a
WHERE a.product_id = p.id) AS sold_by_tn_retailer,
(SELECT i.url FROM elec.product_image i WHERE i.product_id = p.id
ORDER BY i.rank, i.id LIMIT 1) AS image_url,
p.updated_at
FROM elec.product p
JOIN elec.brand b ON b.id = p.brand_id
JOIN elec.category c ON c.id = p.category_id
LEFT JOIN elec.v_best_price bp ON bp.product_id = p.id
WHERE p.verification_status = 'verified';
CREATE VIEW elec.v_brand_summary AS
SELECT brand, brand_slug, category,
count(*) AS product_count,
min(best_price) AS min_price,
max(best_price) AS max_price,
max(platform_count) AS max_platforms
FROM elec.v_brand_catalog
GROUP BY brand, brand_slug, category;

View File

@@ -0,0 +1,44 @@
-- Search results carry cached, sometimes seller-specific prices. A price that
-- disagrees sharply with the product-page price for the same product (or is
-- below what the category can cost) is kept with its evidence but flagged, and
-- is never used as the "best price". Set by repository.flag_price_outliers().
ALTER TABLE elec.source_listing ADD COLUMN price_outlier BOOLEAN NOT NULL DEFAULT FALSE;
CREATE OR REPLACE VIEW elec.v_product_availability AS
SELECT p.id AS product_id,
b.name AS brand,
c.slug AS category,
p.display_name,
s.id AS site_id,
s.name AS site,
s.domain,
s.kind AS site_kind,
s.region AS site_region,
l.id AS listing_id,
l.source_url,
l.source_type,
l.title AS listing_title,
l.colour,
l.price,
l.mrp,
l.in_stock,
l.availability,
l.pincode,
l.pincode_applied,
l.confidence,
l.last_seen_at AS observed_at,
l.price_outlier
FROM elec.product p
JOIN elec.brand b ON b.id = p.brand_id
JOIN elec.category c ON c.id = p.category_id
JOIN elec.product_listing_map m ON m.product_id = p.id AND m.review_status IN ('auto','approved')
JOIN elec.source_listing l ON l.id = m.listing_id
JOIN elec.site s ON s.id = l.site_id
WHERE p.verification_status = 'verified';
CREATE OR REPLACE VIEW elec.v_best_price AS
SELECT DISTINCT ON (product_id)
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
FROM elec.v_product_availability
WHERE price IS NOT NULL AND NOT price_outlier AND in_stock IS DISTINCT FROM FALSE
ORDER BY product_id, (source_type = 'search_snippet'), price, observed_at DESC;

View File

@@ -0,0 +1,30 @@
-- Best price: a product whose every priced listing is out of stock still has a
-- price worth showing. In-stock (or unknown-stock) prices still win; an
-- out-of-stock price is used only when nothing else is priced. Same columns
-- as 0005, so v_brand_catalog keeps working unchanged.
CREATE OR REPLACE VIEW elec.v_best_price AS
SELECT DISTINCT ON (product_id)
product_id, site, domain, source_url, source_type, price, mrp, in_stock, observed_at
FROM elec.v_product_availability
WHERE price IS NOT NULL AND NOT price_outlier
ORDER BY product_id, (in_stock IS FALSE), (source_type = 'search_snippet'), price, observed_at DESC;
-- Individual customer reviews, exactly as a product page publishes them in its
-- schema.org JSON-LD. Nothing here is generated: every row is a review the
-- listing's own page stated. Sentiment is derived only from the reviewer's
-- own star rating (>=4 positive, >=3 neutral, <3 negative); NULL when the
-- review states no rating.
CREATE TABLE elec.listing_review (
id BIGSERIAL PRIMARY KEY,
listing_id BIGINT NOT NULL REFERENCES elec.source_listing(id) ON DELETE CASCADE,
author TEXT,
rating NUMERIC(2,1) CHECK (rating IS NULL OR rating BETWEEN 0 AND 5),
title TEXT,
body TEXT NOT NULL,
review_date TEXT,
sentiment TEXT CHECK (sentiment IS NULL OR sentiment IN ('positive','neutral','negative')),
content_hash TEXT NOT NULL,
fetched_at TIMESTAMPTZ NOT NULL DEFAULT now(),
UNIQUE (listing_id, content_hash)
);
CREATE INDEX listing_review_listing_idx ON elec.listing_review (listing_id);

View File

@@ -0,0 +1,588 @@
"""All SQL used by the pipeline. psycopg3, no ORM - the same style as the
original project, with each function owning one statement or one small unit
of work."""
from __future__ import annotations
import hashlib
import json
import re
from datetime import datetime, timezone
from decimal import Decimal
from typing import Any, Dict, List, Optional
from psycopg.types.json import Jsonb
from app.electronics.db.connection import connect, transaction
from app.electronics.models import Listing
from app.electronics.reference import Reference, slugify
def _json(value: Any) -> Jsonb:
return Jsonb(json.loads(json.dumps(value, default=str)))
# ---------------------------------------------------------------------------
# Reference data
# ---------------------------------------------------------------------------
def seed_reference(ref: Reference) -> Dict[str, int]:
"""Idempotent upsert of brands, aliases, categories and sites."""
counts = {"brands": 0, "aliases": 0, "categories": 0, "sites": 0}
with transaction() as conn:
for c in ref.categories.values():
conn.execute(
"INSERT INTO elec.category (slug, name) VALUES (%s, %s) "
"ON CONFLICT (slug) DO UPDATE SET name = EXCLUDED.name",
(c.slug, c.name),
)
counts["categories"] += 1
for b in ref.brands.values():
row = conn.execute(
"INSERT INTO elec.brand (name, slug, official_domains) VALUES (%s, %s, %s) "
"ON CONFLICT (slug) DO UPDATE SET name = EXCLUDED.name, official_domains = EXCLUDED.official_domains "
"RETURNING id",
(b.name, b.slug, list(b.official)),
).fetchone()
counts["brands"] += 1
for alias in b.aliases:
conn.execute(
"INSERT INTO elec.brand_alias (alias, brand_id, is_sub_brand) VALUES (%s, %s, FALSE) "
"ON CONFLICT (alias) DO UPDATE SET brand_id = EXCLUDED.brand_id, is_sub_brand = FALSE",
(alias, row["id"]),
)
counts["aliases"] += 1
for sub in b.sub_brands:
conn.execute(
"INSERT INTO elec.brand_alias (alias, brand_id, is_sub_brand) VALUES (%s, %s, TRUE) "
"ON CONFLICT (alias) DO UPDATE SET brand_id = EXCLUDED.brand_id, is_sub_brand = TRUE",
(sub, row["id"]),
)
counts["aliases"] += 1
for cat in b.categories:
conn.execute(
"INSERT INTO elec.brand_category (brand_id, category_id) "
"SELECT %s, id FROM elec.category WHERE slug = %s ON CONFLICT DO NOTHING",
(row["id"], cat),
)
for s in ref.sites.values():
conn.execute(
"""
INSERT INTO elec.site (domain, name, kind, region, policy, brand_id, product_url, pincode_param)
VALUES (%s, %s, %s, %s, %s, (SELECT id FROM elec.brand WHERE slug = %s), %s, %s)
ON CONFLICT (domain) DO UPDATE SET
name = EXCLUDED.name, kind = EXCLUDED.kind, region = EXCLUDED.region,
policy = EXCLUDED.policy, brand_id = EXCLUDED.brand_id,
product_url = EXCLUDED.product_url, pincode_param = EXCLUDED.pincode_param
""",
(s.domain, s.name, s.kind, s.region, s.policy, s.brand_slug, s.product_url, s.pincode_param),
)
counts["sites"] += 1
return counts
def id_maps() -> Dict[str, Dict[str, int]]:
with connect() as conn:
return {
"brand": {r["slug"]: r["id"] for r in conn.execute("SELECT id, slug FROM elec.brand")},
"category": {r["slug"]: r["id"] for r in conn.execute("SELECT id, slug FROM elec.category")},
"site": {r["domain"]: r["id"] for r in conn.execute("SELECT id, domain FROM elec.site")},
}
def sites() -> List[dict]:
with connect() as conn:
return list(conn.execute("SELECT * FROM elec.site ORDER BY kind, name"))
def set_probe_result(domain: str, outcome: str, robots_allowed: Optional[bool], evidence: dict) -> None:
with transaction() as conn:
conn.execute(
"UPDATE elec.site SET probe_outcome = %s, robots_allowed = %s, probe_evidence = %s, probed_at = now() "
"WHERE domain = %s",
(outcome, robots_allowed, _json(evidence), domain),
)
def trip_breaker(domain: str, reason: str, until_epoch: float) -> None:
with transaction() as conn:
conn.execute(
"UPDATE elec.site SET breaker_until = to_timestamp(%s), breaker_reason = %s, "
"probe_outcome = 'C' WHERE domain = %s OR %s LIKE '%%.' || domain",
(until_epoch, reason, domain, domain),
)
# ---------------------------------------------------------------------------
# Runs and fetch log
# ---------------------------------------------------------------------------
def start_run(kind: str, params: dict) -> int:
with transaction() as conn:
return conn.execute(
"INSERT INTO elec.crawl_run (kind, params) VALUES (%s, %s) RETURNING id", (kind, _json(params))
).fetchone()["id"]
def finish_run(run_id: int, status: str, stats: dict, error: Optional[str] = None) -> None:
with transaction() as conn:
conn.execute(
"UPDATE elec.crawl_run SET status = %s, stats = %s, error = %s, ended_at = now() WHERE id = %s",
(status, _json(stats), error, run_id),
)
def log_fetch(run_id: Optional[int], url: str, host: str, status: Optional[int], nbytes: int,
outcome: str, robots_allowed: Optional[bool]) -> None:
with transaction() as conn:
conn.execute(
"INSERT INTO elec.fetch_log (crawl_run_id, url, host, status, bytes, outcome, robots_allowed) "
"VALUES (%s, %s, %s, %s, %s, %s, %s)",
(run_id, url, host, status, nbytes, outcome, robots_allowed),
)
def recent_runs(limit: int = 20) -> List[dict]:
with connect() as conn:
return list(conn.execute("SELECT * FROM elec.crawl_run ORDER BY id DESC LIMIT %s", (limit,)))
# ---------------------------------------------------------------------------
# Search cache
# ---------------------------------------------------------------------------
def search_cache_get(provider: str, kind: str, query: str, ttl_hours: int) -> Optional[List[dict]]:
with connect() as conn:
row = conn.execute(
"SELECT results FROM elec.search_cache WHERE provider = %s AND kind = %s AND query = %s "
"AND fetched_at > now() - make_interval(hours => %s)",
(provider, kind, query, ttl_hours),
).fetchone()
return row["results"] if row else None
def search_cache_put(provider: str, kind: str, query: str, results: List[dict]) -> None:
with transaction() as conn:
conn.execute(
"INSERT INTO elec.search_cache (provider, kind, query, results) VALUES (%s, %s, %s, %s) "
"ON CONFLICT (provider, kind, query) DO UPDATE SET results = EXCLUDED.results, fetched_at = now()",
(provider, kind, query, _json(results)),
)
def google_queries_today() -> int:
with connect() as conn:
return conn.execute(
"SELECT count(*) AS n FROM elec.search_cache WHERE provider = 'google' AND fetched_at::date = current_date"
).fetchone()["n"]
# ---------------------------------------------------------------------------
# Listings and prices
# ---------------------------------------------------------------------------
def upsert_listing(listing: Listing, ids: Dict[str, Dict[str, int]], run_id: Optional[int]) -> int:
"""Write the latest state of a listing and append one price observation."""
listing.validate()
site_id = ids["site"][listing.site_domain]
brand_id = ids["brand"][listing.brand_slug]
category_id = ids["category"][listing.category]
with transaction() as conn:
existing = conn.execute(
"SELECT id, source_type, price FROM elec.source_listing WHERE site_id = %s AND source_sku = %s",
(site_id, listing.source_sku),
).fetchone()
# A scraped page is better evidence than a search snippet about the
# same page. Never let a later snippet overwrite scraped values.
if existing and existing["source_type"] in ("scraped_page", "brand_official") and listing.source_type == "search_snippet":
conn.execute("UPDATE elec.source_listing SET last_seen_at = now() WHERE id = %s", (existing["id"],))
return existing["id"]
params = dict(
site_id=site_id, source_sku=listing.source_sku, source_url=listing.source_url,
source_type=listing.source_type, brand_id=brand_id, category_id=category_id,
family=listing.family, title=listing.title[:500], model=listing.model,
model_number=listing.model_number, ram_gb=listing.ram_gb, storage_gb=listing.storage_gb,
colour=listing.colour, price=listing.price, mrp=listing.mrp, availability=listing.availability,
in_stock=listing.in_stock, pincode=listing.pincode, pincode_applied=listing.pincode_applied,
rating=listing.rating, review_count=listing.review_count, gtin=listing.gtin,
image_urls=listing.image_urls[:12], specs_raw=_json(listing.specs_raw), specs=_json(listing.specs),
evidence_text=listing.evidence_text[:4000], search_query=listing.search_query,
confidence=round(listing.confidence, 2), parser=listing.parser, content_hash=listing.content_hash,
crawl_run_id=run_id,
)
row = conn.execute(
"""
INSERT INTO elec.source_listing (
site_id, source_sku, source_url, source_type, brand_id, category_id, family, title, model,
model_number, ram_gb, storage_gb, colour, price, mrp, availability, in_stock, pincode,
pincode_applied, rating, review_count, gtin, image_urls, specs_raw, specs, evidence_text,
search_query, confidence, parser, content_hash, crawl_run_id)
VALUES (
%(site_id)s, %(source_sku)s, %(source_url)s, %(source_type)s, %(brand_id)s, %(category_id)s,
%(family)s, %(title)s, %(model)s, %(model_number)s, %(ram_gb)s, %(storage_gb)s, %(colour)s,
%(price)s, %(mrp)s, %(availability)s, %(in_stock)s, %(pincode)s, %(pincode_applied)s,
%(rating)s, %(review_count)s, %(gtin)s, %(image_urls)s, %(specs_raw)s, %(specs)s,
%(evidence_text)s, %(search_query)s, %(confidence)s, %(parser)s, %(content_hash)s,
%(crawl_run_id)s)
ON CONFLICT (site_id, source_sku) DO UPDATE SET
source_url = EXCLUDED.source_url, source_type = EXCLUDED.source_type,
brand_id = EXCLUDED.brand_id, category_id = EXCLUDED.category_id, family = EXCLUDED.family,
title = EXCLUDED.title, model = EXCLUDED.model, model_number = EXCLUDED.model_number,
ram_gb = EXCLUDED.ram_gb, storage_gb = EXCLUDED.storage_gb, colour = EXCLUDED.colour,
price = EXCLUDED.price, mrp = EXCLUDED.mrp, availability = EXCLUDED.availability,
in_stock = EXCLUDED.in_stock, pincode = EXCLUDED.pincode,
pincode_applied = EXCLUDED.pincode_applied, rating = EXCLUDED.rating,
review_count = EXCLUDED.review_count, gtin = EXCLUDED.gtin, image_urls = EXCLUDED.image_urls,
specs_raw = EXCLUDED.specs_raw, specs = EXCLUDED.specs, evidence_text = EXCLUDED.evidence_text,
search_query = EXCLUDED.search_query, confidence = EXCLUDED.confidence, parser = EXCLUDED.parser,
content_hash = EXCLUDED.content_hash, crawl_run_id = EXCLUDED.crawl_run_id, last_seen_at = now()
RETURNING id
""",
params,
).fetchone()
listing_id = row["id"]
if listing.price is not None or listing.in_stock is not None:
conn.execute(
"INSERT INTO elec.price_history (listing_id, price, mrp, availability, in_stock, source_type, "
"pincode, pincode_applied, evidence_text, crawl_run_id) VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)",
(listing_id, listing.price, listing.mrp, listing.availability, listing.in_stock,
listing.source_type, listing.pincode, listing.pincode_applied,
listing.evidence_text[:2000], run_id),
)
return listing_id
# ---------------------------------------------------------------------------
# Ratings and reviews
# ---------------------------------------------------------------------------
def save_reviews(listing_id: int, reviews: List[Dict[str, Any]]) -> int:
"""Store the reviews a listing's page publishes. Idempotent per review
text; an empty list changes nothing (a later search-only sighting must
not erase what the page said). Returns the number of new rows."""
from app.electronics.reviews import sentiment_for
added = 0
with transaction() as conn:
for r in reviews:
body = (r.get("body") or "").strip()
if not body:
continue
digest = hashlib.sha1(f"{r.get('author') or ''}|{body}".encode("utf-8", "ignore")).hexdigest()
row = conn.execute(
"INSERT INTO elec.listing_review (listing_id, author, rating, title, body, review_date, sentiment, "
"content_hash) VALUES (%s,%s,%s,%s,%s,%s,%s,%s) "
"ON CONFLICT (listing_id, content_hash) DO NOTHING RETURNING id",
(listing_id, (r.get("author") or None) and str(r["author"])[:200], r.get("rating"),
(r.get("title") or None) and str(r["title"])[:300], body[:4000],
(r.get("review_date") or None) and str(r["review_date"])[:40],
sentiment_for(r.get("rating")), digest),
).fetchone()
added += 1 if row else 0
return added
def update_listing_rating(listing_id: int, rating: Optional[Decimal], review_count: Optional[int]) -> None:
"""Refresh only the rating fields of a listing (used by the review backfill)."""
with transaction() as conn:
conn.execute(
"UPDATE elec.source_listing SET rating = %s, review_count = %s WHERE id = %s",
(rating, review_count, listing_id),
)
def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
"""Per-platform ratings and all stored reviews for a verified product's
approved listings, each with the page it was read from."""
sources = conn.execute(
"SELECT a.site, a.source_url, l.rating, l.review_count FROM elec.v_product_availability a "
"JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.product_id = %s AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
(product_id,),
).fetchall()
reviews = conn.execute(
"SELECT a.site, a.source_url, r.author, r.rating, r.title, r.body, r.review_date, r.sentiment "
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
"WHERE a.product_id = %s",
(product_id,),
).fetchall()
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
"""Page-read listings of verified products, for re-reading ratings/reviews."""
sql = (
"SELECT a.listing_id, a.source_url, a.domain, a.site_kind, a.category, l.source_sku, l.title "
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.source_type IN ('scraped_page','brand_official')"
)
params: tuple = ()
if category:
sql += " AND a.category = %s"
params = (category,)
with connect() as conn:
return list(conn.execute(sql + " ORDER BY a.listing_id", params))
# ---------------------------------------------------------------------------
# Products, matching, images
# ---------------------------------------------------------------------------
def product_candidates(brand_slug: str, category: str) -> List[dict]:
with connect() as conn:
return list(conn.execute(
"SELECT p.id, p.variant_key, p.model_norm, p.ram_gb, p.storage_gb, p.processor, p.mpn, p.gtin "
"FROM elec.product p JOIN elec.brand b ON b.id = p.brand_id JOIN elec.category c ON c.id = p.category_id "
"WHERE b.slug = %s AND c.slug = %s AND p.verification_status <> 'rejected'",
(brand_slug, category),
))
def _cpu_label(processor: Optional[str]) -> str:
""""ryzen 5 7530u" -> "Ryzen 5 7530U", "i5-1334u" -> "i5-1334U"."""
if not processor:
return ""
def fmt(t: str) -> str:
if re.fullmatch(r"i[3579]-\w+", t):
return "i" + t[1:].upper() # i5-1334U
if any(ch.isdigit() for ch in t):
return t.upper() # 7530U, M5
return t.title() # Ryzen, Core, Ultra
return " ".join(fmt(t) for t in processor.split())
def product_display_name(listing: Listing) -> str:
variant = [x for x in (
_cpu_label(listing.processor) if listing.category == "laptops" else "",
f"{_fmt_gb(listing.ram_gb)} RAM" if listing.ram_gb else "",
_fmt_gb(listing.storage_gb) if listing.storage_gb else "",
) if x]
if listing.category == "laptops" and not listing.processor and listing.model_number:
variant.insert(0, listing.model_number) # the part number is what tells it apart
return " ".join(x for x in [load_brand_name(listing.brand_slug), listing.model,
f"({', '.join(variant)})" if variant else ""] if x)
def create_product(listing: Listing, ids: Dict[str, Dict[str, int]]) -> int:
display = product_display_name(listing)
with transaction() as conn:
row = conn.execute(
"""
INSERT INTO elec.product (brand_id, category_id, family, model, model_norm, variant_key, display_name,
ram_gb, storage_gb, processor, mpn, gtin)
VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)
ON CONFLICT (variant_key) DO UPDATE SET updated_at = now()
RETURNING id
""",
(ids["brand"][listing.brand_slug], ids["category"][listing.category], listing.family,
listing.model or listing.model_norm, listing.model_norm, listing.variant_key, display,
listing.ram_gb, listing.storage_gb, listing.processor, listing.model_number, listing.gtin),
).fetchone()
return row["id"]
def _fmt_gb(value: Optional[Decimal]) -> str:
if value is None:
return ""
if value >= 1024 and value % 1024 == 0:
return f"{int(value // 1024)}TB"
return f"{format(value.normalize(), 'f')}GB"
_BRAND_NAMES: Dict[str, str] = {}
def load_brand_name(slug: str) -> str:
if not _BRAND_NAMES:
from app.electronics.reference import load_reference
_BRAND_NAMES.update({s: b.name for s, b in load_reference().brands.items()})
return _BRAND_NAMES.get(slug, slug.title())
def map_listing(listing_id: int, product_id: int, method: str, confidence: float, review_status: str) -> None:
with transaction() as conn:
conn.execute(
"""
INSERT INTO elec.product_listing_map (listing_id, product_id, method, confidence, review_status)
VALUES (%s, %s, %s, %s, %s)
ON CONFLICT (listing_id) DO UPDATE SET
product_id = EXCLUDED.product_id, method = EXCLUDED.method, confidence = EXCLUDED.confidence,
review_status = CASE WHEN elec.product_listing_map.review_status IN ('approved','rejected')
AND elec.product_listing_map.product_id = EXCLUDED.product_id
THEN elec.product_listing_map.review_status
ELSE EXCLUDED.review_status END
""",
(listing_id, product_id, method, round(confidence, 2), review_status),
)
def merge_product_specs(product_id: int, specs: Dict[str, Any], sources: Dict[str, str], source_url: str) -> None:
"""Add spec keys the product does not have yet. Existing values win:
specs are only ever filled, never overwritten by a later source."""
if not specs:
return
with transaction() as conn:
row = conn.execute(
"SELECT canonical_specs, spec_sources FROM elec.product WHERE id = %s FOR UPDATE", (product_id,)
).fetchone()
current, cur_src = dict(row["canonical_specs"] or {}), dict(row["spec_sources"] or {})
changed = False
for key, value in specs.items():
if key not in current:
current[key] = value
cur_src[key] = {"url": source_url, "from": sources.get(key, "")}
changed = True
if changed:
conn.execute(
"UPDATE elec.product SET canonical_specs = %s, spec_sources = %s, updated_at = now() WHERE id = %s",
(_json(current), _json(cur_src), product_id),
)
def add_image(product_id: int, url: str, listing_id: int, source_type: str, rank: int) -> None:
with transaction() as conn:
conn.execute(
"INSERT INTO elec.product_image (product_id, url, source_listing_id, source_type, rank) "
"VALUES (%s, %s, %s, %s, %s) ON CONFLICT (product_id, url) DO UPDATE SET validated_at = now()",
(product_id, url, listing_id, source_type, rank),
)
def product_image_count(product_id: int) -> int:
with connect() as conn:
return conn.execute("SELECT count(*) AS n FROM elec.product_image WHERE product_id = %s",
(product_id,)).fetchone()["n"]
# The least a new device in the category can plausibly cost. Anything below is
# an accessory, an EMI or an offer amount that slipped through.
CATEGORY_MIN_PRICE = {"mobiles": 3000, "laptops": 15000}
OUTLIER_TOLERANCE = 0.35
def flag_price_outliers() -> int:
"""Flag prices that cannot be trusted as this product's price:
* below the category's floor (CATEGORY_MIN_PRICE);
* a search-result price more than OUTLIER_TOLERANCE away from the price
read off a product page for the same product;
* with no page price, a search-result price that far from the median of
at least three prices for the product.
Flagged prices stay stored with their evidence; they are just never used as
the best price. Returns the number flagged."""
floor_cases = " ".join(f"WHEN '{k}' THEN {v}" for k, v in CATEGORY_MIN_PRICE.items())
with transaction() as conn:
conn.execute("UPDATE elec.source_listing SET price_outlier = FALSE WHERE price_outlier")
cur = conn.execute(
f"""
WITH prices AS (
SELECT l.id, m.product_id, l.price, l.source_type, c.slug
FROM elec.source_listing l
JOIN elec.product_listing_map m ON m.listing_id = l.id AND m.review_status IN ('auto','approved')
JOIN elec.category c ON c.id = l.category_id
WHERE l.price IS NOT NULL
),
ref AS (
SELECT product_id,
percentile_cont(0.5) WITHIN GROUP (ORDER BY price)
FILTER (WHERE source_type <> 'search_snippet') AS page_median,
percentile_cont(0.5) WITHIN GROUP (ORDER BY price) AS all_median,
count(*) AS n
FROM prices GROUP BY product_id
)
UPDATE elec.source_listing l SET price_outlier = TRUE
FROM prices p JOIN ref r ON r.product_id = p.product_id
WHERE l.id = p.id AND (
p.price < CASE p.slug {floor_cases} ELSE 0 END
OR (p.source_type = 'search_snippet' AND r.page_median IS NOT NULL
AND abs(p.price - r.page_median) / r.page_median > %(tol)s)
OR (p.source_type = 'search_snippet' AND r.page_median IS NULL AND r.n >= 3
AND abs(p.price - r.all_median) / r.all_median > %(tol)s)
)
""",
{"tol": OUTLIER_TOLERANCE},
)
return cur.rowcount
def refresh_verification() -> Dict[str, int]:
flag_price_outliers()
"""A product is VERIFIED when auto/approved listings on at least two
different sites point at it, and at least one of them is a retailer
(so it is actually sold). Everything else stays unverified and hidden."""
with transaction() as conn:
conn.execute(
"""
WITH ev AS (
SELECT m.product_id,
count(DISTINCT l.site_id) AS sites,
count(DISTINCT l.site_id) FILTER (WHERE s.kind <> 'brand_official') AS retail_sites
FROM elec.product_listing_map m
JOIN elec.source_listing l ON l.id = m.listing_id
JOIN elec.site s ON s.id = l.site_id
WHERE m.review_status IN ('auto','approved')
GROUP BY m.product_id
)
UPDATE elec.product p SET
evidence_count = coalesce(ev.sites, 0),
verification_status = CASE
WHEN p.verification_status = 'rejected' THEN 'rejected'
WHEN coalesce(ev.sites, 0) >= 2 AND coalesce(ev.retail_sites, 0) >= 1 THEN 'verified'
ELSE 'unverified' END,
updated_at = now()
FROM elec.product p2 LEFT JOIN ev ON ev.product_id = p2.id
WHERE p.id = p2.id
"""
)
rows = conn.execute(
"SELECT verification_status AS s, count(*) AS n FROM elec.product GROUP BY 1"
).fetchall()
return {r["s"]: r["n"] for r in rows}
def products_without_embedding(limit: int = 500) -> List[dict]:
with connect() as conn:
return list(conn.execute(
"SELECT p.id, p.display_name, p.canonical_specs, b.name AS brand, c.name AS category "
"FROM elec.product p JOIN elec.brand b ON b.id = p.brand_id JOIN elec.category c ON c.id = p.category_id "
"WHERE p.embedding IS NULL AND p.verification_status = 'verified' LIMIT %s", (limit,)))
def set_embedding(product_id: int, vector: List[float]) -> None:
import numpy as np
with transaction() as conn:
conn.execute("UPDATE elec.product SET embedding = %s WHERE id = %s", (np.array(vector), product_id))
def review_queue(limit: int = 100) -> List[dict]:
with connect() as conn:
return list(conn.execute(
"""
SELECT m.listing_id, m.product_id, m.method, m.confidence, l.title AS listing_title, l.source_url,
s.name AS site, p.display_name AS product
FROM elec.product_listing_map m
JOIN elec.source_listing l ON l.id = m.listing_id
JOIN elec.site s ON s.id = l.site_id
JOIN elec.product p ON p.id = m.product_id
WHERE m.review_status = 'pending'
ORDER BY m.confidence DESC, m.listing_id LIMIT %s
""", (limit,)))
def set_review(listing_id: int, approve: bool) -> bool:
with transaction() as conn:
cur = conn.execute(
"UPDATE elec.product_listing_map SET review_status = %s, reviewed_at = now() "
"WHERE listing_id = %s AND review_status = 'pending'",
("approved" if approve else "rejected", listing_id),
)
return cur.rowcount > 0
def now_utc() -> datetime:
return datetime.now(timezone.utc)
def grounding_sample(n: int = 50) -> List[dict]:
with connect() as conn:
return list(conn.execute(
"SELECT id, source_url, source_type, price, evidence_text FROM elec.source_listing "
"WHERE price IS NOT NULL ORDER BY random() LIMIT %s", (n,)))
def slug(text: str) -> str:
return slugify(text)

View File

@@ -0,0 +1,123 @@
"""Product facts from page markup when there is no usable JSON-LD.
Only machine-readable markup is trusted for the price: OpenGraph/product meta
tags and schema.org microdata (itemprop="price"). Free text on the page is not
scanned for rupee amounts - a product page shows EMIs, offers and other
products' prices, and picking the wrong one is worse than picking none.
Spec tables (<table>, <dl>) supply specifications.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, List, Optional
from bs4 import BeautifulSoup
def _dec(value: Optional[str]) -> Optional[Decimal]:
if not value:
return None
try:
return Decimal(re.sub(r"[^\d.]", "", value))
except InvalidOperation:
return None
def _meta(soup: BeautifulSoup, *names: str) -> Optional[str]:
for name in names:
tag = soup.find("meta", attrs={"property": name}) or soup.find("meta", attrs={"name": name})
if tag and tag.get("content"):
return tag["content"].strip()
return None
def spec_tables(soup: BeautifulSoup, limit: int = 200) -> Dict[str, str]:
specs: Dict[str, str] = {}
for row in soup.select("table tr"):
cells = row.find_all(["th", "td"])
if len(cells) == 2:
k, v = (c.get_text(" ", strip=True) for c in cells)
if k and v and len(k) <= 60 and len(v) <= 200:
specs.setdefault(k, v)
if len(specs) >= limit:
return specs
for dl in soup.find_all("dl"):
for dt in dl.find_all("dt"):
dd = dt.find_next_sibling("dd")
if dd:
k, v = dt.get_text(" ", strip=True), dd.get_text(" ", strip=True)
if k and v and len(k) <= 60 and len(v) <= 200:
specs.setdefault(k, v)
return specs
def extract_page(html: str) -> Dict[str, Any]:
soup = BeautifulSoup(html, "lxml")
title = _meta(soup, "og:title", "twitter:title")
if not title:
h1 = soup.find("h1")
title = h1.get_text(" ", strip=True) if h1 else None
images: List[str] = []
for name in ("og:image", "og:image:secure_url", "twitter:image"):
v = _meta(soup, name)
if v and v.startswith("http") and v not in images:
images.append(v)
price = _dec(_meta(soup, "product:price:amount", "og:price:amount"))
currency = _meta(soup, "product:price:currency", "og:price:currency")
evidence = ""
if price is not None:
evidence = f"meta product:price:amount={price} currency={currency}"
else:
tag = soup.find(attrs={"itemprop": "price"})
if tag is not None:
raw = tag.get("content") or tag.get_text(" ", strip=True)
price = _dec(raw)
cur_tag = soup.find(attrs={"itemprop": "priceCurrency"})
currency = (cur_tag.get("content") if cur_tag else None) or currency
if price is not None:
evidence = f'itemprop="price" {raw} currency={currency}'
availability = _meta(soup, "product:availability", "og:availability")
in_stock = None
if availability:
low = availability.lower().replace(" ", "")
in_stock = True if "instock" in low else False if ("outofstock" in low or "oos" == low) else None
return {
"name": title,
"images": images,
"price": price,
"currency": currency,
"availability": availability,
"in_stock": in_stock,
"properties": spec_tables(soup),
"evidence": evidence,
}
_STATE_RE = re.compile(
r"<script[^>]*id=\"__NEXT_DATA__\"[^>]*>(.*?)</script>"
r"|window\.__(?:INITIAL|PRELOADED)_STATE__\s*=\s*(\{.*?\})\s*;?\s*</script>",
re.DOTALL,
)
def embedded_state(html: str) -> Optional[Any]:
"""The page's server-rendered application state, when it embeds one."""
for m in _STATE_RE.finditer(html):
raw = m.group(1) or m.group(2)
try:
return json.loads(raw)
except (json.JSONDecodeError, TypeError):
continue
return None
def visible_text(html: str, limit: int = 6000) -> str:
soup = BeautifulSoup(html, "lxml")
for tag in soup(["script", "style", "noscript", "svg", "header", "footer", "nav"]):
tag.decompose()
return re.sub(r"\s+", " ", soup.get_text(" ", strip=True))[:limit]

View File

@@ -0,0 +1,202 @@
"""schema.org Product data embedded in a page as JSON-LD.
This is the preferred source on any page: it is what the site publishes for
search engines, so it is stable and states price, currency and availability
explicitly.
"""
from __future__ import annotations
import json
import re
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, Iterable, List, Optional
from bs4 import BeautifulSoup
_PRODUCT_TYPES = {"product", "productgroup", "productmodel", "individualproduct"}
def _types(node: dict) -> set:
t = node.get("@type")
if isinstance(t, list):
return {str(x).lower() for x in t}
return {str(t).lower()} if t else set()
def _walk(node: Any) -> Iterable[dict]:
if isinstance(node, dict):
yield node
for v in node.values():
yield from _walk(v)
elif isinstance(node, list):
for item in node:
yield from _walk(item)
def json_ld_blocks(html: str) -> List[Any]:
soup = BeautifulSoup(html, "lxml")
blocks = []
for tag in soup.find_all("script", attrs={"type": re.compile(r"ld\+json", re.I)}):
raw = (tag.string or tag.get_text() or "").strip()
if not raw:
continue
try:
blocks.append(json.loads(raw))
except json.JSONDecodeError:
# Some sites put several objects or trailing commas in one tag.
try:
blocks.append(json.loads(re.sub(r",\s*([}\]])", r"\1", raw)))
except json.JSONDecodeError:
continue
return blocks
def _dec(value: Any) -> Optional[Decimal]:
if value is None or value == "":
return None
try:
return Decimal(str(value).replace(",", "").strip())
except InvalidOperation:
return None
def _text(value: Any) -> Optional[str]:
if isinstance(value, dict):
value = value.get("name") or value.get("@value")
if isinstance(value, list):
value = value[0] if value else None
return str(value).strip() if value not in (None, "") else None
def _images(value: Any) -> List[str]:
out: List[str] = []
for v in value if isinstance(value, list) else [value]:
if isinstance(v, dict):
v = v.get("url") or v.get("contentUrl")
if isinstance(v, str) and v.startswith(("http://", "https://")):
out.append(v)
return out
def _availability(value: Any) -> tuple:
text = (_text(value) or "").lower()
if not text:
return None, None
if "instock" in text or "limitedavailability" in text or "onlineonly" in text:
return "InStock", True
if any(k in text for k in ("outofstock", "soldout", "discontinued", "preorder", "presale")):
return text.rsplit("/", 1)[-1], False
return text.rsplit("/", 1)[-1], None
def _offer(offers: Any) -> Dict[str, Any]:
"""The price/availability of the product's (lowest) offer."""
candidates = offers if isinstance(offers, list) else [offers]
best: Dict[str, Any] = {}
for o in candidates:
if not isinstance(o, dict):
continue
price = _dec(o.get("price"))
if price is None:
price = _dec(o.get("lowPrice"))
if price is None and isinstance(o.get("priceSpecification"), dict):
price = _dec(o["priceSpecification"].get("price"))
currency = _text(o.get("priceCurrency")) or (
_text(o["priceSpecification"].get("priceCurrency")) if isinstance(o.get("priceSpecification"), dict) else None
)
availability, in_stock = _availability(o.get("availability"))
entry = {"price": price, "currency": currency, "availability": availability, "in_stock": in_stock,
"raw": {k: o.get(k) for k in ("price", "lowPrice", "priceCurrency", "availability") if k in o}}
if price is not None and (not best or best.get("price") is None or price < best["price"]):
best = entry
elif not best:
best = entry
return best
MAX_REVIEWS_PER_PAGE = 30
def _review_rating(value: Any) -> Optional[Decimal]:
"""A reviewer's star rating, rescaled to 0-5 when the page uses another scale."""
if not isinstance(value, dict):
return None
rating = _dec(value.get("ratingValue"))
if rating is None:
return None
best = _dec(value.get("bestRating")) or Decimal(5)
if best <= 0:
return None
if best != 5:
rating = rating * Decimal(5) / best
if not (Decimal(0) <= rating <= Decimal(5)):
return None
return rating.quantize(Decimal("0.1"))
def _reviews(node: dict) -> List[Dict[str, Any]]:
"""Customer reviews published on the Product node (schema.org Review).
Only reviews with text are kept - a bare star with no words is not
something a reader can weigh. Nothing is paraphrased or summarised: body,
title and author are the page's own strings.
"""
raw = node.get("review") or node.get("reviews") or []
out: List[Dict[str, Any]] = []
for r in raw if isinstance(raw, list) else [raw]:
if not isinstance(r, dict):
continue
body = _text(r.get("reviewBody")) or _text(r.get("description"))
if not body:
continue
out.append({
"author": _text(r.get("author")),
"rating": _review_rating(r.get("reviewRating")),
"title": _text(r.get("name")) or _text(r.get("headline")),
"body": body[:4000],
"review_date": _text(r.get("datePublished")) or _text(r.get("dateCreated")),
})
if len(out) >= MAX_REVIEWS_PER_PAGE:
break
return out
def extract_products(html: str) -> List[Dict[str, Any]]:
"""All schema.org Product nodes on the page, flattened to plain fields."""
products: List[Dict[str, Any]] = []
for block in json_ld_blocks(html):
for node in _walk(block):
if not (_types(node) & _PRODUCT_TYPES):
continue
name = _text(node.get("name"))
if not name:
continue
offer = _offer(node.get("offers")) if node.get("offers") else {}
if not offer and isinstance(node.get("hasVariant"), list):
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
props = {}
for p in node.get("additionalProperty") or []:
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
props[str(p["name"])] = str(p["value"])
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
products.append({
"name": name,
"brand": _text(node.get("brand")),
"sku": _text(node.get("sku")) or _text(node.get("productID")),
"mpn": _text(node.get("mpn")),
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
"color": _text(node.get("color")),
"images": _images(node.get("image")),
"description": _text(node.get("description")),
"price": offer.get("price"),
"currency": offer.get("currency"),
"availability": offer.get("availability"),
"in_stock": offer.get("in_stock"),
# Sites publish 0 for "no ratings yet"; that is not a rating.
"rating": (_dec(rating.get("ratingValue")) or None),
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
"reviews": _reviews(node),
"properties": props,
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
})
return products

View File

@@ -0,0 +1,207 @@
"""Read prices and stock state out of text we did not render ourselves:
search-result titles/snippets, and visible page text.
The rules lean hard towards NOT returning a price. A snippet usually carries
several rupee amounts - the selling price, the MRP, an EMI, a bank discount, an
exchange value, "₹X off" - and taking the wrong one is worse than taking none.
An amount is only a price when nothing around it says it is something else.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
from decimal import Decimal, InvalidOperation
from typing import List, Optional
PRICE_MIN = Decimal("500")
PRICE_MAX = Decimal("1000000")
# ₹ / Rs / Rs. / INR followed by an amount with Indian (1,29,999) or western
# (129,999) grouping, or none.
_AMOUNT = r"(\d{1,3}(?:,\d{2,3})+(?:\.\d{1,2})?|\d+(?:\.\d{1,2})?)"
_MONEY_RE = re.compile(r"(?:₹|\bRs\.?|\bINR)\s?" + _AMOUNT, re.IGNORECASE)
# Words that make an amount something other than the selling price.
_REJECT_BEFORE = re.compile(
r"(?:emi|save|saving|savings|cashback|cash\s*back|exchange|bank|discount|coupon|"
r"extra|instant|up\s*to|upto|flat|off\s+upto|worth|delivery|shipping|fee|charges?|"
r"starting|starts|from|onwards|min(?:imum)?|as\s+low\s+as|down\s*payment|per\s+month)\W*$",
re.IGNORECASE,
)
_REJECT_AFTER = re.compile(
r"^\W{0,3}(?:off\b|/\s*m(?:o|onth)?\b|per\s+month|p\.?m\.?\b|a\s+month|emi\b|/-?\s*emi|"
r"cashback|discount|savings?|onwards|\+\s*shipping|delivery)",
re.IGNORECASE,
)
_MRP_BEFORE = re.compile(r"(?:m\.?\s?r\.?\s?p\.?|list\s+price|was|original\s+price)[\s:]*$", re.IGNORECASE)
_RANGE_BETWEEN = re.compile(r"^\s*(?:-|–|—|to)\s*$", re.IGNORECASE)
_OUT_OF_STOCK = re.compile(
r"\b(?:out\s+of\s+stock|currently\s+unavailable|sold\s+out|coming\s+soon|notify\s+me|"
r"temporarily\s+unavailable|not\s+available)\b",
re.IGNORECASE,
)
_IN_STOCK = re.compile(r"\b(?:in\s+stock|available\s+now|buy\s+now|add\s+to\s+cart)\b", re.IGNORECASE)
@dataclass(frozen=True)
class Amount:
value: Decimal
kind: str # price | mrp | rejected
reason: str
start: int
end: int
raw: str
def parse_amount(raw: str) -> Optional[Decimal]:
try:
value = Decimal(raw.replace(",", ""))
except InvalidOperation:
return None
return value
def find_amounts(text: str) -> List[Amount]:
"""Every rupee amount in `text`, each classified as price, mrp or rejected."""
out: List[Amount] = []
if not text:
return out
matches = list(_MONEY_RE.finditer(text))
for i, m in enumerate(matches):
value = parse_amount(m.group(1))
if value is None:
continue
before = text[max(0, m.start() - 28): m.start()]
after = text[m.end(): m.end() + 22]
kind, reason = "price", ""
if _MRP_BEFORE.search(before):
kind, reason = "mrp", "labelled MRP"
elif _REJECT_BEFORE.search(before):
kind, reason = "rejected", f"preceded by {_REJECT_BEFORE.search(before).group(0).strip()!r}"
elif _REJECT_AFTER.search(after):
kind, reason = "rejected", f"followed by {_REJECT_AFTER.search(after).group(0).strip()!r}"
# A range ("₹10,999 - ₹12,999") names no single price.
if kind == "price":
if i + 1 < len(matches) and _RANGE_BETWEEN.match(text[m.end(): matches[i + 1].start()]):
kind, reason = "rejected", "start of a price range"
elif i > 0 and _RANGE_BETWEEN.match(text[matches[i - 1].end(): m.start()]):
kind, reason = "rejected", "end of a price range"
if kind != "rejected" and not (PRICE_MIN <= value <= PRICE_MAX):
kind, reason = "rejected", "outside plausible range"
out.append(Amount(value, kind, reason, m.start(), m.end(), m.group(0)))
return out
@dataclass(frozen=True)
class PriceReading:
price: Optional[Decimal]
mrp: Optional[Decimal]
evidence: str # the exact substring the price was read from ("" if none)
def read_price(text: str) -> PriceReading:
"""The single selling price stated in `text`, or None.
If the text states two different unlabelled prices, it is ambiguous (a
listing page snippet often shows several variants) and None is returned.
"""
amounts = find_amounts(text)
prices = [a for a in amounts if a.kind == "price"]
mrps = [a for a in amounts if a.kind == "mrp"]
distinct = {a.value for a in prices}
price: Optional[Decimal] = None
evidence = ""
if len(distinct) == 1:
price = prices[0].value
evidence = prices[0].raw
mrp = mrps[0].value if mrps else None
if price is not None and mrp is not None and mrp < price:
mrp = None # an "MRP" below the selling price was misread; drop it
return PriceReading(price, mrp, evidence)
def read_stock(text: str) -> Optional[bool]:
"""True/False only when the text says so; None when it does not."""
if not text:
return None
if _OUT_OF_STOCK.search(text):
return False
if _IN_STOCK.search(text):
return True
return None
@dataclass(frozen=True)
class RatingReading:
rating: Optional[Decimal]
review_count: Optional[int]
evidence: str # the exact substring the rating was read from ("" if none)
# Only ratings the text states explicitly on a 5-point scale:
# "4.3 out of 5 stars", "Rating: 4.3/5", "Rated 4.3 / 5", "4.3★", "4.3 ★ (1,234 ratings)"
_RATING_PATTERNS = (
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*out\s+of\s*5(?:\.0)?\b(?:\s*stars?)?", re.IGNORECASE),
# "x/5" only with a rating word before it or "stars" after it - a bare
# "1/5" is as likely a sensor size or a fraction.
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\b", re.IGNORECASE),
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\s*stars?\b", re.IGNORECASE),
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*(?:★|☆|⭐)"),
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*(?:stars?|★)", re.IGNORECASE),
)
_RATING_COUNT = re.compile(
r"^[\s()\-|·,.:]*(?:stars?)?[\s()\-|·,.:]*(\d{1,3}(?:,\d{2,3})+|\d+)\s*(?:customer\s+)?(?:ratings?|reviews?|votes?)\b",
re.IGNORECASE,
)
def read_rating(text: str) -> RatingReading:
"""The product rating a search title/snippet states, or None.
Only an explicit "x out of 5" / "x/5" / "x★" statement counts; bare
numbers never do. If the text states two different ratings it is
ambiguous (several products on one results page) and None is returned.
"""
if not text:
return RatingReading(None, None, "")
found = []
for pattern in _RATING_PATTERNS:
for m in pattern.finditer(text):
try:
value = Decimal(m.group(1))
except InvalidOperation:
continue
if Decimal(0) < value <= Decimal(5):
found.append((value, m))
if not found or len({v for v, _ in found}) != 1:
return RatingReading(None, None, "")
value, m = min(found, key=lambda f: f[1].start())
count = None
tail = _RATING_COUNT.match(text[m.end(): m.end() + 40])
if tail:
count = int(tail.group(1).replace(",", ""))
evidence = text[m.start(): m.end() + (tail.end() if tail else 0)].strip()
return RatingReading(value, count, evidence)
# Titles returned by search engines carry the site name; it is not part of the
# product title.
_TITLE_SUFFIX = re.compile(
r"\s*(?:[|\-–:]\s*)?(?:buy\s+online.*|online\s+at\s+best\s+price.*|"
r"at\s+best\s+price.*|price\s+in\s+india.*|"
r"amazon\.in.*|flipkart(?:\.com)?.*|croma.*|reliance\s+digital.*|vijay\s+sales.*|"
r"tata\s+cliq.*|poorvika.*|sangeetha.*|vasanth.*|viveks.*)$",
re.IGNORECASE,
)
_TITLE_PREFIX = re.compile(r"^(?:buy\s+|amazon\.in\s*:\s*)", re.IGNORECASE)
def clean_result_title(title: str) -> str:
t = (title or "").strip()
# Engines truncate with "..." and sometimes run several results' titles
# together after it; everything past the first ellipsis is not this page.
t = re.split(r"\s*(?:\.\.\.|…)", t, maxsplit=1)[0]
t = _TITLE_PREFIX.sub("", t)
t = _TITLE_SUFFIX.sub("", t)
return t.strip(" -|:–")

View File

@@ -0,0 +1,124 @@
"""Link a listing to its canonical product (one real-world variant).
From most to least certain:
1. GTIN - same barcode -> auto
2. MPN - same manufacturer part number (laptops) -> auto
3. variant key - same brand, model, RAM and storage -> auto
4. fuzzy - model names ≥ AUTO_RATIO similar AND every
hard attribute (RAM, storage, processor) equal -> auto
≥ REVIEW_RATIO -> pending (review queue)
5. otherwise a new product is created for the variant.
A listing that states too little to identify a variant (no storage on a
phone title, for example) is stored but not linked to any product.
"""
from __future__ import annotations
from dataclasses import dataclass
from decimal import Decimal
from typing import List, Optional
from rapidfuzz import fuzz
from app.electronics.normalise.title_parser import laptop_line, processor_is_specific
AUTO_RATIO = 92
REVIEW_RATIO = 85
@dataclass
class MatchDecision:
product_id: Optional[int] # None -> create a new product
method: str
confidence: float
review_status: str
def _eq(a: Optional[Decimal], b: Optional[Decimal]) -> bool:
if a is None or b is None:
return a is None and b is None
return Decimal(a) == Decimal(b)
def _number_tokens(model_norm: str) -> set:
"""Tokens that carry a digit ("s25", "a37", "15", "2a", "8a"). Two models
whose number tokens differ are different products, however similar the
rest of the name is: Galaxy S25 vs S26, A27 vs A37, iPhone 15 vs 16."""
return {t for t in (model_norm or "").split() if any(ch.isdigit() for ch in t)}
def _lines_compatible(a: str, b: str) -> bool:
"""One model line is the other plus/minus extra words, and they agree on
every number token they both carry ("15" is not "15s", "slim 3" is not
"slim 5")."""
ta, tb = set(a.split()), set(b.split())
return bool(ta and tb) and (ta <= tb or tb <= ta)
def decide(listing, candidates: List[dict]) -> Optional[MatchDecision]:
"""`listing` is a models.Listing with variant_key/model_norm set;
`candidates` are product rows of the same brand and category."""
if not listing.variant_key:
return None
if listing.gtin:
for c in candidates:
if c.get("gtin") and c["gtin"] == listing.gtin:
return MatchDecision(c["id"], "gtin", 0.99, "auto")
if listing.model_number:
for c in candidates:
if c.get("mpn") and c["mpn"].lower() == listing.model_number.lower():
return MatchDecision(c["id"], "mpn", 0.97, "auto")
for c in candidates:
if c["variant_key"] == listing.variant_key:
return MatchDecision(c["id"], "variant_key", 0.95, "auto")
# Laptops: the same configuration (exact CPU model, RAM, storage) within a
# compatible model line - "ideapad slim 3" and "ideapad slim 3 15amn8" -
# is the same product. Two compatible candidates means the title is too
# vague to choose ("pavilion" vs "pavilion 14" and "pavilion 15"): review.
if listing.category == "laptops" and processor_is_specific(listing.processor) \
and listing.ram_gb is not None and listing.storage_gb is not None:
line = laptop_line(listing.model_norm)
same_config = [
c for c in candidates
if c.get("processor") == listing.processor
and _eq(c.get("ram_gb"), listing.ram_gb) and _eq(c.get("storage_gb"), listing.storage_gb)
and _lines_compatible(line, laptop_line(c["model_norm"]))
]
if len(same_config) == 1:
return MatchDecision(same_config[0]["id"], "variant_key", 0.85, "auto")
if len(same_config) > 1:
return MatchDecision(same_config[0]["id"], "variant_key", 0.6, "pending")
# One side does not state the RAM ("Apple iPhone 15 (128 GB)"). Same model
# and storage with exactly one candidate is the same variant; with several
# candidates it is ambiguous and goes to review.
same_model = [c for c in candidates
if c["model_norm"] == listing.model_norm and _eq(c.get("storage_gb"), listing.storage_gb)
and (c.get("ram_gb") is None) != (listing.ram_gb is None)]
if len(same_model) == 1:
return MatchDecision(same_model[0]["id"], "variant_key", 0.85, "auto")
if len(same_model) > 1:
return MatchDecision(same_model[0]["id"], "variant_key", 0.6, "pending")
best, best_score = None, 0.0
for c in candidates:
if not (_eq(c.get("storage_gb"), listing.storage_gb) and _eq(c.get("ram_gb"), listing.ram_gb)):
continue
if listing.category == "laptops" and (c.get("processor") or listing.processor) and c.get("processor") != listing.processor:
continue
if _number_tokens(c["model_norm"]) != _number_tokens(listing.model_norm):
continue
score = fuzz.token_set_ratio(c["model_norm"], listing.model_norm or "")
# token_set_ratio treats "galaxy s24" and "galaxy s24 ultra" as a
# subset match (100). Different words mean different models.
extra = set((listing.model_norm or "").split()) ^ set(c["model_norm"].split())
if extra:
score = min(score, fuzz.ratio(c["model_norm"], listing.model_norm or ""))
if score > best_score:
best, best_score = c, score
if best is not None and best_score >= AUTO_RATIO:
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.9, 2), "auto")
if best is not None and best_score >= REVIEW_RATIO:
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.8, 2), "pending")
return MatchDecision(None, "variant_key", 0.9, "auto")

View File

@@ -0,0 +1,105 @@
"""Rebuild canonical products for a category from the listings already stored.
Products and listing links are derived data: every fact lives on the listing
(title, snippet evidence, specs, URL). When the parsing or matching rules
improve, this re-runs them over the stored listings - no network requests -
and keeps each image attached to the listing it was found on.
Review decisions (approved/rejected links) are lost, because the products they
pointed at are rebuilt; uncertain matches simply come back to the queue.
"""
from __future__ import annotations
import logging
from decimal import Decimal
from typing import Dict, List
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect, transaction
from app.electronics.match.matcher import decide
from app.electronics.models import Listing
from app.electronics.normalise.title_parser import fill_from_context, parse_title, variant_key
logger = logging.getLogger(__name__)
_ORDER = {"brand_official": 0, "scraped_page": 1, "search_snippet": 2}
def _listing_from_row(row: dict, category: str) -> Listing:
parsed = parse_title(row["title"], category, expected_brand=row["brand_slug"])
snippet = ""
if row["source_type"] == "search_snippet" and " — " in row["evidence_text"]:
snippet = row["evidence_text"].split(" — ", 1)[1].split(" || ", 1)[0]
raw = row["specs_raw"] or {}
spec_texts = tuple(str(v) for k, v in raw.items() if "processor" in k.lower() or "cpu" in k.lower())
spec_texts += (str((row["specs"] or {}).get("processor") or ""),)
fill_from_context(parsed, category, snippet=snippet, spec_texts=spec_texts)
l = Listing(
site_domain=row["domain"], source_sku=row["source_sku"], source_url=row["source_url"],
source_type=row["source_type"], brand_slug=row["brand_slug"], category=category,
title=row["title"], evidence_text=row["evidence_text"], confidence=float(row["confidence"]),
parser=row["parser"], family=parsed.brand.family if parsed.brand else row["family"],
model=parsed.model, model_number=row["model_number"] or parsed.mpn,
ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb, colour=row["colour"],
gtin=row["gtin"], specs=row["specs"] or {},
)
l.model_norm, l.processor = parsed.model_norm, parsed.processor
l.variant_key = variant_key(parsed, category) if parsed.brand else None
return l
def rematch(category: str) -> Dict[str, int]:
stats: Dict[str, int] = {"listings": 0, "linked": 0, "pending": 0, "unlinked": 0, "products": 0, "images": 0}
with connect() as conn:
rows = conn.execute(
"""
SELECT l.*, b.slug AS brand_slug, s.domain
FROM elec.source_listing l
JOIN elec.brand b ON b.id = l.brand_id
JOIN elec.site s ON s.id = l.site_id
JOIN elec.category c ON c.id = l.category_id
WHERE c.slug = %s
""",
(category,),
).fetchall()
images = conn.execute(
"""
SELECT i.url, i.source_listing_id, i.source_type, i.rank FROM elec.product_image i
JOIN elec.product p ON p.id = i.product_id JOIN elec.category c ON c.id = p.category_id
WHERE c.slug = %s
""",
(category,),
).fetchall()
with transaction() as conn:
# Maps and images cascade from the products.
conn.execute(
"DELETE FROM elec.product p USING elec.category c WHERE c.id = p.category_id AND c.slug = %s",
(category,),
)
ids = repo.id_maps()
product_of_listing: Dict[int, int] = {}
rows.sort(key=lambda r: (_ORDER.get(r["source_type"], 9), r["id"]))
for row in rows:
stats["listings"] += 1
listing = _listing_from_row(row, category)
decision = decide(listing, repo.product_candidates(listing.brand_slug, category))
if decision is None:
stats["unlinked"] += 1
continue
product_id = decision.product_id or repo.create_product(listing, ids)
stats["products"] += decision.product_id is None
repo.map_listing(row["id"], product_id, decision.method, decision.confidence, decision.review_status)
product_of_listing[row["id"]] = product_id
if decision.review_status == "pending":
stats["pending"] += 1
else:
stats["linked"] += 1
repo.merge_product_specs(product_id, listing.specs, {}, listing.source_url)
for img in images:
pid = product_of_listing.get(img["source_listing_id"])
if pid is not None:
repo.add_image(pid, img["url"], img["source_listing_id"], img["source_type"], img["rank"])
stats["images"] += 1
stats.update({f"products_{k}": v for k, v in repo.refresh_verification().items()})
return stats

View File

@@ -0,0 +1,68 @@
"""The record a collector produces for one product page / search result."""
from __future__ import annotations
from dataclasses import dataclass, field
from decimal import Decimal
from typing import Any, Dict, List, Optional
SOURCE_TYPES = ("scraped_page", "search_snippet", "brand_official")
@dataclass
class Listing:
site_domain: str
source_sku: str
source_url: str
source_type: str
brand_slug: str
category: str
title: str
evidence_text: str
confidence: float
parser: str
family: Optional[str] = None
model: Optional[str] = None
model_number: Optional[str] = None
ram_gb: Optional[Decimal] = None
storage_gb: Optional[Decimal] = None
colour: Optional[str] = None
price: Optional[Decimal] = None
mrp: Optional[Decimal] = None
availability: Optional[str] = None
in_stock: Optional[bool] = None
pincode: Optional[str] = None
pincode_applied: bool = False
rating: Optional[Decimal] = None
review_count: Optional[int] = None
# Customer reviews the page itself publishes (schema.org Review); stored
# in elec.listing_review, not on the listing row.
reviews: List[Dict[str, Any]] = field(default_factory=list)
gtin: Optional[str] = None
image_urls: List[str] = field(default_factory=list)
specs_raw: Dict[str, Any] = field(default_factory=dict)
specs: Dict[str, Any] = field(default_factory=dict)
spec_sources: Dict[str, str] = field(default_factory=dict)
search_query: Optional[str] = None
content_hash: Optional[str] = None
# Not stored on the listing; used for matching.
variant_key: Optional[str] = None
model_norm: Optional[str] = None
processor: Optional[str] = None
def validate(self) -> None:
"""The anti-fabrication contract, checked before anything is written."""
if self.source_type not in SOURCE_TYPES:
raise ValueError(f"bad source_type {self.source_type!r}")
if not self.source_url.startswith(("http://", "https://")):
raise ValueError("listing without a real source URL")
if not self.evidence_text.strip():
raise ValueError("listing without evidence text")
if self.price is not None:
if not (Decimal(500) <= self.price <= Decimal(1000000)):
raise ValueError(f"implausible price {self.price}")
if self.mrp is not None and self.price is not None and self.mrp < self.price:
self.mrp = None
if self.pincode_applied and not self.pincode:
raise ValueError("pincode_applied without a pincode")
if not 0 <= self.confidence <= 1:
raise ValueError("confidence out of range")

View File

View File

@@ -0,0 +1,61 @@
"""Per-host circuit breaker.
One 403, 429, 503 or CAPTCHA page opens the breaker for that host for
ELEC_BREAKER_COOLDOWN_HOURS. While it is open the host is not requested at all
and its products are collected from web search results instead. There is no
retry-with-a-different-identity: a block is an answer.
"""
from __future__ import annotations
import threading
import time
from typing import Callable, Dict, Optional, Tuple
from app.infrastructure.settings import ELEC_BREAKER_COOLDOWN_HOURS
class CircuitBreaker:
def __init__(
self,
cooldown_seconds: float = ELEC_BREAKER_COOLDOWN_HOURS * 3600,
on_trip: Optional[Callable[[str, str, float], None]] = None,
clock: Callable[[], float] = time.time,
) -> None:
self.cooldown = cooldown_seconds
self.on_trip = on_trip
self._clock = clock
self._open: Dict[str, Tuple[float, str]] = {}
self._lock = threading.Lock()
@staticmethod
def _key(host: str) -> str:
host = host.lower()
return host[4:] if host.startswith("www.") else host
def preload(self, host: str, until_epoch: float, reason: str) -> None:
"""Restore a breaker that was opened in an earlier run (elec.site)."""
if until_epoch > self._clock():
with self._lock:
self._open[self._key(host)] = (until_epoch, reason)
def trip(self, host: str, reason: str) -> None:
until = self._clock() + self.cooldown
with self._lock:
self._open[self._key(host)] = (until, reason)
if self.on_trip:
self.on_trip(self._key(host), reason, until)
def is_open(self, host: str) -> bool:
key = self._key(host)
with self._lock:
entry = self._open.get(key)
if not entry:
return False
if entry[0] <= self._clock():
del self._open[key]
return False
return True
def reason(self, host: str) -> Optional[str]:
entry = self._open.get(self._key(host))
return entry[1] if entry else None

View File

@@ -0,0 +1,223 @@
"""The only way this project fetches a retail or brand web page.
What it guarantees, for every request:
* robots.txt is consulted first (protego). If robots.txt cannot be read
because the server errors or blocks it, the site is treated as disallowed.
* at least ELEC_SITE_MIN_INTERVAL_SECONDS between requests to one host.
* an honest User-Agent naming the project and a contact address.
* no JavaScript, no cookies kept between requests, no proxies, no retries on
403/429 - a block is respected, not worked around.
* a size cap on the response body.
* a circuit breaker: a 403/429/CAPTCHA response opens it for the host, and
every later request to that host is refused until the cooldown passes.
* every request is reported to `on_fetch` (the fetch_log table).
"""
from __future__ import annotations
import logging
import re
import threading
import time
from dataclasses import dataclass
from typing import Callable, Dict, Optional, Tuple
from urllib.parse import urlparse
import httpx
from protego import Protego
from app.electronics.net.breaker import CircuitBreaker
from app.infrastructure.settings import (
ELEC_MAX_PAGE_BYTES,
ELEC_SITE_MIN_INTERVAL_SECONDS,
REQUEST_TIMEOUT_SECONDS,
USER_AGENT,
)
logger = logging.getLogger(__name__)
ROBOTS_TTL_SECONDS = 24 * 3600
# Pages that are a bot check rather than content. Matched on the first 20 KB.
_CAPTCHA_MARKERS = re.compile(
r"captcha|robot\s*check|are\s+you\s+a\s+robot|verify\s+you\s+are\s+human|"
r"/errors/validatecaptcha|px-captcha|cf-challenge|challenge-platform|access\s+denied|"
r"unusual\s+traffic|request\s+blocked|bot\s+detection|akamai.*reference",
re.IGNORECASE,
)
@dataclass
class FetchResult:
url: str
final_url: str
status: Optional[int]
text: str
outcome: str # ok | robots_disallowed | blocked | captcha | breaker_open | http_error | network_error | too_large | not_html
robots_allowed: Optional[bool]
bytes: int = 0
@property
def ok(self) -> bool:
return self.outcome == "ok"
class PoliteClient:
def __init__(
self,
*,
min_interval: float = ELEC_SITE_MIN_INTERVAL_SECONDS,
breaker: Optional[CircuitBreaker] = None,
on_fetch: Optional[Callable[[FetchResult, str], None]] = None,
transport: Optional[httpx.BaseTransport] = None,
sleep: Callable[[float], None] = time.sleep,
clock: Callable[[], float] = time.monotonic,
) -> None:
self.min_interval = min_interval
self.breaker = breaker or CircuitBreaker()
self.on_fetch = on_fetch
self._sleep = sleep
self._clock = clock
self._last: Dict[str, float] = {}
self._locks: Dict[str, threading.Lock] = {}
self._robots: Dict[str, Tuple[float, Optional[Protego], bool]] = {}
self._guard = threading.Lock()
self._client = httpx.Client(
headers={
"User-Agent": USER_AGENT,
"Accept": "text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.5",
"Accept-Language": "en-IN,en;q=0.9",
},
follow_redirects=True,
timeout=REQUEST_TIMEOUT_SECONDS,
transport=transport,
)
def close(self) -> None:
self._client.close()
def __enter__(self) -> "PoliteClient":
return self
def __exit__(self, *exc) -> None:
self.close()
# -- pacing --------------------------------------------------------------
def _host_lock(self, host: str) -> threading.Lock:
with self._guard:
return self._locks.setdefault(host, threading.Lock())
def _wait_turn(self, host: str) -> None:
last = self._last.get(host)
if last is not None:
gap = self.min_interval - (self._clock() - last)
if gap > 0:
self._sleep(gap)
self._last[host] = self._clock()
# -- robots.txt ----------------------------------------------------------
def _robots_for(self, scheme: str, host: str) -> Tuple[Optional[Protego], bool]:
"""(parser, reachable). parser None + reachable True = no robots.txt
(everything allowed); reachable False = could not read it (deny)."""
cached = self._robots.get(host)
if cached and self._clock() - cached[0] < ROBOTS_TTL_SECONDS:
return cached[1], cached[2]
url = f"{scheme}://{host}/robots.txt"
parser: Optional[Protego] = None
reachable = False
self._wait_turn(host)
try:
resp = self._client.get(url)
if resp.status_code == 200:
parser, reachable = Protego.parse(resp.text), True
elif resp.status_code in (404, 410):
parser, reachable = None, True
else:
reachable = False
if resp.status_code in (403, 429):
self.breaker.trip(host, f"robots.txt returned HTTP {resp.status_code}")
except httpx.HTTPError as exc:
logger.info("robots.txt unreachable for %s: %s", host, exc)
self._robots[host] = (self._clock(), parser, reachable)
return parser, reachable
def robots_allowed(self, url: str) -> bool:
p = urlparse(url)
parser, reachable = self._robots_for(p.scheme or "https", p.netloc.lower())
if not reachable:
return False
return True if parser is None else bool(parser.can_fetch(url, USER_AGENT))
# -- fetch ---------------------------------------------------------------
def _report(self, result: FetchResult) -> FetchResult:
if self.on_fetch:
try:
self.on_fetch(result, urlparse(result.url).netloc.lower())
except Exception as exc: # noqa: BLE001 - logging must never break a crawl
logger.debug("fetch log failed: %s", exc)
return result
def get(self, url: str, *, check_robots: bool = True, accept_non_html: bool = False) -> FetchResult:
host = urlparse(url).netloc.lower()
if self.breaker.is_open(host):
return FetchResult(url, url, None, "", "breaker_open", None)
with self._host_lock(host):
allowed: Optional[bool] = None
if check_robots:
allowed = self.robots_allowed(url)
if not allowed:
return self._report(FetchResult(url, url, None, "", "robots_disallowed", False))
self._wait_turn(host)
try:
with self._client.stream("GET", url) as resp:
status = resp.status_code
final = str(resp.url)
ctype = resp.headers.get("content-type", "").lower()
body = bytearray()
too_large = False
for chunk in resp.iter_bytes():
body.extend(chunk)
if len(body) > ELEC_MAX_PAGE_BYTES:
too_large = True
break
encoding = resp.encoding or "utf-8"
except httpx.HTTPError as exc:
logger.info("fetch failed %s: %s", url, exc)
return self._report(FetchResult(url, url, None, "", "network_error", allowed))
text = bytes(body).decode(encoding, errors="replace") if body else ""
n = len(body)
if status in (403, 429, 503) or (status == 200 and _CAPTCHA_MARKERS.search(text[:20000]) and len(text) < 60000):
outcome = "captcha" if status == 200 or _CAPTCHA_MARKERS.search(text[:20000]) else "blocked"
self.breaker.trip(host, f"HTTP {status} ({outcome})")
return self._report(FetchResult(url, final, status, "", outcome, allowed, n))
if status != 200:
return self._report(FetchResult(url, final, status, "", "http_error", allowed, n))
if too_large:
return self._report(FetchResult(url, final, status, "", "too_large", allowed, n))
if not accept_non_html and "html" not in ctype and "json" not in ctype:
return self._report(FetchResult(url, final, status, "", "not_html", allowed, n))
return self._report(FetchResult(url, final, status, text, "ok", allowed, n))
def check_image(self, url: str, min_bytes: int) -> bool:
"""One ranged GET to confirm a URL serves a real image. Paced per host
like any request; robots.txt is not consulted because this fetches a
single file the product page itself references, as a browser would."""
host = urlparse(url).netloc.lower()
if not url.startswith(("http://", "https://")) or self.breaker.is_open(host):
return False
with self._host_lock(host):
self._wait_turn(host)
try:
with self._client.stream("GET", url, headers={"Accept": "image/*", "Range": f"bytes=0-{min_bytes * 4}"}) as resp:
if resp.status_code not in (200, 206):
return False
if not resp.headers.get("content-type", "").lower().startswith("image/"):
return False
got = 0
for chunk in resp.iter_bytes():
got += len(chunk)
if got >= min_bytes:
return True
return got >= min_bytes
except httpx.HTTPError:
return False

View File

@@ -0,0 +1,86 @@
"""Resolve the brand of a product title against the closed allow-list."""
from __future__ import annotations
import re
from dataclasses import dataclass
from functools import lru_cache
from typing import List, Optional, Tuple
from app.electronics.reference import load_reference
@dataclass(frozen=True)
class BrandMatch:
brand_slug: str
brand_name: str
family: Optional[str] # sub-brand (Redmi, iQOO, Pixel...) when the title used one
matched: str # the alias text found in the title
@lru_cache(maxsize=1)
def _alias_table() -> List[Tuple[str, str, bool]]:
"""(alias, brand_slug, is_sub_brand), longest alias first."""
ref = load_reference()
rows: List[Tuple[str, str, bool]] = []
for b in ref.brands.values():
for a in b.aliases:
rows.append((a, b.slug, False))
for s in b.sub_brands:
rows.append((s, b.slug, True))
rows.sort(key=lambda r: -len(r[0]))
return rows
def resolve_brand(title: str, *, expected: Optional[str] = None) -> Optional[BrandMatch]:
"""The allow-listed brand a title starts with (or names within its first
few words), or None. `expected` restricts the match to one brand slug.
Only the start of the title is considered: "Case for Samsung Galaxy S24"
is an accessory, not a Samsung phone.
"""
if not title:
return None
ref = load_reference()
head = " ".join(re.findall(r"[a-z0-9+]+", title.lower())[:3])
for alias, slug, is_sub in _alias_table():
if expected and slug != expected:
continue
pattern = r"(?:^|\s)" + re.escape(alias) + r"(?:\s|$)"
m = re.search(pattern, head)
if not m:
continue
# The brand/sub-brand must be the first or second word ("Apple iPhone",
# "Samsung Galaxy", "Xiaomi Redmi Note") - not buried later.
if len(head[: m.start()].split()) > 1:
continue
# "Google Pixel 8", "Xiaomi Redmi Note 13": the parent brand matched,
# but the family is the sub-brand that follows it.
sub = alias if is_sub else next(
(s for s in ref.brands[slug].sub_brands if re.search(r"(?:^|\s)" + re.escape(s) + r"(?:\s|$)", head)),
None,
)
return BrandMatch(slug, ref.brands[slug].name, _family_casing(sub) if sub else None, alias)
return None
_CASING = {"iphone": "iPhone", "iqoo": "iQOO", "macbook": "MacBook", "rog": "ROG", "tuf": "TUF",
"cmf": "CMF", "loq": "LOQ", "poco": "POCO", "mi": "Mi", "xps": "XPS", "thinkpad": "ThinkPad",
"ideapad": "IdeaPad", "thinkbook": "ThinkBook", "vivobook": "Vivobook", "zenbook": "Zenbook"}
def _family_casing(sub: str) -> str:
return _CASING.get(sub, sub.title())
# Words that mark an accessory or a non-product page, not a device.
_NOT_A_DEVICE = re.compile(
r"\b(?:case|cover|back\s+cover|tempered|screen\s+guard|protector|charger|adapter|cable|"
r"skin|sleeve|bag|backpack|stand|holder|refurbished|renewed|pre-?owned|used|"
r"compare|vs\.?|versus|review|specifications?\s+and|price\s+list|best\s+\w+\s+under|"
r"top\s+\d+|all\s+models)\b",
re.IGNORECASE,
)
def looks_like_device_title(title: str) -> bool:
return bool(title) and not _NOT_A_DEVICE.search(title)

View File

@@ -0,0 +1,55 @@
"""Is a value actually stated in the text it supposedly came from?
Every value the LLM returns passes through value_in_source() against the exact
text the model was shown. Anything that cannot be found there is discarded,
which is what stops a small model's guess becoming a stored fact.
"""
from __future__ import annotations
import re
from decimal import Decimal, InvalidOperation
from typing import Union
_WS = re.compile(r"\s+")
def _norm_text(text: str) -> str:
text = text.lower().replace(" ", " ")
text = re.sub(r"[^\w.+ ]+", " ", text)
text = re.sub(r"(?<!\d)\.|\.(?!\d)", " ", text) # sentence full stops, not decimals
# "5000mAh" and "5000 mAh" must compare equal.
text = re.sub(r"(?<=\d)(?=[a-z])|(?<=[a-z])(?=\d)", " ", text)
return _WS.sub(" ", text).strip()
def _numbers_in(text: str) -> set:
found = set()
for raw in re.findall(r"\d[\d,]*(?:\.\d+)?", text):
try:
found.add(Decimal(raw.replace(",", "")).normalize())
except InvalidOperation:
continue
return found
def value_in_source(value: Union[str, int, float, Decimal, None], source: str) -> bool:
if value is None or not source:
return False
if isinstance(value, bool):
return False
if isinstance(value, (int, float, Decimal)):
try:
return Decimal(str(value)).normalize() in _numbers_in(source)
except InvalidOperation:
return False
text = str(value).strip()
if not text:
return False
# A string with a number in it ("5000 mAh", "Snapdragon 8 Gen 3") must have
# every one of its numbers in the source, and its words too.
nums = _numbers_in(text)
if nums and not nums <= _numbers_in(source):
return False
words = [w for w in _norm_text(text).split() if not re.fullmatch(r"[\d.,]+", w)]
hay = f" {_norm_text(source)} "
return all(f" {w} " in hay for w in words) if words else bool(nums)

View File

@@ -0,0 +1,71 @@
"""Fill MISSING spec keys from page text with the local LLM - and keep only
what the text actually says.
The model sees one block of text that we fetched (a spec section or a
description) and is asked to copy values out of it. Every value it returns is:
1. checked by grounding.value_in_source() against that same text, and
2. normalised by spec_normaliser (units, plausible ranges).
Anything failing either step is dropped. Prices, product names and images are
never asked of the model.
"""
from __future__ import annotations
import json
import logging
from typing import Any, Dict, Iterable, Tuple
from app.electronics.normalise.grounding import value_in_source
from app.electronics.normalise.spec_normaliser import normalise_value
from app.infrastructure.settings import ELEC_USE_LLM
logger = logging.getLogger(__name__)
MAX_SOURCE_CHARS = 3500
SYSTEM_PROMPT = (
"You copy product specifications out of the text you are given. "
"Rules: use ONLY the given text; copy each value exactly as written, including its unit; "
"if the text does not state a value, use null; never guess, estimate or use outside knowledge. "
"Reply with one JSON object whose keys are exactly the requested keys."
)
def fill_missing(
category: str,
source_text: str,
missing_keys: Iterable[str],
*,
generate=None,
) -> Tuple[Dict[str, Any], Dict[str, str]]:
"""(specs, sources) for whichever of `missing_keys` the text states."""
keys = [k for k in missing_keys if k != "colour"]
text = (source_text or "").strip()[:MAX_SOURCE_CHARS]
if not keys or not text or not ELEC_USE_LLM:
return {}, {}
if generate is None:
from app.services.ollama_service import generate_json as generate
prompt = (
f"Requested keys: {json.dumps(keys)}\n\n"
f"Text:\n\"\"\"\n{text}\n\"\"\"\n\n"
"JSON:"
)
reply = generate(SYSTEM_PROMPT, prompt)
if not isinstance(reply, dict):
return {}, {}
specs: Dict[str, Any] = {}
sources: Dict[str, str] = {}
for key in keys:
raw = reply.get(key)
if raw is None or isinstance(raw, (dict, list, bool)):
continue
if not value_in_source(raw, text):
logger.debug("LLM value %r for %s not found in source text; dropped", raw, key)
continue
value = normalise_value(category, key, raw)
if value is None:
continue
specs[key] = value
sources[key] = f"llm-extracted: {str(raw)[:80]}"
return specs, sources

View File

@@ -0,0 +1,101 @@
"""Map raw spec labels/values from a page to canonical keys and units.
Deterministic and table-driven (reference/spec_keys.yaml). A value that cannot
be parsed, or lands outside the plausible range for its key, is dropped - never
estimated.
"""
from __future__ import annotations
import re
from decimal import Decimal, InvalidOperation
from functools import lru_cache
from typing import Any, Dict, Optional, Tuple
from app.electronics.reference import load_reference
def _label(text: str) -> str:
return re.sub(r"[^a-z0-9]+", " ", str(text).lower()).strip()
@lru_cache(maxsize=None)
def _synonyms(category: str) -> Dict[str, str]:
table: Dict[str, str] = {}
for key, spec in load_reference().spec_keys.get(category, {}).items():
for syn in [key.replace("_", " "), *spec.get("synonyms", [])]:
table.setdefault(_label(syn), key)
return table
def canonical_key(category: str, label: str) -> Optional[str]:
return _synonyms(category).get(_label(label))
# unit -> (regex for the unit in text, factor into the canonical unit)
_UNIT_PATTERNS = {
"GB": [(r"tb", Decimal(1024)), (r"gb", Decimal(1)), (r"mb", Decimal(1) / 1024)],
"inch": [(r"(?:inch(?:es)?|in\b|\"|”)", Decimal(1)), (r"cm", Decimal(1) / Decimal("2.54"))],
"Hz": [(r"hz", Decimal(1))],
"MP": [(r"mp|megapixel", Decimal(1))],
"mAh": [(r"mah", Decimal(1))],
"kg": [(r"kg|kilogram", Decimal(1)), (r"(?<![k])g\b|grams?", Decimal("0.001"))],
"Wh": [(r"wh|watt\s*hours?", Decimal(1))],
}
def _to_number(value: str, unit: str) -> Optional[Decimal]:
text = str(value).lower().replace(",", "")
patterns = _UNIT_PATTERNS.get(unit, [])
# Prefer an amount written in the canonical unit ("39.62 cm (15.6 inch)" -> 15.6).
for unit_re, factor in patterns:
m = re.search(r"(\d+(?:\.\d+)?)\s*(?:" + unit_re + r")", text)
if m:
try:
return (Decimal(m.group(1)) * factor).quantize(Decimal("0.01")).normalize()
except InvalidOperation:
return None
# A bare number is accepted only when nothing else is in the value.
m = re.fullmatch(r"\s*(\d+(?:\.\d+)?)\s*", text)
if m:
return Decimal(m.group(1)).normalize()
return None
def normalise_value(category: str, key: str, value: Any) -> Optional[Any]:
spec = load_reference().spec_keys.get(category, {}).get(key)
if spec is None or value is None:
return None
text = str(value).strip()
if not text or text.lower() in {"na", "n/a", "-", "none", "not applicable", "no"}:
return None
kind = spec.get("type")
if kind == "number":
num = _to_number(text, spec.get("unit", ""))
if num is None:
return None
lo, hi = spec.get("range", [None, None])
if (lo is not None and num < Decimal(str(lo))) or (hi is not None and num > Decimal(str(hi))):
return None
return float(num) if num != num.to_integral() else int(num)
if kind == "enum":
low = text.lower()
for canon, words in spec.get("values", {}).items():
if any(re.search(r"\b" + re.escape(w) + r"\b", low) for w in words):
return canon
return None
return re.sub(r"\s+", " ", text)[:120]
def normalise_specs(category: str, raw: Dict[str, Any]) -> Tuple[Dict[str, Any], Dict[str, str]]:
"""(specs, sources): canonical key -> value, and key -> the raw label it came from."""
specs: Dict[str, Any] = {}
sources: Dict[str, str] = {}
for label, value in (raw or {}).items():
key = canonical_key(category, label)
if not key or key in specs:
continue
norm = normalise_value(category, key, value)
if norm is not None:
specs[key] = norm
sources[key] = f"{label}: {value}"[:200]
return specs, sources

View File

@@ -0,0 +1,374 @@
"""Split a retail product title into model, variant and a matching key.
Everything returned is read from the title text; a value the title does not
state is None. Titles differ a lot between sites:
Samsung Galaxy S24 5G (Onyx Black, 8GB RAM, 256GB Storage) Amazon
SAMSUNG Galaxy S24 5G (Onyx Black, 256 GB) (8 GB RAM) Flipkart
Samsung Galaxy S24 5G (8GB RAM, 256GB, Onyx Black) Croma
Redmi Note 13 Pro 5G (8GB + 256GB)
Apple iPhone 15 (128 GB) - Black
HP 15s, 13th Gen Intel Core i5-1334U, 16GB DDR4, 512GB SSD, ... fd0112TU
so the model is taken from the text before the first bracket/comma, and RAM /
storage / colour from anywhere in the title.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from decimal import Decimal
from typing import List, Optional
from app.electronics.normalise.brand_alias import BrandMatch, resolve_brand
_NUM = r"(\d+(?:\.\d+)?)"
# "8GB RAM", "8 GB LPDDR5X RAM", "RAM 8GB", "16GB DDR4" (laptops)
_RAM_RES = [
re.compile(_NUM + r"\s*GB\s*(?:LP)?(?:DDR\s?\d\w?\s*)?RAM\b", re.IGNORECASE),
re.compile(r"\bRAM\s*[:\-]?\s*" + _NUM + r"\s*GB", re.IGNORECASE),
re.compile(_NUM + r"\s*GB\s*(?:LP)?DDR\s?\d", re.IGNORECASE),
re.compile(_NUM + r"\s*GB\s*(?:unified\s+memory|memory)\b", re.IGNORECASE),
]
# "8GB + 256GB", "8/256", "8GB/256GB", "12+512GB"
_PAIR_RE = re.compile(r"(?<![\d.])(\d{1,2})\s*(?:GB)?\s*[+/]\s*(\d{2,4}|1|2)\s*(GB|TB)?\b", re.IGNORECASE)
# explicit storage: "256GB Storage", "512GB SSD", "1TB", "256 GB ROM"
_STORAGE_LABELLED = re.compile(
_NUM + r"\s*(GB|TB)\s*(?:SSD|ROM|storage|internal(?:\s+storage)?|HDD|eMMC|UFS|NVMe|PCIe)\b", re.IGNORECASE
)
_SIZE_ANY = re.compile(r"(?<![\d.])" + _NUM + r"\s*(GB|TB)\b", re.IGNORECASE)
# Order matters: the first pattern that matches wins. AMD comes before the
# Intel "Core N" pattern, because retail titles write core counts as words
# ("Ryzen 3 Quad Core 7320U"), and "Core 7320U" must not read as Intel.
_CORE_COUNT = r"(?:(?:Dual|Quad|Hexa|Octa|Six|Eight)\s+Core\s+)?"
_PROCESSOR_RES = [
re.compile(r"\b(?:AMD\s+)?Ryzen\s+R?(\d)\s*(?:Pro\s+)?" + _CORE_COUNT + r"[- ]?(\d{4}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"\b(?:AMD\s+)?(Athlon)\s+(?:Silver\s+|Gold\s+)?" + _CORE_COUNT + r"(\d{4}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+Ultra\s+(\d)\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+(?:\(?\d+(?:th|nd|rd|st)\s+Gen\)?\s+)?(i[3579])\s*[- ]?\s*(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?Core\s+(i[3579])\s+\d+(?:th|nd|rd|st)\s+Gen\s+(\d{4,5}[A-Z]{0,3})\b", re.IGNORECASE),
re.compile(r"(?<!Dual )(?<!Quad )(?<!Hexa )(?<!Octa )(?<!Six )(?<!Eight )\b(?:Intel\s+)?Core\s+([3579])\s*[- ]?\s*(\d{3}[A-Z]{0,2})\b", re.IGNORECASE),
re.compile(r"\bApple\s+(M[1-9])(?:\s+(Pro|Max|Ultra))?\b", re.IGNORECASE),
re.compile(r"\b(M[1-9])\s*(Pro|Max|Ultra)?\s+chip\b", re.IGNORECASE),
re.compile(r"\bSnapdragon\s+(X\d?)\s*(Elite|Plus)?\s*(X\d{1,2}-\d{3})?", re.IGNORECASE),
re.compile(r"\b(?:Intel\s+)?(Celeron|Pentium(?:\s+Silver|\s+Gold)?)\s+(N?\d{3,5}[A-Z]?)\b", re.IGNORECASE),
re.compile(r"\bMediaTek\s+(Kompanio\s+\d{3,4}|MT\d{4})\b", re.IGNORECASE),
]
_COLOUR_WORDS = re.compile(
r"\b(black|white|blue|green|red|grey|gray|silver|gold|purple|violet|pink|yellow|orange|"
r"cream|titanium|graphite|midnight|starlight|mint|lavender|bronze|copper|beige|teal|"
r"navy|jade|coral|onyx|marble|obsidian|porcelain|hazel|aqua|lime|sand|charcoal)\b",
re.IGNORECASE,
)
# Tokens that describe the device class or connectivity, not the model.
_MODEL_NOISE = re.compile(
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|mobile|phone|dual\s+sim|"
r"laptop|notebook|thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest|with\b.*$)",
re.IGNORECASE,
)
_SPEC_TOKEN = re.compile(
r"^(?:\d+(?:\.\d+)?\s*(?:gb|tb|mp|mah|hz|inch|inches|cm|w|kg|g)|\d+(?:th|nd|rd|st)|gen|ddr\d?\w*|"
r"lpddr\d\w*|ssd|hdd|fhd|qhd|uhd|oled|ips|win|windows|20\d\d)$",
re.IGNORECASE,
)
@dataclass
class ParsedTitle:
title: str
brand: Optional[BrandMatch]
model: Optional[str] = None # "Galaxy S24", "Redmi Note 13 Pro", "15s"
model_norm: Optional[str] = None # "galaxy s24", matching form
ram_gb: Optional[Decimal] = None
storage_gb: Optional[Decimal] = None
colour: Optional[str] = None
processor: Optional[str] = None # "i5-1334u", "ryzen 5 7530u", "m3"
mpn: Optional[str] = None # laptop part number when stated
network: Optional[str] = None
notes: List[str] = field(default_factory=list)
def _dec(value: str, unit: str = "GB") -> Decimal:
d = Decimal(value)
if unit.upper() == "TB":
d = d * 1024
return d.normalize() if d == d.to_integral() else d
def _parse_ram_storage(text: str, category: str):
ram = storage = None
for rx in _RAM_RES:
m = rx.search(text)
if m:
ram = _dec(m.group(1))
break
m = _PAIR_RE.search(text)
if m:
a, b, unit = m.group(1), m.group(2), (m.group(3) or "GB")
pair_ram, pair_storage = _dec(a), _dec(b, unit)
# "Core Ultra 5/ 16GB RAM/ 512GB" is not a 5 GB / 16 GB pair: a real
# pair has device-sized storage.
min_pair_storage = Decimal(16) if category == "mobiles" else Decimal(64)
if pair_storage > pair_ram and pair_storage >= min_pair_storage:
ram = ram if ram is not None else pair_ram
storage = pair_storage
if storage is None:
m = _STORAGE_LABELLED.search(text)
if m:
storage = _dec(m.group(1), m.group(2))
if storage is None:
# Unlabelled sizes: the storage is the largest one that is not the RAM.
min_storage = Decimal(16) if category == "mobiles" else Decimal(32)
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
candidates = [s for s in sizes if s != ram and s >= min_storage]
if candidates:
storage = max(candidates)
if ram is None:
# "(8 GB RAM)" handled above; an unlabelled small size next to a larger
# one ("8GB 256GB") is the RAM.
sizes = [_dec(v, u) for v, u in _SIZE_ANY.findall(text)]
small = [s for s in sizes if s <= (24 if category == "mobiles" else 64) and (storage is None or s < storage)]
if len(set(small)) == 1 and storage is not None:
ram = small[0]
plausible_ram = Decimal(32) if category == "mobiles" else Decimal(128)
if ram is not None and not (Decimal(1) <= ram <= plausible_ram):
ram = None
return ram, storage
def parse_processor(text: str) -> Optional[str]:
"""Normalised CPU name ("i5-1334u", "ryzen 3 7320u", "core ultra 5 125h",
"core 5 120u", "athlon 7120u", "m2"), or None."""
for rx in _PROCESSOR_RES:
m = rx.search(text or "")
if not m:
continue
parts = [g for g in m.groups() if g]
matched = m.group(0).lower()
if "ryzen" in matched:
return f"ryzen {parts[0]} {parts[1]}".lower()
if "athlon" in matched:
return f"athlon {parts[1]}".lower()
if "ultra" in matched and "core" in matched:
return f"core ultra {parts[0]} {parts[1]}".lower()
if "core" in matched and parts[0].lower().startswith("i"):
return f"{parts[0]}-{parts[1]}".lower()
if "core" in matched:
return f"core {parts[0]} {parts[1]}".lower()
return " ".join(parts).lower()
return None
_parse_processor = parse_processor
def processor_is_specific(processor: Optional[str]) -> bool:
"""True for a CPU named down to its model number ("i5-1334u"), which
together with brand, model line, RAM and storage identifies a laptop
configuration. "m2" (Apple) also counts."""
if not processor:
return False
return bool(re.search(r"\d{3,}", processor)) or bool(re.fullmatch(r"m[1-9](?: (?:pro|max|ultra))?", processor))
_MPN_RE = re.compile(
r"(?<![\w-])([A-Z0-9]{2,8}-[A-Z0-9]{2,10}(?:-[A-Z0-9]{1,6})?|"
r"[A-Z0-9]{2,6}[A-Z]{0,4}\d{2,6}[A-Z]{1,4}\d{0,4}[A-Z]{0,3}|\d{2}[A-Z]{2}\d{3,}[A-Z0-9]{2,})(?![\w-])",
re.IGNORECASE,
)
def _parse_mpn(text: str, processor: Optional[str]) -> Optional[str]:
"""A manufacturer part number such as 82XV00BHIN or fd0112TU, when the
title states one (laptops). Tokens that are specs or CPU names are not."""
candidates = []
for m in _MPN_RE.finditer(text):
tok = m.group(1)
low = tok.lower()
if len(tok) < 6 or len(tok) > 20:
continue
if sum(c.isdigit() for c in tok) < 2 or sum(c.isalpha() for c in tok) < 2:
continue
if _SPEC_TOKEN.match(low) or re.match(r"^(?:i[3579]|m[1-9]|rtx|gtx|rx|ddr|lpddr)", low):
continue
if processor and low in processor.replace("-", " ").split() + [processor.replace(" ", "")]:
continue
if re.search(r"\d+(?:gb|tb|mp|mah|hz|w)$", low):
continue
if re.search(r"-(?:core|inch|cell|bit|gen|thread)s?$|^\d+-", low) and not re.search(r"[a-z]\d", low.split("-")[-1]):
continue # "10-Core", "15-inch", "3-Cell" describe hardware, not a part number
candidates.append(tok)
return candidates[-1].upper() if candidates else None
def _parse_colour(title: str) -> Optional[str]:
# Inside brackets first: "(Onyx Black, 8GB RAM, 256GB Storage)"
for group in re.findall(r"\(([^()]*)\)", title):
for part in re.split(r"[,|/]", group):
part = part.strip()
if part and not re.search(r"\d", part) and _COLOUR_WORDS.search(part):
return part.title()
# Trailing "- Black"
m = re.search(r"[-–|,]\s*([A-Za-z][A-Za-z ]{2,30})\s*$", title)
if m and _COLOUR_WORDS.search(m.group(1)) and not re.search(r"\d", m.group(1)):
return m.group(1).strip().title()
return None
def normalise_model(model: str) -> str:
text = model.lower()
text = re.sub(r"[()\[\],|]", " ", text)
text = _MODEL_NOISE.sub(" ", text)
text = re.sub(r"\+", " plus ", text)
text = re.sub(r"[^a-z0-9 ]+", " ", text)
tokens = [t for t in text.split() if not _SPEC_TOKEN.match(t)]
return " ".join(tokens)
_LAPTOP_SPEC_START = re.compile(
r"\b(?:intel|amd|apple\s+m[1-9]|m[1-9]\s+chip|core\s+(?:i[3579]|ultra)|ryzen|snapdragon|celeron|pentium|"
r"mediatek|\d+(?:th|nd|rd|st)\s+gen|\d+(?:\.\d+)?\s*(?:-|\s)?(?:inch|cm|\"))",
re.IGNORECASE,
)
_DISPLAY_NOISE = re.compile(
r"\b(?:5g|4g|lte|smartphone|smart\s+phone|mobile\s+phone|dual\s+sim|laptop|notebook|"
r"thin\s+and\s+light|thin\s+&\s+light|gaming|new|latest)\b",
re.IGNORECASE,
)
def _model_from_title(title: str, brand: Optional[BrandMatch], category: str, colour: Optional[str],
mpn: Optional[str] = None):
"""(display model, matching model_norm) from the head of the title."""
head = re.sub(r"^\s*buy\s+", "", title, flags=re.IGNORECASE)
if mpn:
# A part number is not the model name. HP writes the line into it
# ("15-fc0500AU" is an HP 15), so that prefix is kept.
prefix = mpn.split("-", 1)[0] if "-" in mpn and len(mpn.split("-", 1)[0]) <= 4 else ""
head = re.sub(re.escape(mpn), f" {prefix} ", head, flags=re.IGNORECASE)
# A short model token in brackets right after the name is part of it:
# "Nothing Phone (2a) 5G (Black, 128 GB)".
head = re.sub(r"\((?:19|20)\d\d\)", " ", head) # "(2026)" is a model year, not part of the name
head = re.sub(r"\(([A-Za-z0-9+ ]{1,6})\)", lambda m: " " + m.group(1) + " "
if not re.search(r"\d\s*(?:gb|tb)", m.group(1), re.I) else m.group(0), head, count=1)
head = re.split(r"\s[-–|]\s|[(,|\[:]", head, maxsplit=1)[0]
if category == "laptops":
m = _LAPTOP_SPEC_START.search(head)
if m and m.start() > 0:
head = head[: m.start()]
if brand:
# Drop the parent brand's own name ("Samsung Galaxy S24" -> "Galaxy S24",
# "Apple iPhone 15" -> "iPhone 15"); a sub-brand stays ("Redmi Note 13").
from app.electronics.reference import load_reference
parent_aliases = sorted(load_reference().brands[brand.brand_slug].aliases, key=len, reverse=True)
for alias in parent_aliases:
head = re.sub(r"^\s*" + re.escape(alias) + r"\b", "", head, flags=re.IGNORECASE).strip()
head = re.sub(r"\b\d+\s*GB\s*RAM\b", " ", head, flags=re.IGNORECASE)
head = _PAIR_RE.sub(" ", head)
head = _SIZE_ANY.sub(" ", head)
if colour:
head = re.sub(re.escape(colour), " ", head, flags=re.IGNORECASE)
words = head.split()
while len(words) > 1 and _COLOUR_WORDS.fullmatch(words[-1]):
words.pop() # "iPhone 15 Black" -> "iPhone 15"
head = " ".join(words)
display = re.sub(r"\s+", " ", _DISPLAY_NOISE.sub(" ", head)).strip(" -–")
norm = normalise_model(head)
if not norm:
return None, None
return display or head.strip(), norm
def parse_title(title: str, category: str, *, expected_brand: Optional[str] = None) -> ParsedTitle:
title = re.sub(r"\s+", " ", (title or "")).strip()
brand = resolve_brand(title, expected=expected_brand)
parsed = ParsedTitle(title=title, brand=brand)
if not title:
return parsed
parsed.ram_gb, parsed.storage_gb = _parse_ram_storage(title, category)
parsed.colour = _parse_colour(title)
if re.search(r"\b5G\b", title, re.IGNORECASE):
parsed.network = "5G"
if category == "laptops":
parsed.processor = _parse_processor(title)
parsed.mpn = _parse_mpn(title, parsed.processor)
parsed.model, parsed.model_norm = _model_from_title(title, brand, category, parsed.colour, parsed.mpn)
return parsed
def variant_key(parsed: ParsedTitle, category: str) -> Optional[str]:
"""The identity of one real-world variant, or None if the title does not
state enough to tell variants apart."""
if not parsed.brand or not parsed.model_norm:
return None
b = parsed.brand.brand_slug
fmt = lambda d: "na" if d is None else format(d.normalize(), "f") # noqa: E731
if category == "laptops":
# A laptop configuration is its model line + CPU + RAM + storage. That
# is what every site states (a part number is shown by only a few), so
# it is the key whenever it is complete; the MPN is the fallback.
line = laptop_line(parsed.model_norm)
if line and processor_is_specific(parsed.processor) and parsed.ram_gb and parsed.storage_gb:
return f"{b}|laptops|{line}|{parsed.processor}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
if parsed.mpn:
return f"{b}|laptops|mpn:{parsed.mpn.lower()}"
return None
if parsed.storage_gb is None:
return None
return f"{b}|{category}|{parsed.model_norm}|{fmt(parsed.ram_gb)}|{fmt(parsed.storage_gb)}"
# Lenovo/Asus machine-type codes ("15amn8", "15irh10", "14iah8", "x1504za")
# name a chassis generation, and one site prints them where another does not.
_MACHINE_CODE = re.compile(r"^(?:\d{2}[a-z]{2,4}\d{1,2}|[a-z]\d{4}[a-z]{1,3})$")
def laptop_line(model_norm: Optional[str]) -> str:
"""The model line used for matching: "ideapad slim 3 15amn8" -> "ideapad slim 3"."""
tokens = [t for t in (model_norm or "").split() if not _MACHINE_CODE.match(t)]
return " ".join(tokens)
def fill_from_context(parsed: ParsedTitle, category: str, *, snippet: str = "",
spec_texts: tuple = ()) -> ParsedTitle:
"""Fill variant fields a (often truncated) title leaves out, from text the
same site published about the same page: its search snippet, or the spec
table of the fetched page. Only unambiguous values are taken - a snippet
naming two different storage sizes is describing several variants."""
if snippet and (parsed.ram_gb is None or parsed.storage_gb is None):
sizes = {_dec(v, u) for v, u in _SIZE_ANY.findall(snippet)}
if len(sizes) <= 2:
ram, storage = _parse_ram_storage(snippet, category)
if parsed.storage_gb is None and storage is not None:
parsed.storage_gb = storage
if parsed.ram_gb is None and ram is not None and ram != parsed.storage_gb:
parsed.ram_gb = ram
if category == "laptops" and not processor_is_specific(parsed.processor):
# A title that already names a CPU family ("Snapdragon X", "Core i7")
# is only completed from the page's own spec table, never from a
# snippet - snippets often run several products' titles together.
sources = list(spec_texts) + ([snippet] if parsed.processor is None and snippet else [])
found = set()
for text in sources:
found |= {cpu for cpu in all_processors(text) if processor_is_specific(cpu)}
if len(found) == 1:
parsed.processor = found.pop()
return parsed
def all_processors(text: str) -> set:
"""Every CPU named anywhere in `text` (a snippet can name several)."""
found = set()
for rx in _PROCESSOR_RES:
for m in rx.finditer(text or ""):
cpu = parse_processor(m.group(0))
if cpu:
found.add(cpu)
return found

View File

@@ -0,0 +1,102 @@
"""Fill missing prices on search-only platforms (Amazon.in, Flipkart, Croma...)
from Google Programmable Search, without fetching those sites.
For each listing that has no price (or an unconfirmed one), search Google for
that product on that site. A price is taken only when:
* the result is the SAME product page (its site product id equals the
listing's), and
* Google's structured data for the page (pagemap offer / product:price meta)
states an INR price.
The listing is updated through the normal path, so the price is stored with
its evidence, appended to price_history, and outlier-checked.
"""
from __future__ import annotations
import logging
from typing import Callable, Dict, Optional
from app.electronics.collector import Collector, RunOptions, RunStats, source_sku
from app.electronics.db import repository as repo
from app.electronics.db.connection import connect
from app.electronics.normalise.title_parser import parse_title
from app.electronics.reference import load_reference, site_for_url
from app.electronics.search.engine import SearchEngine
logger = logging.getLogger(__name__)
def _listings_needing_price(limit: int, category: Optional[str]) -> list:
with connect() as conn:
return conn.execute(
"""
SELECT l.id, l.title, l.source_sku, l.source_url, s.domain, b.slug AS brand_slug, c.slug AS category,
(p.verification_status = 'verified') AS verified
FROM elec.source_listing l
JOIN elec.site s ON s.id = l.site_id
JOIN elec.brand b ON b.id = l.brand_id
JOIN elec.category c ON c.id = l.category_id
JOIN elec.product_listing_map m ON m.listing_id = l.id AND m.review_status IN ('auto','approved')
JOIN elec.product p ON p.id = m.product_id
WHERE l.source_type = 'search_snippet'
AND (l.price IS NULL OR l.price_outlier)
AND (s.policy = 'serp_only' OR coalesce(s.probe_outcome, 'C') = 'C')
AND (%(category)s::text IS NULL OR c.slug = %(category)s)
ORDER BY (p.verification_status = 'verified') DESC, l.last_seen_at DESC
LIMIT %(limit)s
""",
{"limit": limit, "category": category},
).fetchall()
def lookup_prices(limit: int = 40, category: Optional[str] = None,
progress: Callable[[str], None] = logger.info) -> Dict[str, object]:
stats: Dict[str, object] = {"checked": 0, "priced": 0, "no_same_page": 0, "no_structured_price": 0}
engine = SearchEngine(budget=limit)
if not engine.google.enabled:
stats["error"] = "Google Programmable Search is not configured (GOOGLE_API_KEY / GOOGLE_CSE_ID)"
return stats
ids = repo.id_maps()
ref = load_reference()
run_id = repo.start_run("price_lookup", {"limit": limit, "category": category})
try:
for row in _listings_needing_price(limit, category):
if not engine.google.enabled:
break
site = ref.sites[row["domain"]]
query = f"site:{row['domain']} {row['title'][:110]}"
hits = engine.text(query, max_results=10, providers="google")
stats["checked"] += 1
if hits is None:
continue
same = [h for h in hits
if (s := site_for_url(h.url)) is not None and s.domain == site.domain
and source_sku(site, h.url) == row["source_sku"]]
if not same:
stats["no_same_page"] += 1
continue
hit = next((h for h in same if h.offer), None)
if hit is None:
stats["no_structured_price"] += 1
continue
collector = Collector.__new__(Collector) # only its listing builder is used
collector.opt = RunOptions(category=row["category"], brands=[row["brand_slug"]])
collector.stats = RunStats()
parsed = parse_title(row["title"], row["category"], expected_brand=row["brand_slug"])
if parsed.brand is None:
continue
listing = collector.listing_from_search(hit, site, parsed, query)
listing.source_sku = row["source_sku"]
if listing.price is None:
stats["no_structured_price"] += 1
continue
repo.upsert_listing(listing, ids, run_id)
stats["priced"] += 1
progress(f"{site.name}: {row['title'][:70]} -> Rs {listing.price}")
stats["products"] = repo.refresh_verification()
if engine.google.error:
stats["error"] = engine.google.error
repo.finish_run(run_id, "done", {k: v for k, v in stats.items() if k != "products"})
except Exception as exc:
repo.finish_run(run_id, "failed", {}, repr(exc))
raise
return stats

View File

@@ -0,0 +1,99 @@
"""Decide, per site, whether it may be scraped or only searched.
A robots.txt allows product pages, HTTP 200 without a bot check, and the
page carries a schema.org Product with an INR offer -> scrape
B fetchable, product name/specs readable from the HTML, but no
structured price -> scrape specs/images,
price from search
C serp_only policy, robots.txt disallows, blocked / CAPTCHA, or the page
has no product data without JavaScript -> web search only
The probe looks at 2-3 real product URLs for the site, found through web
search, so it grades the pages the collector would actually fetch.
"""
from __future__ import annotations
import logging
from typing import Dict, List, Optional
from app.electronics.extract.html_fallback import embedded_state, extract_page
from app.electronics.extract.jsonld import extract_products
from app.electronics.net.polite_client import PoliteClient
from app.electronics.reference import SiteRef, load_reference, site_for_url
from app.electronics.search.engine import SearchEngine
logger = logging.getLogger(__name__)
def sample_product_urls(site: SiteRef, engine: SearchEngine, limit: int = 3) -> List[str]:
ref = load_reference()
urls: List[str] = []
if site.kind == "brand_official":
brand = ref.brands[site.brand_slug]
terms = [ref.categories[c].search_terms[0] for c in brand.categories]
queries = [f"site:{site.domain} {brand.name} {t}" for t in terms]
else:
queries = [f"site:{site.domain} samsung galaxy 5g", f"site:{site.domain} lenovo laptop"]
rx = site.product_url_re
for q in queries:
for hit in engine.text(q, max_results=15) or []:
s = site_for_url(hit.url)
if not s or s.domain != site.domain:
continue
if rx is not None and not rx.search(hit.url):
continue
if hit.url not in urls:
urls.append(hit.url)
if len(urls) >= limit:
return urls
return urls
def grade_page(html: str) -> Dict[str, object]:
products = extract_products(html)
priced = [p for p in products if p.get("price") is not None and (p.get("currency") in (None, "INR"))]
page = extract_page(html)
return {
"jsonld_products": len(products),
"jsonld_priced": len(priced),
"meta_price": page.get("price") is not None,
"has_title": bool(page.get("name")),
"spec_rows": len(page.get("properties") or {}),
"embedded_state": embedded_state(html) is not None,
}
def probe_site(site: SiteRef, client: PoliteClient, engine: SearchEngine) -> Dict[str, object]:
"""Returns {"outcome", "robots_allowed", "evidence"}; never raises."""
if site.policy == "serp_only":
return {"outcome": "C", "robots_allowed": None,
"evidence": {"reason": "policy serp_only: this site is never fetched directly"}}
urls = sample_product_urls(site, engine)
if not urls:
return {"outcome": "C", "robots_allowed": None,
"evidence": {"reason": "no product URLs found through web search"}}
pages: List[dict] = []
robots_any: Optional[bool] = None
for url in urls:
res = client.get(url)
entry = {"url": url, "status": res.status, "outcome": res.outcome}
robots_any = res.robots_allowed if robots_any is None else (robots_any or bool(res.robots_allowed))
if res.ok:
entry.update(grade_page(res.text))
pages.append(entry)
if res.outcome in ("captcha", "blocked", "breaker_open"):
break
ok_pages = [p for p in pages if p["outcome"] == "ok"]
if any(p["outcome"] in ("captcha", "blocked") for p in pages):
outcome, reason = "C", "blocked or bot check - not fetched again until the breaker cools down"
elif all(p["outcome"] == "robots_disallowed" for p in pages):
outcome, reason = "C", "robots.txt disallows product pages"
elif not ok_pages:
outcome, reason = "C", "product pages could not be fetched"
elif any(p.get("jsonld_priced") for p in ok_pages):
outcome, reason = "A", "schema.org Product with an INR offer"
elif any(p.get("has_title") and (p.get("spec_rows") or p.get("meta_price") or p.get("jsonld_products")) for p in ok_pages):
outcome, reason = "B", "product details readable from HTML; no structured price"
else:
outcome, reason = "C", "no product data without JavaScript"
return {"outcome": outcome, "robots_allowed": robots_any, "evidence": {"reason": reason, "pages": pages}}

View File

@@ -0,0 +1,132 @@
"""Reference data (brands, categories, sites, spec dictionary) loaded from YAML."""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from functools import lru_cache
from pathlib import Path
from typing import Dict, List, Optional
import yaml
_DIR = Path(__file__).resolve().parent
def slugify(text: str) -> str:
return re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
@dataclass(frozen=True)
class BrandRef:
name: str
slug: str
categories: tuple
aliases: tuple
sub_brands: tuple
official: tuple
@dataclass(frozen=True)
class CategoryRef:
slug: str
name: str
search_terms: tuple
query_terms: tuple = ()
@dataclass(frozen=True)
class SiteRef:
domain: str
name: str
kind: str
region: str
policy: str
product_url: Optional[str] = None
pincode_param: Optional[str] = None
brand_slug: Optional[str] = None
@property
def product_url_re(self) -> Optional[re.Pattern]:
return re.compile(self.product_url) if self.product_url else None
@dataclass(frozen=True)
class Reference:
brands: Dict[str, BrandRef]
categories: Dict[str, CategoryRef]
sites: Dict[str, SiteRef]
spec_keys: Dict[str, dict] = field(default_factory=dict)
def brands_for(self, category: str) -> List[BrandRef]:
return [b for b in self.brands.values() if category in b.categories]
def _load_yaml(name: str) -> dict:
return yaml.safe_load((_DIR / name).read_text(encoding="utf-8")) or {}
@lru_cache(maxsize=1)
def load_reference() -> Reference:
raw_brands = _load_yaml("brands.yaml")
brands: Dict[str, BrandRef] = {}
for b in raw_brands.get("brands", []):
slug = slugify(b["name"])
brands[slug] = BrandRef(
name=b["name"],
slug=slug,
categories=tuple(b.get("categories", [])),
aliases=tuple(a.lower() for a in b.get("aliases", [])),
sub_brands=tuple(s.lower() for s in b.get("sub_brands", [])),
official=tuple(b.get("official", [])),
)
categories = {
c["slug"]: CategoryRef(c["slug"], c["name"], tuple(c.get("search_terms", [])),
tuple(c.get("query_terms", [])))
for c in raw_brands.get("categories", [])
}
sites: Dict[str, SiteRef] = {}
for s in _load_yaml("sites.yaml").get("sites", []):
sites[s["domain"]] = SiteRef(
domain=s["domain"],
name=s["name"],
kind=s["kind"],
region=s["region"],
policy=s["policy"],
product_url=s.get("product_url"),
pincode_param=s.get("pincode_param"),
)
# Every brand's official domains become sites of their own. They are
# probed like any retailer - an official page is the best evidence there is.
for b in brands.values():
for domain in b.official:
sites.setdefault(
domain,
SiteRef(
domain=domain,
name=f"{b.name} (official)",
kind="brand_official",
region="national",
policy="probe",
brand_slug=b.slug,
),
)
spec_keys = _load_yaml("spec_keys.yaml").get("categories", {})
return Reference(brands=brands, categories=categories, sites=sites, spec_keys=spec_keys)
def site_for_url(url: str) -> Optional[SiteRef]:
"""The registered site a URL belongs to (subdomains included), or None."""
from urllib.parse import urlparse
host = (urlparse(url).hostname or "").lower()
if not host:
return None
ref = load_reference()
best: Optional[SiteRef] = None
for domain, site in ref.sites.items():
if host == domain or host.endswith("." + domain):
if best is None or len(domain) > len(best.domain):
best = site
return best

View File

@@ -0,0 +1,100 @@
# Brand allow-list. A listing whose brand does not resolve to one of these is
# rejected - the catalogue is closed-world by design.
#
# aliases spellings seen on retail pages (matched case-insensitively,
# longest alias first, as a whole word at the start of a title)
# sub_brands product families sold under a parent brand. They resolve to the
# parent, and are kept as the product family.
# official the brand's own Indian web domains. A product page on one of
# these is the strongest evidence that a product exists.
brands:
- name: Samsung
categories: [mobiles, laptops]
aliases: [samsung]
official: [samsung.com]
- name: Apple
categories: [mobiles, laptops]
aliases: [apple]
sub_brands: [iphone, macbook]
official: [apple.com]
- name: Xiaomi
categories: [mobiles]
aliases: [xiaomi]
sub_brands: [redmi, poco, mi]
official: [mi.com]
- name: OnePlus
categories: [mobiles]
aliases: [oneplus, one plus]
official: [oneplus.in]
- name: Vivo
categories: [mobiles]
aliases: [vivo]
sub_brands: [iqoo]
official: [vivo.com, iqoo.com]
- name: Oppo
categories: [mobiles]
aliases: [oppo]
official: [oppo.com]
- name: Realme
categories: [mobiles]
aliases: [realme]
sub_brands: [narzo]
official: [realme.com]
- name: Motorola
categories: [mobiles]
aliases: [motorola, moto]
official: [motorola.co.in, motorola.com]
- name: Google
categories: [mobiles]
aliases: [google]
sub_brands: [pixel]
official: [store.google.com]
- name: Nothing
categories: [mobiles]
aliases: [nothing]
sub_brands: [cmf]
official: [nothing.tech]
- name: HP
categories: [laptops]
aliases: [hp, hewlett packard]
sub_brands: [omen, victus, pavilion, envy, spectre]
official: [hp.com]
- name: Dell
categories: [laptops]
aliases: [dell]
sub_brands: [alienware, inspiron, vostro, latitude, xps]
official: [dell.com]
- name: Lenovo
categories: [laptops]
aliases: [lenovo]
sub_brands: [thinkpad, ideapad, legion, yoga, thinkbook, loq]
official: [lenovo.com]
- name: Asus
categories: [laptops]
aliases: [asus]
sub_brands: [rog, tuf, vivobook, zenbook]
official: [asus.com]
- name: Acer
categories: [laptops]
aliases: [acer]
sub_brands: [aspire, nitro, predator, swift]
official: [acer.com]
- name: MSI
categories: [laptops]
aliases: [msi]
official: [msi.com]
# search_terms: the category word used when probing brand sites.
# query_terms: appended to `site:<platform> <brand>` during discovery. They
# read like the variant part of a product title, which is what
# makes search engines return single product pages rather than
# category or blog pages.
categories:
- slug: mobiles
name: Mobiles
search_terms: [smartphone, mobile phone]
query_terms: ["5G 8GB RAM 128GB", "5G 8GB 256GB", "12GB RAM 256GB"]
- slug: laptops
name: Laptops
search_terms: [laptop]
query_terms: ["laptop 16GB RAM 512GB SSD", "laptop 8GB RAM 512GB SSD"]

View File

@@ -0,0 +1,77 @@
# Retail platforms.
#
# kind marketplace | national_chain | tn_regional
# region national | TN (TN = a Tamil Nadu retail chain)
# policy serp_only -> NEVER fetched directly; everything comes from web
# search results (titles, snippets, image results)
# probe -> fetched only if the site probe grades it A or B
# (robots.txt allows, HTTP 200, no CAPTCHA); otherwise
# it falls back to search results like serp_only
# product_url regex a URL must match to count as a single product page.
# Group 1, when present, is the site's own product id.
# pincode_param optional query parameter the site accepts for a delivery
# pincode. Only sites that actually honour it get
# pincode_applied=true on their prices.
#
# Brand official sites are generated from brands.yaml (kind brand_official).
sites:
- domain: amazon.in
name: Amazon.in
kind: marketplace
region: national
policy: serp_only
product_url: '/(?:dp|gp/product)/([A-Z0-9]{10})'
- domain: flipkart.com
name: Flipkart
kind: marketplace
region: national
policy: serp_only
product_url: '/p/(itm[0-9a-z]+)'
- domain: croma.com
name: Croma
kind: national_chain
region: national
policy: probe
product_url: '/p/(\d{5,})'
- domain: reliancedigital.in
name: Reliance Digital
kind: national_chain
region: national
policy: probe
product_url: '(?:/p/|/product/[^?#]*?-)(\d{6,})'
- domain: vijaysales.com
name: Vijay Sales
kind: national_chain
region: national
policy: probe
product_url: '/p/(?:P?)(\d{3,})/'
- domain: tatacliq.com
name: Tata CLiQ
kind: marketplace
region: national
policy: probe
product_url: '/p-(mp\d+)'
- domain: poorvika.com
name: Poorvika
kind: tn_regional
region: TN
policy: probe
product_url: '/([a-z0-9-]{8,})/p/?$'
- domain: sangeethamobiles.com
name: Sangeetha Mobiles
kind: tn_regional
region: TN
policy: probe
product_url: '(?i)/product-?details/(?:[^/?#]+/)?(\d+)'
- domain: vasanthandco.in
name: Vasanth & Co
kind: tn_regional
region: TN
policy: probe
product_url: '/(?:product|products)/([a-z0-9-]{8,})'
- domain: viveks.com
name: Viveks
kind: tn_regional
region: TN
policy: probe
product_url: '/([a-z0-9-]{8,})\.html$'

View File

@@ -0,0 +1,127 @@
# Canonical specification keys per category.
#
# type number | text | enum
# unit canonical unit for numbers (values are converted into it)
# synonyms spec labels seen on retail/brand pages (case-insensitive,
# punctuation ignored). A label maps to the first key that lists it.
# range plausible [min, max] after conversion; values outside are dropped
# values allowed canonical values for enums, each with its match words
#
# A value is only ever stored if it was read from a page or snippet. Nothing
# here supplies a default.
categories:
mobiles:
ram_gb:
type: number
unit: GB
range: [1, 32]
synonyms: [ram, memory ram, ram size, ram capacity, installed ram, system memory]
storage_gb:
type: number
unit: GB
range: [8, 2048]
synonyms: [internal storage, storage, rom, internal memory, storage capacity, inbuilt memory, memory storage capacity]
display_inch:
type: number
unit: inch
range: [3, 9]
synonyms: [display size, screen size, display, screen size inches, standing screen display size]
display_type:
type: enum
synonyms: [display type, screen type, display technology, panel type]
values:
AMOLED: [amoled, super amoled, dynamic amoled, pole amoled, fluid amoled]
OLED: [oled, super retina, ltpo oled]
LCD: [lcd, ips lcd, tft, ips]
refresh_hz:
type: number
unit: Hz
range: [30, 240]
synonyms: [refresh rate, screen refresh rate, display refresh rate]
processor:
type: text
synonyms: [processor, chipset, processor name, soc, cpu, processor brand]
rear_camera_mp:
type: number
unit: MP
range: [2, 250]
synonyms: [rear camera, primary camera, main camera, back camera, rear camera resolution, primary camera resolution]
front_camera_mp:
type: number
unit: MP
range: [2, 60]
synonyms: [front camera, secondary camera, selfie camera, front camera resolution]
battery_mah:
type: number
unit: mAh
range: [1000, 10000]
synonyms: [battery capacity, battery, battery power, battery capacity mah]
os:
type: enum
synonyms: [operating system, os, os version]
values:
Android: [android]
iOS: [ios]
network:
type: enum
synonyms: [network type, network, cellular technology, connectivity technology, network connectivity]
values:
5G: [5g]
4G: [4g, lte]
colour:
type: text
synonyms: [colour, color, colour name, color name]
laptops:
processor:
type: text
synonyms: [processor, processor name, cpu, processor model, processor type]
ram_gb:
type: number
unit: GB
range: [2, 128]
synonyms: [ram, ram size, memory, system memory, installed ram, ram capacity]
storage_gb:
type: number
unit: GB
range: [32, 8192]
synonyms: [ssd capacity, storage, hard disk size, hard drive size, storage capacity, ssd, internal storage]
storage_type:
type: enum
synonyms: [storage type, hard disk type, hard drive interface, drive type]
values:
SSD: [ssd, nvme, solid state]
HDD: [hdd, hard disk drive]
eMMC: [emmc]
display_inch:
type: number
unit: inch
range: [10, 19]
synonyms: [screen size, display size, standing screen display size, display]
resolution:
type: text
synonyms: [resolution, screen resolution, display resolution, maximum display resolution]
gpu:
type: text
synonyms: [graphics, graphics processor, gpu, graphic processor, graphics coprocessor, graphics card]
os:
type: enum
synonyms: [operating system, os]
values:
Windows: [windows]
macOS: [macos, mac os]
ChromeOS: [chrome os, chromeos]
Linux: [linux, ubuntu]
DOS: [dos, free dos, freedos]
weight_kg:
type: number
unit: kg
range: [0.5, 5]
synonyms: [weight, item weight, product weight, laptop weight]
battery_wh:
type: number
unit: Wh
range: [20, 120]
synonyms: [battery capacity, battery, battery power]
colour:
type: text
synonyms: [colour, color]

View File

@@ -0,0 +1,96 @@
"""Which real customer reviews to show for a product, and in what mix.
Every review passed in here was read from a product page's own schema.org
data (see extract/jsonld.py); this module only classifies and selects - it
never writes, rewrites or summarises review text.
Sentiment is the reviewer's own star rating, nothing inferred:
>= 4 positive, >= 3 neutral, < 3 negative.
The mix follows the product's overall rating, so the reviews shown read like
the rating does:
rating >= 4.0 mostly positive, some neutral, a little negative
3.0 < rating < 4.0 mostly neutral, some positive, a little negative
rating <= 3.0 mostly negative, a little positive and neutral
When a group has too few reviews its slots go to the other groups, in the
same priority order. Nothing is ever padded: if only 3 real reviews exist,
3 are shown.
"""
from __future__ import annotations
from decimal import Decimal
from typing import Any, Dict, List, Optional, Sequence, Tuple
POSITIVE, NEUTRAL, NEGATIVE = "positive", "neutral", "negative"
MAX_REVIEWS = 10
# (group, share of MAX_REVIEWS), highest priority first.
_MIX_HIGH: Tuple[Tuple[str, int], ...] = ((POSITIVE, 6), (NEUTRAL, 3), (NEGATIVE, 1))
_MIX_MID: Tuple[Tuple[str, int], ...] = ((NEUTRAL, 5), (POSITIVE, 3), (NEGATIVE, 2))
_MIX_LOW: Tuple[Tuple[str, int], ...] = ((NEGATIVE, 6), (POSITIVE, 2), (NEUTRAL, 2))
def sentiment_for(rating: Any) -> Optional[str]:
"""The group a reviewer's own star rating puts a review in; None when the
review states no rating."""
if rating is None:
return None
try:
value = Decimal(str(rating))
except Exception: # noqa: BLE001
return None
if value >= 4:
return POSITIVE
if value >= 3:
return NEUTRAL
return NEGATIVE
def mix_for(product_rating: Any) -> Tuple[Tuple[str, int], ...]:
if product_rating is None:
return _MIX_MID # no overall rating stated: a balanced view
value = Decimal(str(product_rating))
if value >= 4:
return _MIX_HIGH
if value > 3:
return _MIX_MID
return _MIX_LOW
def _rank_key(review: Dict[str, Any]) -> tuple:
# Newest first (ISO dates sort as text), then the more substantial review.
return (str(review.get("review_date") or ""), len(review.get("body") or ""))
def select_reviews(product_rating: Any, reviews: Sequence[Dict[str, Any]],
max_n: int = MAX_REVIEWS) -> List[Dict[str, Any]]:
"""Up to `max_n` of `reviews`, mixed by sentiment as described above.
Reviews without a star rating have no sentiment and are not shown: there
is no honest way to place them in the mix.
"""
groups: Dict[str, List[Dict[str, Any]]] = {POSITIVE: [], NEUTRAL: [], NEGATIVE: []}
seen = set()
for r in reviews:
s = r.get("sentiment") or sentiment_for(r.get("rating"))
key = (r.get("body") or "").strip().lower()
if s is None or not key or key in seen:
continue
seen.add(key)
groups[s].append({**r, "sentiment": s})
for g in groups.values():
g.sort(key=_rank_key, reverse=True)
mix = mix_for(product_rating)
scale = max_n / MAX_REVIEWS
quota = {g: int(round(n * scale)) for g, n in mix}
picked: Dict[str, List[Dict[str, Any]]] = {g: groups[g][: quota[g]] for g, _ in mix}
# Hand unused slots to the other groups, in priority order.
spare = max_n - sum(len(v) for v in picked.values())
for g, _ in mix:
if spare <= 0:
break
extra = groups[g][len(picked[g]): len(picked[g]) + spare]
picked[g].extend(extra)
spare -= len(extra)
return [r for g, _ in mix for r in picked[g]]

View File

@@ -0,0 +1,65 @@
"""Cached, budgeted access to the search providers.
Results are cached in elec.search_cache so a re-run does not query again
within SEARCH_CACHE_TTL_HOURS, and each run has a query budget so a large
brand list cannot hammer the providers.
"""
from __future__ import annotations
import logging
from typing import Dict, List, Optional
from app.electronics.db import repository as repo
from app.electronics.search.providers import DuckDuckGoProvider, GoogleCseProvider, SearchHit
from app.infrastructure.settings import GOOGLE_CSE_DAILY_QUOTA, SEARCH_CACHE_TTL_HOURS
logger = logging.getLogger(__name__)
class SearchEngine:
def __init__(self, *, budget: int = 200, use_cache: bool = True) -> None:
self.budget = budget
self.use_cache = use_cache
self.used = 0
self.stats: Dict[str, int] = {"cache_hits": 0, "queries": 0, "unavailable": 0}
self.ddg = DuckDuckGoProvider()
self.google = GoogleCseProvider(
quota_left=lambda: GOOGLE_CSE_DAILY_QUOTA - repo.google_queries_today()
)
def _ask(self, provider, kind: str, query: str, max_results: int) -> Optional[List[SearchHit]]:
if not provider.enabled:
return None
cached = repo.search_cache_get(provider.name, kind, query, SEARCH_CACHE_TTL_HOURS) if self.use_cache else None
if cached is not None:
self.stats["cache_hits"] += 1
return [SearchHit.from_dict(d) for d in cached]
if self.used >= self.budget:
logger.info("Search budget (%d) spent; skipping %r", self.budget, query)
return None
self.used += 1
self.stats["queries"] += 1
self.stats[f"queries_{provider.name}"] = self.stats.get(f"queries_{provider.name}", 0) + 1
hits = provider.text(query, max_results) if kind == "text" else provider.images(query, max_results)
if hits is None:
self.stats["unavailable"] += 1
return None
repo.search_cache_put(provider.name, kind, query, [h.to_dict() for h in hits])
return hits
def _run(self, kind: str, query: str, max_results: int, providers: str) -> Optional[List[SearchHit]]:
"""providers: "default" = DuckDuckGo, with Google only when DuckDuckGo
gives no answer (keeps the 100/day Google quota for price lookups);
"google" = Google only."""
if providers == "google":
return self._ask(self.google, kind, query, max_results)
hits = self._ask(self.ddg, kind, query, max_results)
if hits is None:
hits = self._ask(self.google, kind, query, max_results)
return hits
def text(self, query: str, max_results: int = 20, *, providers: str = "default") -> Optional[List[SearchHit]]:
return self._run("text", query, max_results, providers)
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
return self._run("images", query, max_results, "default")

View File

@@ -0,0 +1,234 @@
"""Web search: DuckDuckGo (ddgs, no key) and, when configured, Google
Programmable Search.
Three outcomes, never collapsed: a list of hits, an empty list ("we asked and
nothing matched"), or None ("we could not ask" - throttled, offline, no
quota). A throttle is not evidence that a product is not sold anywhere.
"""
from __future__ import annotations
import logging
import threading
import time
from dataclasses import asdict, dataclass
from typing import Callable, List, Optional
import requests
from app.infrastructure.settings import (
GOOGLE_API_KEY,
GOOGLE_CSE_ID,
SEARCH_MIN_INTERVAL_SECONDS,
SEARCH_REGION,
USE_DDG_SEARCH,
USE_GOOGLE_CSE,
)
logger = logging.getLogger(__name__)
@dataclass
class SearchHit:
url: str
title: str
snippet: str
provider: str
rank: int
image_url: Optional[str] = None # image searches: the image itself (url = page it is on)
# Structured offer data the search engine itself extracted from the page
# (Google CSE "pagemap"): {"price", "currency", "availability", "raw"}.
offer: Optional[dict] = None
# Aggregate rating the search engine extracted from the page's own
# structured data (Google CSE "pagemap"): {"rating", "review_count", "raw"}.
rating: Optional[dict] = None
def to_dict(self) -> dict:
return asdict(self)
@classmethod
def from_dict(cls, d: dict) -> "SearchHit":
return cls(**{k: d.get(k) for k in ("url", "title", "snippet", "provider", "rank", "image_url", "offer", "rating")})
class _Pacer:
def __init__(self, interval: float, sleep: Callable[[float], None] = time.sleep) -> None:
self.interval = interval
self._sleep = sleep
self._last = 0.0
self._lock = threading.Lock()
def wait(self) -> None:
with self._lock:
gap = self.interval - (time.monotonic() - self._last)
if gap > 0:
self._sleep(gap)
self._last = time.monotonic()
class DuckDuckGoProvider:
name = "ddg"
def __init__(self, interval: float = SEARCH_MIN_INTERVAL_SECONDS) -> None:
self._pacer = _Pacer(interval)
self.enabled = USE_DDG_SEARCH
# "auto" rotates ddgs's engines; yahoo is a second opinion when it is throttled.
BACKENDS = ("auto", "yahoo")
def text(self, query: str, max_results: int = 20) -> Optional[List[SearchHit]]:
"""Hits, or None when no backend answered. ddgs reports a throttle and
a genuinely empty result the same way ("No results found"), so an
empty answer is treated as unknown rather than as "not listed"."""
if not self.enabled:
return None
try:
from ddgs import DDGS
except ImportError:
logger.warning("ddgs is not installed; DuckDuckGo search unavailable")
return None
for backend in self.BACKENDS:
self._pacer.wait()
try:
with DDGS(timeout=20) as ddgs:
rows = list(ddgs.text(query, region=SEARCH_REGION, safesearch="moderate",
max_results=max_results, backend=backend) or [])
except Exception as exc: # noqa: BLE001 - ddgs raises many types on throttling
logger.info("DuckDuckGo(%s) text search gave no answer (%s): %s", backend, query, exc)
continue
hits = [
SearchHit(r.get("href") or r.get("url") or "", r.get("title") or "", r.get("body") or "",
f"{self.name}", i)
for i, r in enumerate(rows)
if (r.get("href") or r.get("url") or "").startswith("http")
]
if hits:
return hits
return None
def images(self, query: str, max_results: int = 15) -> Optional[List[SearchHit]]:
if not self.enabled:
return None
try:
from ddgs import DDGS
except ImportError:
return None
self._pacer.wait()
try:
with DDGS(timeout=20) as ddgs:
rows = list(ddgs.images(query, region=SEARCH_REGION, safesearch="moderate",
max_results=max_results) or [])
except Exception as exc: # noqa: BLE001
if "no results" in str(exc).lower():
return []
logger.info("DuckDuckGo image search failed (%s): %s", query, exc)
return None
return [
SearchHit(r.get("url") or "", r.get("title") or "", "", self.name, i, image_url=r.get("image"))
for i, r in enumerate(rows)
if str(r.get("image") or "").startswith("http") and str(r.get("url") or "").startswith("http")
]
class GoogleCseProvider:
name = "google"
ENDPOINT = "https://www.googleapis.com/customsearch/v1"
def __init__(self, quota_left: Callable[[], int] = lambda: 100) -> None:
self.enabled = USE_GOOGLE_CSE
self.error: Optional[str] = None
self._quota_left = quota_left
self._pacer = _Pacer(1.0)
def _call(self, query: str, extra: dict) -> Optional[List[dict]]:
if not self.enabled:
return None
if self._quota_left() <= 0:
self.error = "daily query quota used up"
return None
self._pacer.wait()
try:
resp = requests.get(
self.ENDPOINT,
params={"key": GOOGLE_API_KEY, "cx": GOOGLE_CSE_ID, "q": query, "gl": "in", "num": 10, **extra},
timeout=20,
)
except requests.RequestException as exc:
logger.info("Google CSE failed: %s", exc)
return None
if resp.status_code in (400, 401, 403):
# A key/project problem will not fix itself mid-run: stop asking.
try:
message = resp.json().get("error", {}).get("message", "")
except ValueError:
message = resp.text[:200]
self.enabled = False
self.error = f"HTTP {resp.status_code}: {message}"
logger.warning("Google Programmable Search disabled for this run - %s", self.error)
return None
if resp.status_code != 200:
logger.info("Google CSE HTTP %s: %s", resp.status_code, resp.text[:200])
return None
return resp.json().get("items", []) or []
def text(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
items = self._call(query, {})
if items is None:
return None
return [SearchHit(i.get("link", ""), i.get("title", ""), i.get("snippet", ""), self.name, n,
offer=pagemap_offer(i.get("pagemap") or {}),
rating=pagemap_rating(i.get("pagemap") or {}))
for n, i in enumerate(items[:max_results]) if i.get("link")]
def images(self, query: str, max_results: int = 10) -> Optional[List[SearchHit]]:
items = self._call(query, {"searchType": "image"})
if items is None:
return None
return [SearchHit((i.get("image") or {}).get("contextLink", ""), i.get("title", ""), "", self.name, n,
image_url=i.get("link"))
for n, i in enumerate(items[:max_results]) if i.get("link")]
def pagemap_offer(pagemap: dict) -> Optional[dict]:
"""The offer Google extracted from the page's own structured data
(schema.org Offer, or product:price meta tags), if any. INR only."""
candidates = []
for offer in pagemap.get("offer") or []:
candidates.append((offer.get("price"), offer.get("pricecurrency"), offer.get("availability"), offer))
for meta in pagemap.get("metatags") or []:
price = meta.get("product:price:amount") or meta.get("og:price:amount")
if price:
candidates.append((price, meta.get("product:price:currency") or meta.get("og:price:currency"),
meta.get("product:availability") or meta.get("og:availability"),
{k: v for k, v in meta.items() if "price" in k or "availability" in k}))
for price, currency, availability, raw in candidates:
if price and (currency or "").upper() == "INR":
return {"price": str(price), "currency": "INR", "availability": availability, "raw": raw}
return None
def pagemap_rating(pagemap: dict) -> Optional[dict]:
"""The aggregate rating Google extracted from the page's own structured
data (schema.org AggregateRating), if any. Only a value on a 5-point
scale is accepted."""
for node in pagemap.get("aggregaterating") or []:
try:
value = float(str(node.get("ratingvalue", "")).replace(",", "."))
except ValueError:
continue
best = node.get("bestrating")
try:
if best not in (None, "") and float(best) != 5:
continue
except ValueError:
continue
if not 0 < value <= 5:
continue
count = None
for key in ("reviewcount", "ratingcount"):
digits = "".join(ch for ch in str(node.get(key) or "") if ch.isdigit())
if digits:
count = int(digits)
break
return {"rating": round(value, 2), "review_count": count,
"raw": {k: v for k, v in node.items() if k in ("ratingvalue", "reviewcount", "ratingcount", "bestrating")}}
return None