Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
0
backend/app/electronics/extract/__init__.py
Normal file
0
backend/app/electronics/extract/__init__.py
Normal file
123
backend/app/electronics/extract/html_fallback.py
Normal file
123
backend/app/electronics/extract/html_fallback.py
Normal file
@@ -0,0 +1,123 @@
|
||||
"""Product facts from page markup when there is no usable JSON-LD.
|
||||
|
||||
Only machine-readable markup is trusted for the price: OpenGraph/product meta
|
||||
tags and schema.org microdata (itemprop="price"). Free text on the page is not
|
||||
scanned for rupee amounts - a product page shows EMIs, offers and other
|
||||
products' prices, and picking the wrong one is worse than picking none.
|
||||
Spec tables (<table>, <dl>) supply specifications.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
|
||||
def _dec(value: Optional[str]) -> Optional[Decimal]:
|
||||
if not value:
|
||||
return None
|
||||
try:
|
||||
return Decimal(re.sub(r"[^\d.]", "", value))
|
||||
except InvalidOperation:
|
||||
return None
|
||||
|
||||
|
||||
def _meta(soup: BeautifulSoup, *names: str) -> Optional[str]:
|
||||
for name in names:
|
||||
tag = soup.find("meta", attrs={"property": name}) or soup.find("meta", attrs={"name": name})
|
||||
if tag and tag.get("content"):
|
||||
return tag["content"].strip()
|
||||
return None
|
||||
|
||||
|
||||
def spec_tables(soup: BeautifulSoup, limit: int = 200) -> Dict[str, str]:
|
||||
specs: Dict[str, str] = {}
|
||||
for row in soup.select("table tr"):
|
||||
cells = row.find_all(["th", "td"])
|
||||
if len(cells) == 2:
|
||||
k, v = (c.get_text(" ", strip=True) for c in cells)
|
||||
if k and v and len(k) <= 60 and len(v) <= 200:
|
||||
specs.setdefault(k, v)
|
||||
if len(specs) >= limit:
|
||||
return specs
|
||||
for dl in soup.find_all("dl"):
|
||||
for dt in dl.find_all("dt"):
|
||||
dd = dt.find_next_sibling("dd")
|
||||
if dd:
|
||||
k, v = dt.get_text(" ", strip=True), dd.get_text(" ", strip=True)
|
||||
if k and v and len(k) <= 60 and len(v) <= 200:
|
||||
specs.setdefault(k, v)
|
||||
return specs
|
||||
|
||||
|
||||
def extract_page(html: str) -> Dict[str, Any]:
|
||||
soup = BeautifulSoup(html, "lxml")
|
||||
title = _meta(soup, "og:title", "twitter:title")
|
||||
if not title:
|
||||
h1 = soup.find("h1")
|
||||
title = h1.get_text(" ", strip=True) if h1 else None
|
||||
images: List[str] = []
|
||||
for name in ("og:image", "og:image:secure_url", "twitter:image"):
|
||||
v = _meta(soup, name)
|
||||
if v and v.startswith("http") and v not in images:
|
||||
images.append(v)
|
||||
|
||||
price = _dec(_meta(soup, "product:price:amount", "og:price:amount"))
|
||||
currency = _meta(soup, "product:price:currency", "og:price:currency")
|
||||
evidence = ""
|
||||
if price is not None:
|
||||
evidence = f"meta product:price:amount={price} currency={currency}"
|
||||
else:
|
||||
tag = soup.find(attrs={"itemprop": "price"})
|
||||
if tag is not None:
|
||||
raw = tag.get("content") or tag.get_text(" ", strip=True)
|
||||
price = _dec(raw)
|
||||
cur_tag = soup.find(attrs={"itemprop": "priceCurrency"})
|
||||
currency = (cur_tag.get("content") if cur_tag else None) or currency
|
||||
if price is not None:
|
||||
evidence = f'itemprop="price" {raw} currency={currency}'
|
||||
|
||||
availability = _meta(soup, "product:availability", "og:availability")
|
||||
in_stock = None
|
||||
if availability:
|
||||
low = availability.lower().replace(" ", "")
|
||||
in_stock = True if "instock" in low else False if ("outofstock" in low or "oos" == low) else None
|
||||
|
||||
return {
|
||||
"name": title,
|
||||
"images": images,
|
||||
"price": price,
|
||||
"currency": currency,
|
||||
"availability": availability,
|
||||
"in_stock": in_stock,
|
||||
"properties": spec_tables(soup),
|
||||
"evidence": evidence,
|
||||
}
|
||||
|
||||
|
||||
_STATE_RE = re.compile(
|
||||
r"<script[^>]*id=\"__NEXT_DATA__\"[^>]*>(.*?)</script>"
|
||||
r"|window\.__(?:INITIAL|PRELOADED)_STATE__\s*=\s*(\{.*?\})\s*;?\s*</script>",
|
||||
re.DOTALL,
|
||||
)
|
||||
|
||||
|
||||
def embedded_state(html: str) -> Optional[Any]:
|
||||
"""The page's server-rendered application state, when it embeds one."""
|
||||
for m in _STATE_RE.finditer(html):
|
||||
raw = m.group(1) or m.group(2)
|
||||
try:
|
||||
return json.loads(raw)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def visible_text(html: str, limit: int = 6000) -> str:
|
||||
soup = BeautifulSoup(html, "lxml")
|
||||
for tag in soup(["script", "style", "noscript", "svg", "header", "footer", "nav"]):
|
||||
tag.decompose()
|
||||
return re.sub(r"\s+", " ", soup.get_text(" ", strip=True))[:limit]
|
||||
202
backend/app/electronics/extract/jsonld.py
Normal file
202
backend/app/electronics/extract/jsonld.py
Normal file
@@ -0,0 +1,202 @@
|
||||
"""schema.org Product data embedded in a page as JSON-LD.
|
||||
|
||||
This is the preferred source on any page: it is what the site publishes for
|
||||
search engines, so it is stable and states price, currency and availability
|
||||
explicitly.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Dict, Iterable, List, Optional
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
_PRODUCT_TYPES = {"product", "productgroup", "productmodel", "individualproduct"}
|
||||
|
||||
|
||||
def _types(node: dict) -> set:
|
||||
t = node.get("@type")
|
||||
if isinstance(t, list):
|
||||
return {str(x).lower() for x in t}
|
||||
return {str(t).lower()} if t else set()
|
||||
|
||||
|
||||
def _walk(node: Any) -> Iterable[dict]:
|
||||
if isinstance(node, dict):
|
||||
yield node
|
||||
for v in node.values():
|
||||
yield from _walk(v)
|
||||
elif isinstance(node, list):
|
||||
for item in node:
|
||||
yield from _walk(item)
|
||||
|
||||
|
||||
def json_ld_blocks(html: str) -> List[Any]:
|
||||
soup = BeautifulSoup(html, "lxml")
|
||||
blocks = []
|
||||
for tag in soup.find_all("script", attrs={"type": re.compile(r"ld\+json", re.I)}):
|
||||
raw = (tag.string or tag.get_text() or "").strip()
|
||||
if not raw:
|
||||
continue
|
||||
try:
|
||||
blocks.append(json.loads(raw))
|
||||
except json.JSONDecodeError:
|
||||
# Some sites put several objects or trailing commas in one tag.
|
||||
try:
|
||||
blocks.append(json.loads(re.sub(r",\s*([}\]])", r"\1", raw)))
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return blocks
|
||||
|
||||
|
||||
def _dec(value: Any) -> Optional[Decimal]:
|
||||
if value is None or value == "":
|
||||
return None
|
||||
try:
|
||||
return Decimal(str(value).replace(",", "").strip())
|
||||
except InvalidOperation:
|
||||
return None
|
||||
|
||||
|
||||
def _text(value: Any) -> Optional[str]:
|
||||
if isinstance(value, dict):
|
||||
value = value.get("name") or value.get("@value")
|
||||
if isinstance(value, list):
|
||||
value = value[0] if value else None
|
||||
return str(value).strip() if value not in (None, "") else None
|
||||
|
||||
|
||||
def _images(value: Any) -> List[str]:
|
||||
out: List[str] = []
|
||||
for v in value if isinstance(value, list) else [value]:
|
||||
if isinstance(v, dict):
|
||||
v = v.get("url") or v.get("contentUrl")
|
||||
if isinstance(v, str) and v.startswith(("http://", "https://")):
|
||||
out.append(v)
|
||||
return out
|
||||
|
||||
|
||||
def _availability(value: Any) -> tuple:
|
||||
text = (_text(value) or "").lower()
|
||||
if not text:
|
||||
return None, None
|
||||
if "instock" in text or "limitedavailability" in text or "onlineonly" in text:
|
||||
return "InStock", True
|
||||
if any(k in text for k in ("outofstock", "soldout", "discontinued", "preorder", "presale")):
|
||||
return text.rsplit("/", 1)[-1], False
|
||||
return text.rsplit("/", 1)[-1], None
|
||||
|
||||
|
||||
def _offer(offers: Any) -> Dict[str, Any]:
|
||||
"""The price/availability of the product's (lowest) offer."""
|
||||
candidates = offers if isinstance(offers, list) else [offers]
|
||||
best: Dict[str, Any] = {}
|
||||
for o in candidates:
|
||||
if not isinstance(o, dict):
|
||||
continue
|
||||
price = _dec(o.get("price"))
|
||||
if price is None:
|
||||
price = _dec(o.get("lowPrice"))
|
||||
if price is None and isinstance(o.get("priceSpecification"), dict):
|
||||
price = _dec(o["priceSpecification"].get("price"))
|
||||
currency = _text(o.get("priceCurrency")) or (
|
||||
_text(o["priceSpecification"].get("priceCurrency")) if isinstance(o.get("priceSpecification"), dict) else None
|
||||
)
|
||||
availability, in_stock = _availability(o.get("availability"))
|
||||
entry = {"price": price, "currency": currency, "availability": availability, "in_stock": in_stock,
|
||||
"raw": {k: o.get(k) for k in ("price", "lowPrice", "priceCurrency", "availability") if k in o}}
|
||||
if price is not None and (not best or best.get("price") is None or price < best["price"]):
|
||||
best = entry
|
||||
elif not best:
|
||||
best = entry
|
||||
return best
|
||||
|
||||
|
||||
MAX_REVIEWS_PER_PAGE = 30
|
||||
|
||||
|
||||
def _review_rating(value: Any) -> Optional[Decimal]:
|
||||
"""A reviewer's star rating, rescaled to 0-5 when the page uses another scale."""
|
||||
if not isinstance(value, dict):
|
||||
return None
|
||||
rating = _dec(value.get("ratingValue"))
|
||||
if rating is None:
|
||||
return None
|
||||
best = _dec(value.get("bestRating")) or Decimal(5)
|
||||
if best <= 0:
|
||||
return None
|
||||
if best != 5:
|
||||
rating = rating * Decimal(5) / best
|
||||
if not (Decimal(0) <= rating <= Decimal(5)):
|
||||
return None
|
||||
return rating.quantize(Decimal("0.1"))
|
||||
|
||||
|
||||
def _reviews(node: dict) -> List[Dict[str, Any]]:
|
||||
"""Customer reviews published on the Product node (schema.org Review).
|
||||
|
||||
Only reviews with text are kept - a bare star with no words is not
|
||||
something a reader can weigh. Nothing is paraphrased or summarised: body,
|
||||
title and author are the page's own strings.
|
||||
"""
|
||||
raw = node.get("review") or node.get("reviews") or []
|
||||
out: List[Dict[str, Any]] = []
|
||||
for r in raw if isinstance(raw, list) else [raw]:
|
||||
if not isinstance(r, dict):
|
||||
continue
|
||||
body = _text(r.get("reviewBody")) or _text(r.get("description"))
|
||||
if not body:
|
||||
continue
|
||||
out.append({
|
||||
"author": _text(r.get("author")),
|
||||
"rating": _review_rating(r.get("reviewRating")),
|
||||
"title": _text(r.get("name")) or _text(r.get("headline")),
|
||||
"body": body[:4000],
|
||||
"review_date": _text(r.get("datePublished")) or _text(r.get("dateCreated")),
|
||||
})
|
||||
if len(out) >= MAX_REVIEWS_PER_PAGE:
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def extract_products(html: str) -> List[Dict[str, Any]]:
|
||||
"""All schema.org Product nodes on the page, flattened to plain fields."""
|
||||
products: List[Dict[str, Any]] = []
|
||||
for block in json_ld_blocks(html):
|
||||
for node in _walk(block):
|
||||
if not (_types(node) & _PRODUCT_TYPES):
|
||||
continue
|
||||
name = _text(node.get("name"))
|
||||
if not name:
|
||||
continue
|
||||
offer = _offer(node.get("offers")) if node.get("offers") else {}
|
||||
if not offer and isinstance(node.get("hasVariant"), list):
|
||||
offer = _offer([v.get("offers") for v in node["hasVariant"] if isinstance(v, dict) and v.get("offers")])
|
||||
props = {}
|
||||
for p in node.get("additionalProperty") or []:
|
||||
if isinstance(p, dict) and p.get("name") and p.get("value") not in (None, ""):
|
||||
props[str(p["name"])] = str(p["value"])
|
||||
rating = node.get("aggregateRating") if isinstance(node.get("aggregateRating"), dict) else {}
|
||||
products.append({
|
||||
"name": name,
|
||||
"brand": _text(node.get("brand")),
|
||||
"sku": _text(node.get("sku")) or _text(node.get("productID")),
|
||||
"mpn": _text(node.get("mpn")),
|
||||
"gtin": next((_text(node.get(k)) for k in ("gtin13", "gtin", "gtin12", "gtin14", "gtin8") if node.get(k)), None),
|
||||
"color": _text(node.get("color")),
|
||||
"images": _images(node.get("image")),
|
||||
"description": _text(node.get("description")),
|
||||
"price": offer.get("price"),
|
||||
"currency": offer.get("currency"),
|
||||
"availability": offer.get("availability"),
|
||||
"in_stock": offer.get("in_stock"),
|
||||
# Sites publish 0 for "no ratings yet"; that is not a rating.
|
||||
"rating": (_dec(rating.get("ratingValue")) or None),
|
||||
"review_count": int(_dec(rating.get("reviewCount") or rating.get("ratingCount")) or 0) or None,
|
||||
"reviews": _reviews(node),
|
||||
"properties": props,
|
||||
"evidence": json.dumps({"name": name, "offers": offer.get("raw")}, default=str)[:1500],
|
||||
})
|
||||
return products
|
||||
207
backend/app/electronics/extract/serp_parser.py
Normal file
207
backend/app/electronics/extract/serp_parser.py
Normal file
@@ -0,0 +1,207 @@
|
||||
"""Read prices and stock state out of text we did not render ourselves:
|
||||
search-result titles/snippets, and visible page text.
|
||||
|
||||
The rules lean hard towards NOT returning a price. A snippet usually carries
|
||||
several rupee amounts - the selling price, the MRP, an EMI, a bank discount, an
|
||||
exchange value, "₹X off" - and taking the wrong one is worse than taking none.
|
||||
An amount is only a price when nothing around it says it is something else.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import List, Optional
|
||||
|
||||
PRICE_MIN = Decimal("500")
|
||||
PRICE_MAX = Decimal("1000000")
|
||||
|
||||
# ₹ / Rs / Rs. / INR followed by an amount with Indian (1,29,999) or western
|
||||
# (129,999) grouping, or none.
|
||||
_AMOUNT = r"(\d{1,3}(?:,\d{2,3})+(?:\.\d{1,2})?|\d+(?:\.\d{1,2})?)"
|
||||
_MONEY_RE = re.compile(r"(?:₹|\bRs\.?|\bINR)\s?" + _AMOUNT, re.IGNORECASE)
|
||||
|
||||
# Words that make an amount something other than the selling price.
|
||||
_REJECT_BEFORE = re.compile(
|
||||
r"(?:emi|save|saving|savings|cashback|cash\s*back|exchange|bank|discount|coupon|"
|
||||
r"extra|instant|up\s*to|upto|flat|off\s+upto|worth|delivery|shipping|fee|charges?|"
|
||||
r"starting|starts|from|onwards|min(?:imum)?|as\s+low\s+as|down\s*payment|per\s+month)\W*$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_REJECT_AFTER = re.compile(
|
||||
r"^\W{0,3}(?:off\b|/\s*m(?:o|onth)?\b|per\s+month|p\.?m\.?\b|a\s+month|emi\b|/-?\s*emi|"
|
||||
r"cashback|discount|savings?|onwards|\+\s*shipping|delivery)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_MRP_BEFORE = re.compile(r"(?:m\.?\s?r\.?\s?p\.?|list\s+price|was|original\s+price)[\s:]*$", re.IGNORECASE)
|
||||
_RANGE_BETWEEN = re.compile(r"^\s*(?:-|–|—|to)\s*$", re.IGNORECASE)
|
||||
|
||||
_OUT_OF_STOCK = re.compile(
|
||||
r"\b(?:out\s+of\s+stock|currently\s+unavailable|sold\s+out|coming\s+soon|notify\s+me|"
|
||||
r"temporarily\s+unavailable|not\s+available)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_IN_STOCK = re.compile(r"\b(?:in\s+stock|available\s+now|buy\s+now|add\s+to\s+cart)\b", re.IGNORECASE)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Amount:
|
||||
value: Decimal
|
||||
kind: str # price | mrp | rejected
|
||||
reason: str
|
||||
start: int
|
||||
end: int
|
||||
raw: str
|
||||
|
||||
|
||||
def parse_amount(raw: str) -> Optional[Decimal]:
|
||||
try:
|
||||
value = Decimal(raw.replace(",", ""))
|
||||
except InvalidOperation:
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def find_amounts(text: str) -> List[Amount]:
|
||||
"""Every rupee amount in `text`, each classified as price, mrp or rejected."""
|
||||
out: List[Amount] = []
|
||||
if not text:
|
||||
return out
|
||||
matches = list(_MONEY_RE.finditer(text))
|
||||
for i, m in enumerate(matches):
|
||||
value = parse_amount(m.group(1))
|
||||
if value is None:
|
||||
continue
|
||||
before = text[max(0, m.start() - 28): m.start()]
|
||||
after = text[m.end(): m.end() + 22]
|
||||
kind, reason = "price", ""
|
||||
if _MRP_BEFORE.search(before):
|
||||
kind, reason = "mrp", "labelled MRP"
|
||||
elif _REJECT_BEFORE.search(before):
|
||||
kind, reason = "rejected", f"preceded by {_REJECT_BEFORE.search(before).group(0).strip()!r}"
|
||||
elif _REJECT_AFTER.search(after):
|
||||
kind, reason = "rejected", f"followed by {_REJECT_AFTER.search(after).group(0).strip()!r}"
|
||||
# A range ("₹10,999 - ₹12,999") names no single price.
|
||||
if kind == "price":
|
||||
if i + 1 < len(matches) and _RANGE_BETWEEN.match(text[m.end(): matches[i + 1].start()]):
|
||||
kind, reason = "rejected", "start of a price range"
|
||||
elif i > 0 and _RANGE_BETWEEN.match(text[matches[i - 1].end(): m.start()]):
|
||||
kind, reason = "rejected", "end of a price range"
|
||||
if kind != "rejected" and not (PRICE_MIN <= value <= PRICE_MAX):
|
||||
kind, reason = "rejected", "outside plausible range"
|
||||
out.append(Amount(value, kind, reason, m.start(), m.end(), m.group(0)))
|
||||
return out
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PriceReading:
|
||||
price: Optional[Decimal]
|
||||
mrp: Optional[Decimal]
|
||||
evidence: str # the exact substring the price was read from ("" if none)
|
||||
|
||||
|
||||
def read_price(text: str) -> PriceReading:
|
||||
"""The single selling price stated in `text`, or None.
|
||||
|
||||
If the text states two different unlabelled prices, it is ambiguous (a
|
||||
listing page snippet often shows several variants) and None is returned.
|
||||
"""
|
||||
amounts = find_amounts(text)
|
||||
prices = [a for a in amounts if a.kind == "price"]
|
||||
mrps = [a for a in amounts if a.kind == "mrp"]
|
||||
distinct = {a.value for a in prices}
|
||||
price: Optional[Decimal] = None
|
||||
evidence = ""
|
||||
if len(distinct) == 1:
|
||||
price = prices[0].value
|
||||
evidence = prices[0].raw
|
||||
mrp = mrps[0].value if mrps else None
|
||||
if price is not None and mrp is not None and mrp < price:
|
||||
mrp = None # an "MRP" below the selling price was misread; drop it
|
||||
return PriceReading(price, mrp, evidence)
|
||||
|
||||
|
||||
def read_stock(text: str) -> Optional[bool]:
|
||||
"""True/False only when the text says so; None when it does not."""
|
||||
if not text:
|
||||
return None
|
||||
if _OUT_OF_STOCK.search(text):
|
||||
return False
|
||||
if _IN_STOCK.search(text):
|
||||
return True
|
||||
return None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RatingReading:
|
||||
rating: Optional[Decimal]
|
||||
review_count: Optional[int]
|
||||
evidence: str # the exact substring the rating was read from ("" if none)
|
||||
|
||||
|
||||
# Only ratings the text states explicitly on a 5-point scale:
|
||||
# "4.3 out of 5 stars", "Rating: 4.3/5", "Rated 4.3 / 5", "4.3★", "4.3 ★ (1,234 ratings)"
|
||||
_RATING_PATTERNS = (
|
||||
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*out\s+of\s*5(?:\.0)?\b(?:\s*stars?)?", re.IGNORECASE),
|
||||
# "x/5" only with a rating word before it or "stars" after it - a bare
|
||||
# "1/5" is as likely a sensor size or a fraction.
|
||||
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\b", re.IGNORECASE),
|
||||
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*/\s*5(?:\.0)?\s*stars?\b", re.IGNORECASE),
|
||||
re.compile(r"\b([0-5](?:\.\d{1,2})?)\s*(?:★|☆|⭐)"),
|
||||
re.compile(r"\brat(?:ing|ed)\s*[:\-]?\s*([0-5](?:\.\d{1,2})?)\s*(?:stars?|★)", re.IGNORECASE),
|
||||
)
|
||||
_RATING_COUNT = re.compile(
|
||||
r"^[\s()\-|·,.:]*(?:stars?)?[\s()\-|·,.:]*(\d{1,3}(?:,\d{2,3})+|\d+)\s*(?:customer\s+)?(?:ratings?|reviews?|votes?)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def read_rating(text: str) -> RatingReading:
|
||||
"""The product rating a search title/snippet states, or None.
|
||||
|
||||
Only an explicit "x out of 5" / "x/5" / "x★" statement counts; bare
|
||||
numbers never do. If the text states two different ratings it is
|
||||
ambiguous (several products on one results page) and None is returned.
|
||||
"""
|
||||
if not text:
|
||||
return RatingReading(None, None, "")
|
||||
found = []
|
||||
for pattern in _RATING_PATTERNS:
|
||||
for m in pattern.finditer(text):
|
||||
try:
|
||||
value = Decimal(m.group(1))
|
||||
except InvalidOperation:
|
||||
continue
|
||||
if Decimal(0) < value <= Decimal(5):
|
||||
found.append((value, m))
|
||||
if not found or len({v for v, _ in found}) != 1:
|
||||
return RatingReading(None, None, "")
|
||||
value, m = min(found, key=lambda f: f[1].start())
|
||||
count = None
|
||||
tail = _RATING_COUNT.match(text[m.end(): m.end() + 40])
|
||||
if tail:
|
||||
count = int(tail.group(1).replace(",", ""))
|
||||
evidence = text[m.start(): m.end() + (tail.end() if tail else 0)].strip()
|
||||
return RatingReading(value, count, evidence)
|
||||
|
||||
|
||||
# Titles returned by search engines carry the site name; it is not part of the
|
||||
# product title.
|
||||
_TITLE_SUFFIX = re.compile(
|
||||
r"\s*(?:[|\-–:]\s*)?(?:buy\s+online.*|online\s+at\s+best\s+price.*|"
|
||||
r"at\s+best\s+price.*|price\s+in\s+india.*|"
|
||||
r"amazon\.in.*|flipkart(?:\.com)?.*|croma.*|reliance\s+digital.*|vijay\s+sales.*|"
|
||||
r"tata\s+cliq.*|poorvika.*|sangeetha.*|vasanth.*|viveks.*)$",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_TITLE_PREFIX = re.compile(r"^(?:buy\s+|amazon\.in\s*:\s*)", re.IGNORECASE)
|
||||
|
||||
|
||||
def clean_result_title(title: str) -> str:
|
||||
t = (title or "").strip()
|
||||
# Engines truncate with "..." and sometimes run several results' titles
|
||||
# together after it; everything past the first ellipsis is not this page.
|
||||
t = re.split(r"\s*(?:\.\.\.|…)", t, maxsplit=1)[0]
|
||||
t = _TITLE_PREFIX.sub("", t)
|
||||
t = _TITLE_SUFFIX.sub("", t)
|
||||
return t.strip(" -|:–")
|
||||
Reference in New Issue
Block a user