Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
97 lines
3.6 KiB
Python
97 lines
3.6 KiB
Python
"""Which real customer reviews to show for a product, and in what mix.
|
|
|
|
Every review passed in here was read from a product page's own schema.org
|
|
data (see extract/jsonld.py); this module only classifies and selects - it
|
|
never writes, rewrites or summarises review text.
|
|
|
|
Sentiment is the reviewer's own star rating, nothing inferred:
|
|
>= 4 positive, >= 3 neutral, < 3 negative.
|
|
|
|
The mix follows the product's overall rating, so the reviews shown read like
|
|
the rating does:
|
|
rating >= 4.0 mostly positive, some neutral, a little negative
|
|
3.0 < rating < 4.0 mostly neutral, some positive, a little negative
|
|
rating <= 3.0 mostly negative, a little positive and neutral
|
|
When a group has too few reviews its slots go to the other groups, in the
|
|
same priority order. Nothing is ever padded: if only 3 real reviews exist,
|
|
3 are shown.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from decimal import Decimal
|
|
from typing import Any, Dict, List, Optional, Sequence, Tuple
|
|
|
|
POSITIVE, NEUTRAL, NEGATIVE = "positive", "neutral", "negative"
|
|
MAX_REVIEWS = 10
|
|
|
|
# (group, share of MAX_REVIEWS), highest priority first.
|
|
_MIX_HIGH: Tuple[Tuple[str, int], ...] = ((POSITIVE, 6), (NEUTRAL, 3), (NEGATIVE, 1))
|
|
_MIX_MID: Tuple[Tuple[str, int], ...] = ((NEUTRAL, 5), (POSITIVE, 3), (NEGATIVE, 2))
|
|
_MIX_LOW: Tuple[Tuple[str, int], ...] = ((NEGATIVE, 6), (POSITIVE, 2), (NEUTRAL, 2))
|
|
|
|
|
|
def sentiment_for(rating: Any) -> Optional[str]:
|
|
"""The group a reviewer's own star rating puts a review in; None when the
|
|
review states no rating."""
|
|
if rating is None:
|
|
return None
|
|
try:
|
|
value = Decimal(str(rating))
|
|
except Exception: # noqa: BLE001
|
|
return None
|
|
if value >= 4:
|
|
return POSITIVE
|
|
if value >= 3:
|
|
return NEUTRAL
|
|
return NEGATIVE
|
|
|
|
|
|
def mix_for(product_rating: Any) -> Tuple[Tuple[str, int], ...]:
|
|
if product_rating is None:
|
|
return _MIX_MID # no overall rating stated: a balanced view
|
|
value = Decimal(str(product_rating))
|
|
if value >= 4:
|
|
return _MIX_HIGH
|
|
if value > 3:
|
|
return _MIX_MID
|
|
return _MIX_LOW
|
|
|
|
|
|
def _rank_key(review: Dict[str, Any]) -> tuple:
|
|
# Newest first (ISO dates sort as text), then the more substantial review.
|
|
return (str(review.get("review_date") or ""), len(review.get("body") or ""))
|
|
|
|
|
|
def select_reviews(product_rating: Any, reviews: Sequence[Dict[str, Any]],
|
|
max_n: int = MAX_REVIEWS) -> List[Dict[str, Any]]:
|
|
"""Up to `max_n` of `reviews`, mixed by sentiment as described above.
|
|
|
|
Reviews without a star rating have no sentiment and are not shown: there
|
|
is no honest way to place them in the mix.
|
|
"""
|
|
groups: Dict[str, List[Dict[str, Any]]] = {POSITIVE: [], NEUTRAL: [], NEGATIVE: []}
|
|
seen = set()
|
|
for r in reviews:
|
|
s = r.get("sentiment") or sentiment_for(r.get("rating"))
|
|
key = (r.get("body") or "").strip().lower()
|
|
if s is None or not key or key in seen:
|
|
continue
|
|
seen.add(key)
|
|
groups[s].append({**r, "sentiment": s})
|
|
for g in groups.values():
|
|
g.sort(key=_rank_key, reverse=True)
|
|
|
|
mix = mix_for(product_rating)
|
|
scale = max_n / MAX_REVIEWS
|
|
quota = {g: int(round(n * scale)) for g, n in mix}
|
|
picked: Dict[str, List[Dict[str, Any]]] = {g: groups[g][: quota[g]] for g, _ in mix}
|
|
# Hand unused slots to the other groups, in priority order.
|
|
spare = max_n - sum(len(v) for v in picked.values())
|
|
for g, _ in mix:
|
|
if spare <= 0:
|
|
break
|
|
extra = groups[g][len(picked[g]): len(picked[g]) + spare]
|
|
picked[g].extend(extra)
|
|
spare -= len(extra)
|
|
return [r for g, _ in mix for r in picked[g]]
|