backend updates on recommendation system

This commit is contained in:
sriram
2026-10-07 11:03:42 +05:30
parent 82f5db1250
commit 71dbb2a6e9
6 changed files with 599 additions and 6 deletions

View File

@@ -305,6 +305,96 @@ def product_rating_and_reviews(conn, product_id: int) -> Dict[str, Any]:
return {"sources": [dict(s) for s in sources], "reviews": [dict(r) for r in reviews]}
def rating_sources_for(conn, product_ids: List[int]) -> Dict[int, List[dict]]:
"""product_rating_and_reviews' per-platform ratings for many products at once."""
out: Dict[int, List[dict]] = {pid: [] for pid in product_ids}
rows = conn.execute(
"SELECT a.product_id, a.site, a.source_url, l.rating, l.review_count, l.rating_breakdown "
"FROM elec.v_product_availability a JOIN elec.source_listing l ON l.id = a.listing_id "
"WHERE a.product_id = ANY(%s) AND l.rating > 0 ORDER BY l.review_count DESC NULLS LAST, a.site",
(list(product_ids),),
).fetchall()
for r in rows:
out[r["product_id"]].append(dict(r))
return out
# p is a variant of t when the two differ only in RAM/storage: same brand and
# model and the same processor. One laptop model line ("HP 15") spans many
# CPUs, so laptops count as variants only when both state the same processor;
# phones state none, so for them the model alone decides. Written so it is
# never NULL: NOT NULL would drop the product from recommendations too.
_SAME_MODEL = """(p.brand_id = t.brand_id AND p.model_norm = t.model_norm
AND p.processor IS NOT DISTINCT FROM t.processor
AND (p.processor IS NOT NULL
OR t.category_id IS DISTINCT FROM (SELECT id FROM elec.category WHERE slug = 'laptops')))"""
# Products that may be recommended for target t: verified, same category, not
# t itself or another RAM/storage variant of the same model, with an in-stock
# (or unknown-stock) best price within +/- %(band)s of t's own (no limit when t
# has no price).
_RECOMMENDABLE = """
FROM elec.product t
JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id
AND p.verification_status = 'verified'
AND NOT """ + _SAME_MODEL + """
JOIN elec.v_best_price bp ON bp.product_id = p.id AND bp.in_stock IS DISTINCT FROM FALSE
LEFT JOIN elec.v_best_price tp ON tp.product_id = t.id
WHERE t.id = %(pid)s
AND (tp.price IS NULL OR bp.price BETWEEN tp.price * (1 - %(band)s) AND tp.price * (1 + %(band)s))
"""
_TN_ONLY = (" AND EXISTS (SELECT 1 FROM elec.v_product_availability a"
" WHERE a.product_id = p.id AND a.site_region = 'TN')")
def similar_products(conn, product_id: int, *, band: float, limit: int = 30, tn_only: bool = False) -> List[dict]:
"""Recommendable products closest to this one by embedding (cosine), with
their best price. Empty when the product has no embedding yet.
An exact scan, not the HNSW index: the category/stock filters would make an
approximate index search drop matches, and a category is small enough."""
sql = ("SELECT p.id AS product_id, 1 - (p.embedding <=> t.embedding) AS similarity, bp.price AS best_price"
+ _RECOMMENDABLE + " AND t.embedding IS NOT NULL AND p.embedding IS NOT NULL"
+ (_TN_ONLY if tn_only else "")
+ " ORDER BY p.embedding <=> t.embedding LIMIT %(limit)s")
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band, "limit": limit})]
def rated_products(conn, product_id: int, *, band: float, tn_only: bool = False) -> List[dict]:
"""Recommendable products that at least one platform has rated."""
sql = ("SELECT p.id AS product_id, bp.price AS best_price" + _RECOMMENDABLE
+ " AND EXISTS (SELECT 1 FROM elec.v_product_availability a JOIN elec.source_listing l"
" ON l.id = a.listing_id WHERE a.product_id = p.id AND l.rating > 0)"
+ (_TN_ONLY if tn_only else ""))
return [dict(r) for r in conn.execute(sql, {"pid": product_id, "band": band})]
def review_sentiment_counts(conn, product_ids: List[int]) -> Dict[int, Dict[str, int]]:
"""How many stored reviews of each product are positive / neutral / negative."""
out: Dict[int, Dict[str, int]] = {pid: {} for pid in product_ids}
rows = conn.execute(
"SELECT a.product_id, r.sentiment, count(*)::int AS n "
"FROM elec.v_product_availability a JOIN elec.listing_review r ON r.listing_id = a.listing_id "
"WHERE a.product_id = ANY(%s) AND r.sentiment IS NOT NULL GROUP BY 1, 2",
(list(product_ids),),
).fetchall()
for r in rows:
out[r["product_id"]][r["sentiment"]] = r["n"]
return out
def other_variants(conn, product_id: int) -> List[dict]:
"""The same model's other verified RAM/storage variants (see _SAME_MODEL)."""
return [dict(r) for r in conn.execute(
"SELECT v.product_id, v.display_name, v.ram_gb, v.storage_gb, v.best_price "
"FROM elec.product t JOIN elec.product p ON p.category_id = t.category_id AND p.id <> t.id "
"AND " + _SAME_MODEL + " "
"JOIN elec.v_brand_catalog v ON v.product_id = p.id "
"WHERE t.id = %s ORDER BY v.ram_gb NULLS LAST, v.storage_gb NULLS LAST, v.display_name",
(product_id,),
)]
def listings_for_review_backfill(category: Optional[str] = None) -> List[dict]:
"""Page-read listings of verified products, for re-reading ratings/reviews."""
sql = (