Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
125 lines
5.7 KiB
Python
125 lines
5.7 KiB
Python
"""Link a listing to its canonical product (one real-world variant).
|
|
|
|
From most to least certain:
|
|
1. GTIN - same barcode -> auto
|
|
2. MPN - same manufacturer part number (laptops) -> auto
|
|
3. variant key - same brand, model, RAM and storage -> auto
|
|
4. fuzzy - model names ≥ AUTO_RATIO similar AND every
|
|
hard attribute (RAM, storage, processor) equal -> auto
|
|
≥ REVIEW_RATIO -> pending (review queue)
|
|
5. otherwise a new product is created for the variant.
|
|
|
|
A listing that states too little to identify a variant (no storage on a
|
|
phone title, for example) is stored but not linked to any product.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
from decimal import Decimal
|
|
from typing import List, Optional
|
|
|
|
from rapidfuzz import fuzz
|
|
|
|
from app.electronics.normalise.title_parser import laptop_line, processor_is_specific
|
|
|
|
AUTO_RATIO = 92
|
|
REVIEW_RATIO = 85
|
|
|
|
|
|
@dataclass
|
|
class MatchDecision:
|
|
product_id: Optional[int] # None -> create a new product
|
|
method: str
|
|
confidence: float
|
|
review_status: str
|
|
|
|
|
|
def _eq(a: Optional[Decimal], b: Optional[Decimal]) -> bool:
|
|
if a is None or b is None:
|
|
return a is None and b is None
|
|
return Decimal(a) == Decimal(b)
|
|
|
|
|
|
def _number_tokens(model_norm: str) -> set:
|
|
"""Tokens that carry a digit ("s25", "a37", "15", "2a", "8a"). Two models
|
|
whose number tokens differ are different products, however similar the
|
|
rest of the name is: Galaxy S25 vs S26, A27 vs A37, iPhone 15 vs 16."""
|
|
return {t for t in (model_norm or "").split() if any(ch.isdigit() for ch in t)}
|
|
|
|
|
|
def _lines_compatible(a: str, b: str) -> bool:
|
|
"""One model line is the other plus/minus extra words, and they agree on
|
|
every number token they both carry ("15" is not "15s", "slim 3" is not
|
|
"slim 5")."""
|
|
ta, tb = set(a.split()), set(b.split())
|
|
return bool(ta and tb) and (ta <= tb or tb <= ta)
|
|
|
|
|
|
def decide(listing, candidates: List[dict]) -> Optional[MatchDecision]:
|
|
"""`listing` is a models.Listing with variant_key/model_norm set;
|
|
`candidates` are product rows of the same brand and category."""
|
|
if not listing.variant_key:
|
|
return None
|
|
if listing.gtin:
|
|
for c in candidates:
|
|
if c.get("gtin") and c["gtin"] == listing.gtin:
|
|
return MatchDecision(c["id"], "gtin", 0.99, "auto")
|
|
if listing.model_number:
|
|
for c in candidates:
|
|
if c.get("mpn") and c["mpn"].lower() == listing.model_number.lower():
|
|
return MatchDecision(c["id"], "mpn", 0.97, "auto")
|
|
for c in candidates:
|
|
if c["variant_key"] == listing.variant_key:
|
|
return MatchDecision(c["id"], "variant_key", 0.95, "auto")
|
|
|
|
# Laptops: the same configuration (exact CPU model, RAM, storage) within a
|
|
# compatible model line - "ideapad slim 3" and "ideapad slim 3 15amn8" -
|
|
# is the same product. Two compatible candidates means the title is too
|
|
# vague to choose ("pavilion" vs "pavilion 14" and "pavilion 15"): review.
|
|
if listing.category == "laptops" and processor_is_specific(listing.processor) \
|
|
and listing.ram_gb is not None and listing.storage_gb is not None:
|
|
line = laptop_line(listing.model_norm)
|
|
same_config = [
|
|
c for c in candidates
|
|
if c.get("processor") == listing.processor
|
|
and _eq(c.get("ram_gb"), listing.ram_gb) and _eq(c.get("storage_gb"), listing.storage_gb)
|
|
and _lines_compatible(line, laptop_line(c["model_norm"]))
|
|
]
|
|
if len(same_config) == 1:
|
|
return MatchDecision(same_config[0]["id"], "variant_key", 0.85, "auto")
|
|
if len(same_config) > 1:
|
|
return MatchDecision(same_config[0]["id"], "variant_key", 0.6, "pending")
|
|
|
|
# One side does not state the RAM ("Apple iPhone 15 (128 GB)"). Same model
|
|
# and storage with exactly one candidate is the same variant; with several
|
|
# candidates it is ambiguous and goes to review.
|
|
same_model = [c for c in candidates
|
|
if c["model_norm"] == listing.model_norm and _eq(c.get("storage_gb"), listing.storage_gb)
|
|
and (c.get("ram_gb") is None) != (listing.ram_gb is None)]
|
|
if len(same_model) == 1:
|
|
return MatchDecision(same_model[0]["id"], "variant_key", 0.85, "auto")
|
|
if len(same_model) > 1:
|
|
return MatchDecision(same_model[0]["id"], "variant_key", 0.6, "pending")
|
|
|
|
best, best_score = None, 0.0
|
|
for c in candidates:
|
|
if not (_eq(c.get("storage_gb"), listing.storage_gb) and _eq(c.get("ram_gb"), listing.ram_gb)):
|
|
continue
|
|
if listing.category == "laptops" and (c.get("processor") or listing.processor) and c.get("processor") != listing.processor:
|
|
continue
|
|
if _number_tokens(c["model_norm"]) != _number_tokens(listing.model_norm):
|
|
continue
|
|
score = fuzz.token_set_ratio(c["model_norm"], listing.model_norm or "")
|
|
# token_set_ratio treats "galaxy s24" and "galaxy s24 ultra" as a
|
|
# subset match (100). Different words mean different models.
|
|
extra = set((listing.model_norm or "").split()) ^ set(c["model_norm"].split())
|
|
if extra:
|
|
score = min(score, fuzz.ratio(c["model_norm"], listing.model_norm or ""))
|
|
if score > best_score:
|
|
best, best_score = c, score
|
|
if best is not None and best_score >= AUTO_RATIO:
|
|
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.9, 2), "auto")
|
|
if best is not None and best_score >= REVIEW_RATIO:
|
|
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.8, 2), "pending")
|
|
return MatchDecision(None, "variant_key", 0.9, "auto")
|