"""Link a listing to its canonical product (one real-world variant). From most to least certain: 1. GTIN - same barcode -> auto 2. MPN - same manufacturer part number (laptops) -> auto 3. variant key - same brand, model, RAM and storage -> auto 4. fuzzy - model names ≥ AUTO_RATIO similar AND every hard attribute (RAM, storage, processor) equal -> auto ≥ REVIEW_RATIO -> pending (review queue) 5. otherwise a new product is created for the variant. A listing that states too little to identify a variant (no storage on a phone title, for example) is stored but not linked to any product. """ from __future__ import annotations from dataclasses import dataclass from decimal import Decimal from typing import List, Optional from rapidfuzz import fuzz from app.electronics.normalise.title_parser import laptop_line, processor_is_specific AUTO_RATIO = 92 REVIEW_RATIO = 85 @dataclass class MatchDecision: product_id: Optional[int] # None -> create a new product method: str confidence: float review_status: str def _eq(a: Optional[Decimal], b: Optional[Decimal]) -> bool: if a is None or b is None: return a is None and b is None return Decimal(a) == Decimal(b) def _number_tokens(model_norm: str) -> set: """Tokens that carry a digit ("s25", "a37", "15", "2a", "8a"). Two models whose number tokens differ are different products, however similar the rest of the name is: Galaxy S25 vs S26, A27 vs A37, iPhone 15 vs 16.""" return {t for t in (model_norm or "").split() if any(ch.isdigit() for ch in t)} def _lines_compatible(a: str, b: str) -> bool: """One model line is the other plus/minus extra words, and they agree on every number token they both carry ("15" is not "15s", "slim 3" is not "slim 5").""" ta, tb = set(a.split()), set(b.split()) return bool(ta and tb) and (ta <= tb or tb <= ta) def decide(listing, candidates: List[dict]) -> Optional[MatchDecision]: """`listing` is a models.Listing with variant_key/model_norm set; `candidates` are product rows of the same brand and category.""" if not listing.variant_key: return None if listing.gtin: for c in candidates: if c.get("gtin") and c["gtin"] == listing.gtin: return MatchDecision(c["id"], "gtin", 0.99, "auto") if listing.model_number: for c in candidates: if c.get("mpn") and c["mpn"].lower() == listing.model_number.lower(): return MatchDecision(c["id"], "mpn", 0.97, "auto") for c in candidates: if c["variant_key"] == listing.variant_key: return MatchDecision(c["id"], "variant_key", 0.95, "auto") # Laptops: the same configuration (exact CPU model, RAM, storage) within a # compatible model line - "ideapad slim 3" and "ideapad slim 3 15amn8" - # is the same product. Two compatible candidates means the title is too # vague to choose ("pavilion" vs "pavilion 14" and "pavilion 15"): review. if listing.category == "laptops" and processor_is_specific(listing.processor) \ and listing.ram_gb is not None and listing.storage_gb is not None: line = laptop_line(listing.model_norm) same_config = [ c for c in candidates if c.get("processor") == listing.processor and _eq(c.get("ram_gb"), listing.ram_gb) and _eq(c.get("storage_gb"), listing.storage_gb) and _lines_compatible(line, laptop_line(c["model_norm"])) ] if len(same_config) == 1: return MatchDecision(same_config[0]["id"], "variant_key", 0.85, "auto") if len(same_config) > 1: return MatchDecision(same_config[0]["id"], "variant_key", 0.6, "pending") # One side does not state the RAM ("Apple iPhone 15 (128 GB)"). Same model # and storage with exactly one candidate is the same variant; with several # candidates it is ambiguous and goes to review. same_model = [c for c in candidates if c["model_norm"] == listing.model_norm and _eq(c.get("storage_gb"), listing.storage_gb) and (c.get("ram_gb") is None) != (listing.ram_gb is None)] if len(same_model) == 1: return MatchDecision(same_model[0]["id"], "variant_key", 0.85, "auto") if len(same_model) > 1: return MatchDecision(same_model[0]["id"], "variant_key", 0.6, "pending") best, best_score = None, 0.0 for c in candidates: if not (_eq(c.get("storage_gb"), listing.storage_gb) and _eq(c.get("ram_gb"), listing.ram_gb)): continue if listing.category == "laptops" and (c.get("processor") or listing.processor) and c.get("processor") != listing.processor: continue if _number_tokens(c["model_norm"]) != _number_tokens(listing.model_norm): continue score = fuzz.token_set_ratio(c["model_norm"], listing.model_norm or "") # token_set_ratio treats "galaxy s24" and "galaxy s24 ultra" as a # subset match (100). Different words mean different models. extra = set((listing.model_norm or "").split()) ^ set(c["model_norm"].split()) if extra: score = min(score, fuzz.ratio(c["model_norm"], listing.model_norm or "")) if score > best_score: best, best_score = c, score if best is not None and best_score >= AUTO_RATIO: return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.9, 2), "auto") if best is not None and best_score >= REVIEW_RATIO: return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.8, 2), "pending") return MatchDecision(None, "variant_key", 0.9, "auto")