Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
0
backend/app/electronics/match/__init__.py
Normal file
0
backend/app/electronics/match/__init__.py
Normal file
124
backend/app/electronics/match/matcher.py
Normal file
124
backend/app/electronics/match/matcher.py
Normal file
@@ -0,0 +1,124 @@
|
||||
"""Link a listing to its canonical product (one real-world variant).
|
||||
|
||||
From most to least certain:
|
||||
1. GTIN - same barcode -> auto
|
||||
2. MPN - same manufacturer part number (laptops) -> auto
|
||||
3. variant key - same brand, model, RAM and storage -> auto
|
||||
4. fuzzy - model names ≥ AUTO_RATIO similar AND every
|
||||
hard attribute (RAM, storage, processor) equal -> auto
|
||||
≥ REVIEW_RATIO -> pending (review queue)
|
||||
5. otherwise a new product is created for the variant.
|
||||
|
||||
A listing that states too little to identify a variant (no storage on a
|
||||
phone title, for example) is stored but not linked to any product.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from decimal import Decimal
|
||||
from typing import List, Optional
|
||||
|
||||
from rapidfuzz import fuzz
|
||||
|
||||
from app.electronics.normalise.title_parser import laptop_line, processor_is_specific
|
||||
|
||||
AUTO_RATIO = 92
|
||||
REVIEW_RATIO = 85
|
||||
|
||||
|
||||
@dataclass
|
||||
class MatchDecision:
|
||||
product_id: Optional[int] # None -> create a new product
|
||||
method: str
|
||||
confidence: float
|
||||
review_status: str
|
||||
|
||||
|
||||
def _eq(a: Optional[Decimal], b: Optional[Decimal]) -> bool:
|
||||
if a is None or b is None:
|
||||
return a is None and b is None
|
||||
return Decimal(a) == Decimal(b)
|
||||
|
||||
|
||||
def _number_tokens(model_norm: str) -> set:
|
||||
"""Tokens that carry a digit ("s25", "a37", "15", "2a", "8a"). Two models
|
||||
whose number tokens differ are different products, however similar the
|
||||
rest of the name is: Galaxy S25 vs S26, A27 vs A37, iPhone 15 vs 16."""
|
||||
return {t for t in (model_norm or "").split() if any(ch.isdigit() for ch in t)}
|
||||
|
||||
|
||||
def _lines_compatible(a: str, b: str) -> bool:
|
||||
"""One model line is the other plus/minus extra words, and they agree on
|
||||
every number token they both carry ("15" is not "15s", "slim 3" is not
|
||||
"slim 5")."""
|
||||
ta, tb = set(a.split()), set(b.split())
|
||||
return bool(ta and tb) and (ta <= tb or tb <= ta)
|
||||
|
||||
|
||||
def decide(listing, candidates: List[dict]) -> Optional[MatchDecision]:
|
||||
"""`listing` is a models.Listing with variant_key/model_norm set;
|
||||
`candidates` are product rows of the same brand and category."""
|
||||
if not listing.variant_key:
|
||||
return None
|
||||
if listing.gtin:
|
||||
for c in candidates:
|
||||
if c.get("gtin") and c["gtin"] == listing.gtin:
|
||||
return MatchDecision(c["id"], "gtin", 0.99, "auto")
|
||||
if listing.model_number:
|
||||
for c in candidates:
|
||||
if c.get("mpn") and c["mpn"].lower() == listing.model_number.lower():
|
||||
return MatchDecision(c["id"], "mpn", 0.97, "auto")
|
||||
for c in candidates:
|
||||
if c["variant_key"] == listing.variant_key:
|
||||
return MatchDecision(c["id"], "variant_key", 0.95, "auto")
|
||||
|
||||
# Laptops: the same configuration (exact CPU model, RAM, storage) within a
|
||||
# compatible model line - "ideapad slim 3" and "ideapad slim 3 15amn8" -
|
||||
# is the same product. Two compatible candidates means the title is too
|
||||
# vague to choose ("pavilion" vs "pavilion 14" and "pavilion 15"): review.
|
||||
if listing.category == "laptops" and processor_is_specific(listing.processor) \
|
||||
and listing.ram_gb is not None and listing.storage_gb is not None:
|
||||
line = laptop_line(listing.model_norm)
|
||||
same_config = [
|
||||
c for c in candidates
|
||||
if c.get("processor") == listing.processor
|
||||
and _eq(c.get("ram_gb"), listing.ram_gb) and _eq(c.get("storage_gb"), listing.storage_gb)
|
||||
and _lines_compatible(line, laptop_line(c["model_norm"]))
|
||||
]
|
||||
if len(same_config) == 1:
|
||||
return MatchDecision(same_config[0]["id"], "variant_key", 0.85, "auto")
|
||||
if len(same_config) > 1:
|
||||
return MatchDecision(same_config[0]["id"], "variant_key", 0.6, "pending")
|
||||
|
||||
# One side does not state the RAM ("Apple iPhone 15 (128 GB)"). Same model
|
||||
# and storage with exactly one candidate is the same variant; with several
|
||||
# candidates it is ambiguous and goes to review.
|
||||
same_model = [c for c in candidates
|
||||
if c["model_norm"] == listing.model_norm and _eq(c.get("storage_gb"), listing.storage_gb)
|
||||
and (c.get("ram_gb") is None) != (listing.ram_gb is None)]
|
||||
if len(same_model) == 1:
|
||||
return MatchDecision(same_model[0]["id"], "variant_key", 0.85, "auto")
|
||||
if len(same_model) > 1:
|
||||
return MatchDecision(same_model[0]["id"], "variant_key", 0.6, "pending")
|
||||
|
||||
best, best_score = None, 0.0
|
||||
for c in candidates:
|
||||
if not (_eq(c.get("storage_gb"), listing.storage_gb) and _eq(c.get("ram_gb"), listing.ram_gb)):
|
||||
continue
|
||||
if listing.category == "laptops" and (c.get("processor") or listing.processor) and c.get("processor") != listing.processor:
|
||||
continue
|
||||
if _number_tokens(c["model_norm"]) != _number_tokens(listing.model_norm):
|
||||
continue
|
||||
score = fuzz.token_set_ratio(c["model_norm"], listing.model_norm or "")
|
||||
# token_set_ratio treats "galaxy s24" and "galaxy s24 ultra" as a
|
||||
# subset match (100). Different words mean different models.
|
||||
extra = set((listing.model_norm or "").split()) ^ set(c["model_norm"].split())
|
||||
if extra:
|
||||
score = min(score, fuzz.ratio(c["model_norm"], listing.model_norm or ""))
|
||||
if score > best_score:
|
||||
best, best_score = c, score
|
||||
if best is not None and best_score >= AUTO_RATIO:
|
||||
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.9, 2), "auto")
|
||||
if best is not None and best_score >= REVIEW_RATIO:
|
||||
return MatchDecision(best["id"], "fuzzy", round(best_score / 100 * 0.8, 2), "pending")
|
||||
return MatchDecision(None, "variant_key", 0.9, "auto")
|
||||
105
backend/app/electronics/match/rematch.py
Normal file
105
backend/app/electronics/match/rematch.py
Normal file
@@ -0,0 +1,105 @@
|
||||
"""Rebuild canonical products for a category from the listings already stored.
|
||||
|
||||
Products and listing links are derived data: every fact lives on the listing
|
||||
(title, snippet evidence, specs, URL). When the parsing or matching rules
|
||||
improve, this re-runs them over the stored listings - no network requests -
|
||||
and keeps each image attached to the listing it was found on.
|
||||
|
||||
Review decisions (approved/rejected links) are lost, because the products they
|
||||
pointed at are rebuilt; uncertain matches simply come back to the queue.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from decimal import Decimal
|
||||
from typing import Dict, List
|
||||
|
||||
from app.electronics.db import repository as repo
|
||||
from app.electronics.db.connection import connect, transaction
|
||||
from app.electronics.match.matcher import decide
|
||||
from app.electronics.models import Listing
|
||||
from app.electronics.normalise.title_parser import fill_from_context, parse_title, variant_key
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_ORDER = {"brand_official": 0, "scraped_page": 1, "search_snippet": 2}
|
||||
|
||||
|
||||
def _listing_from_row(row: dict, category: str) -> Listing:
|
||||
parsed = parse_title(row["title"], category, expected_brand=row["brand_slug"])
|
||||
snippet = ""
|
||||
if row["source_type"] == "search_snippet" and " — " in row["evidence_text"]:
|
||||
snippet = row["evidence_text"].split(" — ", 1)[1].split(" || ", 1)[0]
|
||||
raw = row["specs_raw"] or {}
|
||||
spec_texts = tuple(str(v) for k, v in raw.items() if "processor" in k.lower() or "cpu" in k.lower())
|
||||
spec_texts += (str((row["specs"] or {}).get("processor") or ""),)
|
||||
fill_from_context(parsed, category, snippet=snippet, spec_texts=spec_texts)
|
||||
l = Listing(
|
||||
site_domain=row["domain"], source_sku=row["source_sku"], source_url=row["source_url"],
|
||||
source_type=row["source_type"], brand_slug=row["brand_slug"], category=category,
|
||||
title=row["title"], evidence_text=row["evidence_text"], confidence=float(row["confidence"]),
|
||||
parser=row["parser"], family=parsed.brand.family if parsed.brand else row["family"],
|
||||
model=parsed.model, model_number=row["model_number"] or parsed.mpn,
|
||||
ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb, colour=row["colour"],
|
||||
gtin=row["gtin"], specs=row["specs"] or {},
|
||||
)
|
||||
l.model_norm, l.processor = parsed.model_norm, parsed.processor
|
||||
l.variant_key = variant_key(parsed, category) if parsed.brand else None
|
||||
return l
|
||||
|
||||
|
||||
def rematch(category: str) -> Dict[str, int]:
|
||||
stats: Dict[str, int] = {"listings": 0, "linked": 0, "pending": 0, "unlinked": 0, "products": 0, "images": 0}
|
||||
with connect() as conn:
|
||||
rows = conn.execute(
|
||||
"""
|
||||
SELECT l.*, b.slug AS brand_slug, s.domain
|
||||
FROM elec.source_listing l
|
||||
JOIN elec.brand b ON b.id = l.brand_id
|
||||
JOIN elec.site s ON s.id = l.site_id
|
||||
JOIN elec.category c ON c.id = l.category_id
|
||||
WHERE c.slug = %s
|
||||
""",
|
||||
(category,),
|
||||
).fetchall()
|
||||
images = conn.execute(
|
||||
"""
|
||||
SELECT i.url, i.source_listing_id, i.source_type, i.rank FROM elec.product_image i
|
||||
JOIN elec.product p ON p.id = i.product_id JOIN elec.category c ON c.id = p.category_id
|
||||
WHERE c.slug = %s
|
||||
""",
|
||||
(category,),
|
||||
).fetchall()
|
||||
with transaction() as conn:
|
||||
# Maps and images cascade from the products.
|
||||
conn.execute(
|
||||
"DELETE FROM elec.product p USING elec.category c WHERE c.id = p.category_id AND c.slug = %s",
|
||||
(category,),
|
||||
)
|
||||
|
||||
ids = repo.id_maps()
|
||||
product_of_listing: Dict[int, int] = {}
|
||||
rows.sort(key=lambda r: (_ORDER.get(r["source_type"], 9), r["id"]))
|
||||
for row in rows:
|
||||
stats["listings"] += 1
|
||||
listing = _listing_from_row(row, category)
|
||||
decision = decide(listing, repo.product_candidates(listing.brand_slug, category))
|
||||
if decision is None:
|
||||
stats["unlinked"] += 1
|
||||
continue
|
||||
product_id = decision.product_id or repo.create_product(listing, ids)
|
||||
stats["products"] += decision.product_id is None
|
||||
repo.map_listing(row["id"], product_id, decision.method, decision.confidence, decision.review_status)
|
||||
product_of_listing[row["id"]] = product_id
|
||||
if decision.review_status == "pending":
|
||||
stats["pending"] += 1
|
||||
else:
|
||||
stats["linked"] += 1
|
||||
repo.merge_product_specs(product_id, listing.specs, {}, listing.source_url)
|
||||
for img in images:
|
||||
pid = product_of_listing.get(img["source_listing_id"])
|
||||
if pid is not None:
|
||||
repo.add_image(pid, img["url"], img["source_listing_id"], img["source_type"], img["rank"])
|
||||
stats["images"] += 1
|
||||
stats.update({f"products_{k}": v for k, v in repo.refresh_verification().items()})
|
||||
return stats
|
||||
Reference in New Issue
Block a user