Verified catalogue of mobiles and laptops sold in India, collected from real retail listings (FastAPI backend, React frontend, Postgres/pgvector). - REST API under /api/elec (read-only catalogue; admin endpoints need login) - MCP server (FastMCP) at /mcp/ with list_categories, search_products, get_product and price_history tools - Real ratings and reviews read from product pages and search results - Production Dockerfile (requirements-api.txt, no PyTorch) and .env.production.example; remote database only via an explicit ELEC_ALLOW_REMOTE_DB host/name allowlist - docs/API.md: endpoint and MCP reference with live examples Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
106 lines
4.8 KiB
Python
106 lines
4.8 KiB
Python
"""Rebuild canonical products for a category from the listings already stored.
|
|
|
|
Products and listing links are derived data: every fact lives on the listing
|
|
(title, snippet evidence, specs, URL). When the parsing or matching rules
|
|
improve, this re-runs them over the stored listings - no network requests -
|
|
and keeps each image attached to the listing it was found on.
|
|
|
|
Review decisions (approved/rejected links) are lost, because the products they
|
|
pointed at are rebuilt; uncertain matches simply come back to the queue.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
from decimal import Decimal
|
|
from typing import Dict, List
|
|
|
|
from app.electronics.db import repository as repo
|
|
from app.electronics.db.connection import connect, transaction
|
|
from app.electronics.match.matcher import decide
|
|
from app.electronics.models import Listing
|
|
from app.electronics.normalise.title_parser import fill_from_context, parse_title, variant_key
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_ORDER = {"brand_official": 0, "scraped_page": 1, "search_snippet": 2}
|
|
|
|
|
|
def _listing_from_row(row: dict, category: str) -> Listing:
|
|
parsed = parse_title(row["title"], category, expected_brand=row["brand_slug"])
|
|
snippet = ""
|
|
if row["source_type"] == "search_snippet" and " — " in row["evidence_text"]:
|
|
snippet = row["evidence_text"].split(" — ", 1)[1].split(" || ", 1)[0]
|
|
raw = row["specs_raw"] or {}
|
|
spec_texts = tuple(str(v) for k, v in raw.items() if "processor" in k.lower() or "cpu" in k.lower())
|
|
spec_texts += (str((row["specs"] or {}).get("processor") or ""),)
|
|
fill_from_context(parsed, category, snippet=snippet, spec_texts=spec_texts)
|
|
l = Listing(
|
|
site_domain=row["domain"], source_sku=row["source_sku"], source_url=row["source_url"],
|
|
source_type=row["source_type"], brand_slug=row["brand_slug"], category=category,
|
|
title=row["title"], evidence_text=row["evidence_text"], confidence=float(row["confidence"]),
|
|
parser=row["parser"], family=parsed.brand.family if parsed.brand else row["family"],
|
|
model=parsed.model, model_number=row["model_number"] or parsed.mpn,
|
|
ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb, colour=row["colour"],
|
|
gtin=row["gtin"], specs=row["specs"] or {},
|
|
)
|
|
l.model_norm, l.processor = parsed.model_norm, parsed.processor
|
|
l.variant_key = variant_key(parsed, category) if parsed.brand else None
|
|
return l
|
|
|
|
|
|
def rematch(category: str) -> Dict[str, int]:
|
|
stats: Dict[str, int] = {"listings": 0, "linked": 0, "pending": 0, "unlinked": 0, "products": 0, "images": 0}
|
|
with connect() as conn:
|
|
rows = conn.execute(
|
|
"""
|
|
SELECT l.*, b.slug AS brand_slug, s.domain
|
|
FROM elec.source_listing l
|
|
JOIN elec.brand b ON b.id = l.brand_id
|
|
JOIN elec.site s ON s.id = l.site_id
|
|
JOIN elec.category c ON c.id = l.category_id
|
|
WHERE c.slug = %s
|
|
""",
|
|
(category,),
|
|
).fetchall()
|
|
images = conn.execute(
|
|
"""
|
|
SELECT i.url, i.source_listing_id, i.source_type, i.rank FROM elec.product_image i
|
|
JOIN elec.product p ON p.id = i.product_id JOIN elec.category c ON c.id = p.category_id
|
|
WHERE c.slug = %s
|
|
""",
|
|
(category,),
|
|
).fetchall()
|
|
with transaction() as conn:
|
|
# Maps and images cascade from the products.
|
|
conn.execute(
|
|
"DELETE FROM elec.product p USING elec.category c WHERE c.id = p.category_id AND c.slug = %s",
|
|
(category,),
|
|
)
|
|
|
|
ids = repo.id_maps()
|
|
product_of_listing: Dict[int, int] = {}
|
|
rows.sort(key=lambda r: (_ORDER.get(r["source_type"], 9), r["id"]))
|
|
for row in rows:
|
|
stats["listings"] += 1
|
|
listing = _listing_from_row(row, category)
|
|
decision = decide(listing, repo.product_candidates(listing.brand_slug, category))
|
|
if decision is None:
|
|
stats["unlinked"] += 1
|
|
continue
|
|
product_id = decision.product_id or repo.create_product(listing, ids)
|
|
stats["products"] += decision.product_id is None
|
|
repo.map_listing(row["id"], product_id, decision.method, decision.confidence, decision.review_status)
|
|
product_of_listing[row["id"]] = product_id
|
|
if decision.review_status == "pending":
|
|
stats["pending"] += 1
|
|
else:
|
|
stats["linked"] += 1
|
|
repo.merge_product_specs(product_id, listing.specs, {}, listing.source_url)
|
|
for img in images:
|
|
pid = product_of_listing.get(img["source_listing_id"])
|
|
if pid is not None:
|
|
repo.add_image(pid, img["url"], img["source_listing_id"], img["source_type"], img["rank"])
|
|
stats["images"] += 1
|
|
stats.update({f"products_{k}": v for k, v in repo.refresh_verification().items()})
|
|
return stats
|