"""Rebuild canonical products for a category from the listings already stored. Products and listing links are derived data: every fact lives on the listing (title, snippet evidence, specs, URL). When the parsing or matching rules improve, this re-runs them over the stored listings - no network requests - and keeps each image attached to the listing it was found on. Review decisions (approved/rejected links) are lost, because the products they pointed at are rebuilt; uncertain matches simply come back to the queue. """ from __future__ import annotations import logging from decimal import Decimal from typing import Dict, List from app.electronics.db import repository as repo from app.electronics.db.connection import connect, transaction from app.electronics.match.matcher import decide from app.electronics.models import Listing from app.electronics.normalise.title_parser import fill_from_context, parse_title, variant_key logger = logging.getLogger(__name__) _ORDER = {"brand_official": 0, "scraped_page": 1, "search_snippet": 2} def _listing_from_row(row: dict, category: str) -> Listing: parsed = parse_title(row["title"], category, expected_brand=row["brand_slug"]) snippet = "" if row["source_type"] == "search_snippet" and " — " in row["evidence_text"]: snippet = row["evidence_text"].split(" — ", 1)[1].split(" || ", 1)[0] raw = row["specs_raw"] or {} spec_texts = tuple(str(v) for k, v in raw.items() if "processor" in k.lower() or "cpu" in k.lower()) spec_texts += (str((row["specs"] or {}).get("processor") or ""),) fill_from_context(parsed, category, snippet=snippet, spec_texts=spec_texts) l = Listing( site_domain=row["domain"], source_sku=row["source_sku"], source_url=row["source_url"], source_type=row["source_type"], brand_slug=row["brand_slug"], category=category, title=row["title"], evidence_text=row["evidence_text"], confidence=float(row["confidence"]), parser=row["parser"], family=parsed.brand.family if parsed.brand else row["family"], model=parsed.model, model_number=row["model_number"] or parsed.mpn, ram_gb=parsed.ram_gb, storage_gb=parsed.storage_gb, colour=row["colour"], gtin=row["gtin"], specs=row["specs"] or {}, ) l.model_norm, l.processor = parsed.model_norm, parsed.processor l.variant_key = variant_key(parsed, category) if parsed.brand else None return l def rematch(category: str) -> Dict[str, int]: stats: Dict[str, int] = {"listings": 0, "linked": 0, "pending": 0, "unlinked": 0, "products": 0, "images": 0} with connect() as conn: rows = conn.execute( """ SELECT l.*, b.slug AS brand_slug, s.domain FROM elec.source_listing l JOIN elec.brand b ON b.id = l.brand_id JOIN elec.site s ON s.id = l.site_id JOIN elec.category c ON c.id = l.category_id WHERE c.slug = %s """, (category,), ).fetchall() images = conn.execute( """ SELECT i.url, i.source_listing_id, i.source_type, i.rank FROM elec.product_image i JOIN elec.product p ON p.id = i.product_id JOIN elec.category c ON c.id = p.category_id WHERE c.slug = %s """, (category,), ).fetchall() with transaction() as conn: # Maps and images cascade from the products. conn.execute( "DELETE FROM elec.product p USING elec.category c WHERE c.id = p.category_id AND c.slug = %s", (category,), ) ids = repo.id_maps() product_of_listing: Dict[int, int] = {} rows.sort(key=lambda r: (_ORDER.get(r["source_type"], 9), r["id"])) for row in rows: stats["listings"] += 1 listing = _listing_from_row(row, category) decision = decide(listing, repo.product_candidates(listing.brand_slug, category)) if decision is None: stats["unlinked"] += 1 continue product_id = decision.product_id or repo.create_product(listing, ids) stats["products"] += decision.product_id is None repo.map_listing(row["id"], product_id, decision.method, decision.confidence, decision.review_status) product_of_listing[row["id"]] = product_id if decision.review_status == "pending": stats["pending"] += 1 else: stats["linked"] += 1 repo.merge_product_specs(product_id, listing.specs, {}, listing.source_url) for img in images: pid = product_of_listing.get(img["source_listing_id"]) if pid is not None: repo.add_image(pid, img["url"], img["source_listing_id"], img["source_type"], img["rank"]) stats["images"] += 1 stats.update({f"products_{k}": v for k, v in repo.refresh_verification().items()}) return stats