backend apis updation

This commit is contained in:
sriram
2026-08-13 15:54:29 +05:30
parent b8d93fbbf2
commit 241fd237f8
13 changed files with 973 additions and 105 deletions

View File

@@ -1,3 +1,6 @@
import re
from functools import lru_cache
BRAND_ALIASES = {
# Cadbury family
"cadbury gems": "cadbury",
@@ -253,6 +256,21 @@ BRAND_ALIASES = {
DEFAULT_ALIASES = BRAND_ALIASES
def _contains_word(haystack: str, needle: str) -> bool:
"""True when `needle` occurs in `haystack` as a whole word.
Plain `in` would treat any fragment as a match, so a short brand name
could be swallowed by an unrelated alias that merely contains those
letters - e.g. "sun" inside "hul sunsilk". Anchoring both ends on a
word boundary keeps the genuine multi-word hits ("tata" inside
"hul tata tea") while dropping the fragment ones.
"""
if not needle:
return False
return re.search(r"(?<!\w)" + re.escape(needle) + r"(?!\w)", haystack) is not None
@lru_cache(maxsize=1024)
def resolve_parent_brand(brand: str) -> str:
"""Return the parent (canonical) brand for storage purposes.
@@ -260,15 +278,19 @@ def resolve_parent_brand(brand: str) -> str:
returns the parent brand name so that sub-brands share the same
database table, S3 folder, and JSON file as their parent.
Falls back to fuzzy substring matching, then returns the input
unchanged if no alias is known.
Falls back to whole-word matching in either direction, then returns
the input unchanged if no alias is known.
Cached because `_table_name()` in vector_store calls this on every
query and the fallback loop scans all ~230 aliases. BRAND_ALIASES is
a module constant that is never mutated, so the result is stable.
"""
key = brand.lower().strip()
direct = BRAND_ALIASES.get(key)
if direct:
return direct
for alias, parent in BRAND_ALIASES.items():
if alias in key or key in alias:
if _contains_word(key, alias) or _contains_word(alias, key):
return parent
return brand

433
app/services/brand_sync.py Normal file
View File

@@ -0,0 +1,433 @@
"""
Keeps the `brand_*` Postgres tables and `data/seed_catalogs/*.json` in step.
A brand can enter the system from either side: `POST /api/user/products/add`
and `POST /api/catalog/generate` write rows, while a catalog file may be
dropped in by hand or shipped in the image. Neither side used to produce the
other, so a brand ingested at runtime had no seed file, and a seed file added
after first boot was never loaded (the startup auto-seed only runs against a
completely empty database).
This module is the single place that knows the correspondence between the two,
and `reconcile_brand_catalogs()` repairs it in both directions.
The correspondence is *not* the filename. `brand_catalog_p_and_g.json` holds
`"brand": "p&g"`, which sanitises to table `brand_p_g`; `brand_catalog_tata.json`
resolves through BRAND_ALIASES into `brand_hindustan_unilever`. Everything here
therefore indexes files by their `brand` field, exactly as
`scripts/seed_sample_data.py` does. Keying on filenames instead would split
P&G across two files and let Tata clobber Hindustan Unilever.
"""
from __future__ import annotations
import json
import logging
import os
from collections import defaultdict
from decimal import Decimal
from pathlib import Path
from typing import Any, Dict, List, Optional
from app.services.brand_registry import resolve_parent_brand
from app.services.vector_store import (
_connect,
_list_brand_table_suffixes,
_sanitize_name,
display_name_for_suffix,
ensure_brand_schema,
get_products_by_brand,
invalidate_brand_overview_cache,
upsert_brand_products,
)
logger = logging.getLogger(__name__)
# app/services/brand_sync.py -> parents[2] is backend/
SEED_DIR = Path(__file__).resolve().parents[2] / "data" / "seed_catalogs"
# The column list written by upsert_brand_products, minus `embedding` (handled
# separately because it is huge and optional). Keeping these in sync is what
# makes an exported file round-trip back through the seeder without losing
# hsn/price/barcode/sku data - the failing mode of scripts/export_seed_data.py.
EXPORT_COLUMNS = (
"product_name", "title", "description", "category", "image_id",
"image_url", "image_urls", "price_range", "size_variants", "providers",
"fssai_license", "product_sku", "sku_source", "hsn_code",
"final_selling_price", "selling_price", "barcode", "barcode_type",
"highlights", "nutrients", "search_query",
)
def brand_slug(brand: str) -> str:
"""The table suffix a brand name resolves to (e.g. 'ITC' -> 'itc')."""
return _sanitize_name(resolve_parent_brand(brand))
def _read_catalog(path: Path) -> Optional[Dict[str, Any]]:
"""Parse a seed catalog, returning None for anything that isn't one.
Files without a brand or products (notably `hsn_gst_master.json`) are not
catalogs, and a corrupt file must not take down a startup reconcile.
"""
try:
data = json.loads(path.read_text(encoding="utf-8-sig"))
except Exception as e: # noqa: BLE001 - one bad file cannot break the sweep
logger.warning("Could not read seed catalog %s: %s", path.name, e)
return None
if not isinstance(data, dict):
return None
products = data.get("products")
if not isinstance(products, list) or not products:
return None
brand = data.get("brand") or (products[0].get("brand_name") if isinstance(products[0], dict) else None)
if not brand:
return None
data["brand"] = brand
return data
def index_seed_files(seed_dir: Path = SEED_DIR) -> Dict[str, List[Path]]:
"""Map each brand table suffix to the seed files that feed it.
More than one file can feed a suffix - tata.json and hindustan_unilever.json
both land in brand_hindustan_unilever - which is why the value is a list.
"""
index: Dict[str, List[Path]] = defaultdict(list)
if not seed_dir.exists():
return {}
for path in sorted(seed_dir.glob("*.json")):
data = _read_catalog(path)
if data is None:
continue
index[brand_slug(data["brand"])].append(path)
return dict(index)
# Resolving a brand to its file means parsing every catalog, and /batch-add
# calls that once per product. The mapping only changes when a file is created
# or removed, so it is cached and invalidated explicitly on write.
_TARGET_CACHE: Dict[str, Path] = {}
def invalidate_seed_index_cache() -> None:
_TARGET_CACHE.clear()
def canonical_seed_file(brand: str, index: Optional[Dict[str, List[Path]]] = None) -> Path:
"""The file to write for `brand`.
Prefers a file whose own brand field sanitises to the same slug, so P&G
products append to brand_catalog_p_and_g.json rather than creating an
orphan brand_catalog_p_g.json, and Hindustan Unilever products never get
written into brand_catalog_tata.json.
"""
slug = brand_slug(brand)
if index is None:
cached = _TARGET_CACHE.get(slug)
if cached is not None:
return cached
index = index_seed_files()
resolved = SEED_DIR / f"brand_catalog_{slug}.json"
candidates = index.get(slug, [])
for path in candidates:
data = _read_catalog(path)
if data and _sanitize_name(data["brand"]) == slug:
resolved = path
break
else:
if candidates:
resolved = candidates[0]
_TARGET_CACHE[slug] = resolved
return resolved
def upsert_products_into_catalog_file(brand: str, products: List[Dict[str, Any]]) -> Optional[Path]:
"""Merge `products` into the brand's seed catalog, creating it if needed.
Products are matched on image_id or product_name, so re-adding an existing
product updates it in place instead of duplicating. Embeddings are stripped
(they are ~384 floats each and the file is meant to stay readable), and the
write is atomic so a crash cannot leave a half-written catalog that then
fails to parse on the next boot.
"""
if not products:
return None
SEED_DIR.mkdir(parents=True, exist_ok=True)
file_path = canonical_seed_file(brand)
if file_path.exists():
try:
data = json.loads(file_path.read_text(encoding="utf-8-sig"))
except Exception as e: # noqa: BLE001
logger.warning("Could not read existing catalog JSON %s: %s", file_path.name, e)
data = {"brand": brand, "products": []}
else:
data = {
"brand": brand.lower(),
"search_query": f"{brand} products catalog",
"generation_timestamp": str(Path(__file__).resolve()),
"total_products": 0,
"total_images": 0,
"products": [],
}
products_list = data.get("products") or []
by_image_id = {
p.get("image_id"): i for i, p in enumerate(products_list) if p.get("image_id")
}
by_name = {
p.get("product_name"): i for i, p in enumerate(products_list) if p.get("product_name")
}
for product in products:
clean = {k: v for k, v in product.items() if k != "embedding"}
idx = by_image_id.get(clean.get("image_id"))
if idx is None:
idx = by_name.get(clean.get("product_name"))
if idx is None:
products_list.append(clean)
if clean.get("image_id"):
by_image_id[clean["image_id"]] = len(products_list) - 1
if clean.get("product_name"):
by_name[clean["product_name"]] = len(products_list) - 1
else:
products_list[idx] = clean
data["products"] = products_list
data["total_products"] = len(products_list)
data["total_images"] = sum(len(p.get("image_urls") or []) for p in products_list)
payload = json.dumps(data, indent=2, ensure_ascii=False)
tmp_path = file_path.with_suffix(".json.tmp")
tmp_path.write_text(payload, encoding="utf-8")
os.replace(tmp_path, file_path)
logger.info("✅ Updated JSON seed file '%s' (total products: %d)",
file_path.name, data["total_products"])
return file_path
def _jsonable(value: Any) -> Any:
"""Coerce a psycopg row value into something json.dumps accepts."""
if isinstance(value, Decimal):
return float(value)
if isinstance(value, (list, tuple)):
return [_jsonable(v) for v in value]
return value
def export_brand_to_seed_file(brand: str, include_embeddings: bool = True) -> Optional[Path]:
"""Write a brand's database rows out to its seed catalog, field-complete.
Deliberately not scripts/export_seed_data.py, whose _read_products emits
only ~10 keys - running that would silently strip hsn_code, prices,
barcodes, SKUs and FSSAI numbers out of the existing catalogs.
Embeddings are carried through when present because that is what lets
seed_sample_data.py re-seed without invoking the embedding model.
"""
rows = get_products_by_brand(brand)
if not rows:
logger.info("Nothing to export for brand '%s' (no rows)", brand)
return None
display = display_name_for_suffix(brand_slug(brand))
products: List[Dict[str, Any]] = []
for row in rows:
product: Dict[str, Any] = {"brand": display, "brand_name": display}
for col in EXPORT_COLUMNS:
if col in row:
product[col] = _jsonable(row[col])
if include_embeddings and row.get("embedding") is not None:
try:
product["embedding"] = [float(x) for x in row["embedding"]]
except (TypeError, ValueError):
pass
products.append(product)
path = upsert_products_into_catalog_file(display, products)
if path:
logger.info("📤 Exported %d product(s) for '%s' -> %s", len(products), display, path.name)
return path
def load_seed_catalogs(seed_dir: Path = SEED_DIR,
only: Optional[List[str]] = None) -> Dict[str, List[Dict[str, Any]]]:
"""Read seed catalogs and group their products by resolved parent brand.
Grouping matters: several files can feed one table, and the seeder's
stale-row cleanup deletes anything not in the batch it is given. Merging
first is what stops tata.json and hindustan_unilever.json erasing each
other.
"""
if not seed_dir.exists():
logger.error("Seed directory not found: %s", seed_dir)
return {}
files = sorted(seed_dir.glob("*.json"))
if only:
wanted = [w.lower() for w in only]
files = [f for f in files if any(w in f.name.lower() for w in wanted)]
brand_products: Dict[str, List[Dict[str, Any]]] = defaultdict(list)
for path in files:
data = _read_catalog(path)
if data is None:
logger.warning("Skipping %s - no brand/products found", path.name)
continue
resolved = resolve_parent_brand(data["brand"])
brand_products[resolved].extend(data["products"])
logger.info("Read %d products from %s -> resolved brand '%s'",
len(data["products"]), path.name, resolved)
return dict(brand_products)
def seed_brands(brand_products: Dict[str, List[Dict[str, Any]]], cleanup: bool = True) -> int:
"""Upsert grouped products into their brand tables. Returns the row count."""
total = 0
for resolved_brand, all_products in brand_products.items():
logger.info("Seeding %d product(s) for brand '%s'", len(all_products), resolved_brand)
table = ensure_brand_schema(resolved_brand)
if not table:
logger.error("Could not create/verify table for brand '%s' - is pgvector reachable?",
resolved_brand)
continue
upsert_brand_products(resolved_brand, all_products, cleanup=cleanup)
logger.info("Seeded %d products for brand '%s' (table=%s)",
len(all_products), resolved_brand, table)
total += len(all_products)
return total
def _db_brand_counts() -> Dict[str, int]:
"""{table suffix: row count} for every brand_* table, on one connection."""
conn = _connect()
if not conn:
return {}
counts: Dict[str, int] = {}
try:
with conn, conn.cursor() as cur:
for suffix in sorted(set(_list_brand_table_suffixes(cur))):
try:
cur.execute(f"SELECT COUNT(*) FROM brand_{suffix}")
row = cur.fetchone()
counts[suffix] = int(row[0]) if row else 0
except Exception as e: # noqa: BLE001
logger.warning("Could not count brand_%s: %s", suffix, e)
except Exception as e: # noqa: BLE001
logger.error("Failed to enumerate brand tables: %s", e)
finally:
conn.close()
return counts
def _detect_collisions(seed_dir: Path = SEED_DIR) -> List[Dict[str, Any]]:
"""Distinct parent brands that sanitise onto one table (e.g. 'P&G' vs 'P G').
Grouped by *resolved parent*, not by the raw brand field: tata.json and
hindustan_unilever.json share a table because BRAND_ALIASES deliberately
merges them, which is not a collision. A genuine one is two unrelated
parents whose names differ only in characters _sanitize_name strips.
Reported rather than repaired - renaming a live table is a one-way door,
and nothing collides today. This is here so it surfaces the day it does.
"""
by_slug: Dict[str, set] = defaultdict(set)
if not seed_dir.exists():
return []
for path in sorted(seed_dir.glob("*.json")):
data = _read_catalog(path)
if data is None:
continue
parent = resolve_parent_brand(data["brand"]).strip().lower()
by_slug[_sanitize_name(parent)].add(parent)
return [
{"slug": slug, "brands": sorted(names)}
for slug, names in sorted(by_slug.items()) if len(names) > 1
]
def reconcile_brand_catalogs(dry_run: bool = False) -> Dict[str, Any]:
"""Repair the table <-> seed-file correspondence in both directions.
Idempotent and non-destructive:
* A populated table with no seed file gets one exported.
* A seed file whose table is empty or absent gets seeded, with
cleanup disabled - the table holds nothing this batch could be a
partial view of, so there is no stale row to remove and no way to
delete data by passing an incomplete set.
* A file that already maps to a populated table is left alone. That
rule is what stops brand_catalog_tata.json being overwritten with
Hindustan Unilever's merged rows.
"""
# Files may have appeared on disk since the last resolve (that is half of
# what this function exists to handle), so start from a cold index.
invalidate_seed_index_cache()
db_counts = _db_brand_counts()
file_index = index_seed_files()
collisions = _detect_collisions()
to_export = sorted(
suffix for suffix, count in db_counts.items()
if count > 0 and suffix not in file_index
)
to_seed = sorted(
slug for slug in file_index
if db_counts.get(slug, 0) == 0
)
summary: Dict[str, Any] = {
"tables": len(db_counts),
"files": len(file_index),
"exported": [],
"seeded": [],
"collisions": collisions,
"skipped": [],
"dry_run": dry_run,
}
if dry_run:
summary["exported"] = [display_name_for_suffix(s) for s in to_export]
summary["seeded"] = to_seed
return summary
for suffix in to_export:
display = display_name_for_suffix(suffix)
try:
path = export_brand_to_seed_file(display)
if path:
summary["exported"].append(display)
except Exception as e: # noqa: BLE001 - one brand must not stop the sweep
logger.error("Export failed for brand '%s': %s", display, e)
summary["skipped"].append({"brand": display, "reason": str(e)})
for slug in to_seed:
paths = file_index.get(slug, [])
try:
grouped = load_seed_catalogs(only=[p.name for p in paths])
if not grouped:
continue
seed_brands(grouped, cleanup=False)
summary["seeded"].append(slug)
except Exception as e: # noqa: BLE001
logger.error("Seeding failed for slug '%s': %s", slug, e)
summary["skipped"].append({"brand": slug, "reason": str(e)})
if summary["exported"] or summary["seeded"]:
invalidate_brand_overview_cache()
try:
from app.services.query_intent import invalidate_brand_mention_cache
invalidate_brand_mention_cache()
except Exception: # noqa: BLE001
pass
if collisions:
logger.warning("Brand slug collisions detected: %s", collisions)
return summary

View File

@@ -15,7 +15,8 @@ and free on CPU-only hardware. Two things are extracted:
from __future__ import annotations
import re
from typing import Dict, Optional
import time
from typing import Any, Dict, Optional
from app.services.category_registry import detect_category_from_text # noqa: F401 (re-exported)
@@ -170,18 +171,70 @@ BRAND_SEARCH_MAP = {
}
BRAND_MENTION_TTL_SECONDS = 300
# Populated lazily by _brand_index(); holds the static tables above merged with
# whatever brands actually exist in the vector store right now.
_MENTION_CACHE: Dict[str, Any] = {"at": 0.0, "known": None, "map": None}
def invalidate_brand_mention_cache() -> None:
"""Force the next _brand_index() call to re-read the live brand list.
Called after a brand is added or the seed catalogs are reconciled, so a
newly ingested brand becomes mentionable without waiting out the TTL.
"""
_MENTION_CACHE.update(at=0.0, known=None, map=None)
def _brand_index() -> tuple[list[str], Dict[str, str]]:
"""The static brand tables, plus any live brand they don't already cover.
KNOWN_BRANDS and BRAND_SEARCH_MAP carry hand-tuned synonyms ("coke",
"hul", "pepsi") that cannot be derived from the database, so they are the
base and always win - the live list only *adds* brands, via setdefault.
That is what lets a brand ingested at runtime be recognised in chat and
search without anyone editing this file.
Cached for BRAND_MENTION_TTL_SECONDS because this runs on every query and
list_available_brands() opens a fresh connection. If the database is
unreachable the static behaviour stands rather than failing the query.
"""
now = time.monotonic()
if _MENTION_CACHE["map"] is not None and now - _MENTION_CACHE["at"] < BRAND_MENTION_TTL_SECONDS:
return _MENTION_CACHE["known"], _MENTION_CACHE["map"]
known = list(KNOWN_BRANDS)
mapping = dict(BRAND_SEARCH_MAP)
try:
from app.services.vector_store import list_available_brands
covered = {k.lower() for k in known}
for brand in list_available_brands():
if brand.lower() not in covered:
known.append(brand)
covered.add(brand.lower())
mapping.setdefault(brand.lower(), brand)
except Exception: # noqa: BLE001 - a DB blip must not break intent parsing
pass
_MENTION_CACHE.update(at=now, known=known, map=mapping)
return known, mapping
def extract_brand_mention(query: str) -> Optional[str]:
"""Detect if a brand name is explicitly mentioned in the query text."""
if not query:
return None
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
lower = query.lower()
known_brands, brand_search_map = _brand_index()
# 1. Check direct search map
for alias in sorted(BRAND_SEARCH_MAP.keys(), key=len, reverse=True):
for alias in sorted(brand_search_map.keys(), key=len, reverse=True):
pattern = r"\b" + re.escape(alias) + r"\b"
if re.search(pattern, lower):
return BRAND_SEARCH_MAP[alias]
return brand_search_map[alias]
# 2. Check sub-brand aliases (e.g. "oreo", "maggi", "good day")
sorted_aliases = sorted(BRAND_ALIASES.keys(), key=len, reverse=True)
@@ -190,7 +243,7 @@ def extract_brand_mention(query: str) -> Optional[str]:
if re.search(pattern, lower):
parent = resolve_parent_brand(alias)
# Normalize to canonical known brand name case
for kb in KNOWN_BRANDS:
for kb in known_brands:
if kb.lower() == parent.lower():
return kb
return parent.title()

View File

@@ -4,7 +4,9 @@ from typing import List, Optional, Dict, Any
import json
import logging
import os
import re
import time
import psycopg
from app.infrastructure.settings import DATABASE_URL, USE_PGVECTOR, DB_HOST, DB_PORT, DB_NAME, DB_USER, DB_PASSWORD
@@ -367,9 +369,19 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
deleted = cur.rowcount
if deleted:
logger.info(f"🗑️ Removed {deleted} stale product(s) from {table_name}")
conn.close()
# Every write path into the catalog funnels through here, so this is the
# one place that has to invalidate the derived views: the brand cards'
# counts, and the brand list that chat/search intent parsing scopes on.
invalidate_brand_overview_cache()
try:
from app.services.query_intent import invalidate_brand_mention_cache
invalidate_brand_mention_cache()
except Exception: # noqa: BLE001 - cache invalidation must never fail a write
pass
def get_existing_product_image_id(brand: str, product_name: str) -> Optional[str]:
"""Check if a product with this name exists in the brand table and return its image_id"""
@@ -412,20 +424,45 @@ def _build_sanitized_brand_map() -> Dict[str, str]:
return seen
# Acronyms that .title() would mangle ("itc" -> "Itc"). Only needed for brands
# whose parent value in BRAND_ALIASES is lowercase and not a real word.
BRAND_DISPLAY_OVERRIDES = {
"itc": "ITC",
"grb": "GRB",
"hul": "HUL",
}
def display_name_for_suffix(suffix: str, brand_map: Optional[Dict[str, str]] = None) -> str:
"""Turn a brand table suffix back into the name shown in the UI.
Shared by list_available_brands() and get_brand_overview() so the sidebar
list and the brand cards can never disagree - the frontend passes these
strings straight back to /api/brands/{brand}/products, so they must round
trip through resolve_parent_brand + _sanitize_name to the same table.
"""
if brand_map is None:
brand_map = _build_sanitized_brand_map()
override = BRAND_DISPLAY_OVERRIDES.get(suffix)
if override:
return override
return brand_map.get(suffix) or suffix.replace('_', ' ').title()
def list_available_brands() -> List[str]:
"""List all available brands that have tables in the database"""
conn = _connect()
if not conn:
return []
brand_map = _build_sanitized_brand_map()
brands = []
with conn, conn.cursor() as cur:
try:
# Query information_schema for tables starting with brand_
cur.execute("""
SELECT table_name
FROM information_schema.tables
SELECT table_name
FROM information_schema.tables
WHERE table_schema = 'public' AND table_name LIKE 'brand_%'
""")
tables = cur.fetchall()
@@ -434,15 +471,143 @@ def list_available_brands() -> List[str]:
table_name = table[0]
suffix = table_name[len("brand_"):].lower() if table_name.startswith("brand_") else table_name.lower()
# Prefer the original display name from the brand map if available
brand_name = brand_map.get(suffix, suffix.replace('_', ' ').title())
brands.append(brand_name)
brands.append(display_name_for_suffix(suffix, brand_map))
except Exception as e:
logger.error(f"Failed to list brands: {e}")
conn.close()
return sorted(list(set(brands)))
BRAND_OVERVIEW_TTL_SECONDS = int(os.getenv("BRAND_OVERVIEW_TTL_SECONDS", "30"))
_OVERVIEW_CACHE: Dict[str, Any] = {"at": 0.0, "data": None}
def invalidate_brand_overview_cache() -> None:
"""Drop the cached brand overview so the next read reflects a fresh write."""
_OVERVIEW_CACHE.update(at=0.0, data=None)
def get_brand_overview(force_refresh: bool = False) -> List[Dict[str, Any]]:
"""Per-brand summary rows backing the brand cards on the home page.
Everything runs on a single connection: _connect() opens a new one each
call, so doing this per brand would cost dozens of handshakes on every
page load. Results are cached for BRAND_OVERVIEW_TTL_SECONDS and dropped
explicitly by upsert_brand_products(), which covers every write path.
"""
now = time.monotonic()
if (
not force_refresh
and _OVERVIEW_CACHE["data"] is not None
and now - _OVERVIEW_CACHE["at"] < BRAND_OVERVIEW_TTL_SECONDS
):
return _OVERVIEW_CACHE["data"]
conn = _connect()
if not conn:
return []
brand_map = _build_sanitized_brand_map()
overview: List[Dict[str, Any]] = []
try:
with conn, conn.cursor() as cur:
# Brand tables predating the current DDL are missing columns until a
# write runs _ensure_columns on them, so build each query from the
# columns that actually exist. The rest of this module survives that
# by selecting *; naming columns here would 500 the whole endpoint.
cur.execute(
"""
SELECT table_name, column_name FROM information_schema.columns
WHERE table_schema = 'public' AND table_name LIKE 'brand_%'
"""
)
table_columns: Dict[str, set] = {}
for table_name, column_name in cur.fetchall():
table_columns.setdefault(table_name, set()).add(column_name)
for suffix in sorted(set(_list_brand_table_suffixes(cur))):
table_name = f"brand_{suffix}"
columns = table_columns.get(table_name, set())
has_category = "category" in columns
try:
if has_category:
cur.execute(
f"""
SELECT COUNT(*),
COUNT(DISTINCT category)
FILTER (WHERE category IS NOT NULL AND category <> '')
FROM {table_name}
"""
)
row = cur.fetchone()
product_count = int(row[0]) if row else 0
category_count = int(row[1]) if row and row[1] is not None else 0
else:
cur.execute(f"SELECT COUNT(*) FROM {table_name}")
row = cur.fetchone()
product_count = int(row[0]) if row else 0
category_count = 0
if product_count == 0:
# An empty table is a leftover, not a brand to advertise.
continue
categories: List[str] = []
if has_category:
cur.execute(
f"""
SELECT category FROM {table_name}
WHERE category IS NOT NULL AND category <> ''
GROUP BY category ORDER BY COUNT(*) DESC LIMIT 4
"""
)
categories = [r[0] for r in cur.fetchall() if r[0]]
img = None
image_cols = [c for c in ("image_id", "image_url", "image_urls") if c in columns]
if "image_url" in columns or "image_urls" in columns:
where = " OR ".join(
f"({c} IS NOT NULL)" for c in ("image_url", "image_urls") if c in columns
)
order = " ORDER BY updated_at DESC" if "updated_at" in columns else ""
cur.execute(
f"SELECT {', '.join(image_cols)} FROM {table_name} "
f"WHERE {where}{order} LIMIT 1"
)
row = cur.fetchone()
img = dict(zip(image_cols, row)) if row else None
except Exception as e:
logger.warning("Brand overview skipped %s: %s", table_name, e)
continue
sample_image_id = (img or {}).get("image_id")
sample_image_url = (img or {}).get("image_url") or None
if not sample_image_url and img and img.get("image_urls"):
urls = [u for u in img["image_urls"] if u]
sample_image_url = urls[0] if urls else None
overview.append({
"suffix": suffix,
"display_name": display_name_for_suffix(suffix, brand_map),
"product_count": product_count,
"category_count": category_count,
"categories": categories,
"sample_image_id": sample_image_id,
"sample_image_url": sample_image_url,
})
except Exception as e:
logger.error(f"Failed to build brand overview: {e}")
finally:
conn.close()
overview.sort(key=lambda b: b["display_name"].lower())
_OVERVIEW_CACHE.update(at=now, data=overview)
return overview
def get_products_by_brand(brand: str, limit: Optional[int] = None, offset: int = 0,
category: Optional[str] = None) -> List[Dict[str, Any]]:
"""Fetch products for a specific brand from its table (plain listing, no ranking).