unpopular brand generation

This commit is contained in:
sriram
2026-09-10 17:51:05 +05:30
parent d5a23f6456
commit e1a5962f82
11 changed files with 7629 additions and 3 deletions

View File

@@ -0,0 +1,150 @@
"""
Read each brand's own shop catalogue and cache it for discovery to use.
WHY THIS IS A SCRIPT
--------------------
Same split as `backfill_retail_presence.py`, for the same reason: the network
work happens offline and paced, and the runtime path
(`brand_discovery._from_brand_store`) reads only the cache. A discovery preview
must never block on a small company's storefront being slow, and a storefront
must never be hammered because somebody opened a preview.
Unlike the retail check, this is cheap - one or two requests per BRAND, not one
per product - so a full pass over every configured brand costs a handful of
requests and a few seconds.
WHERE THE DOMAINS COME FROM
---------------------------
`brand_registry.BRAND_STORE_DOMAINS`, which is human-curated on purpose. A
wrong domain here does not cost one bad row; it imports a hundred of another
company's real products under this brand's name, and they look perfect because
they ARE perfect - just somebody else's. `resolve_brand_domain` answers "Naga"
with the website of Naga City in the Philippines, which is why search is not
trusted for this.
`--domain` exists for trying a candidate site before adding it to the map. A
domain given that way is NOT trusted: it goes through `brand_store`'s identity
guard, which a curated one skips.
USAGE
python scripts/backfill_brand_stores.py --all # dry run
python scripts/backfill_brand_stores.py --all --apply
python scripts/backfill_brand_stores.py --brand Gopuram --apply --refresh
python scripts/backfill_brand_stores.py --brand Naga --domain nagafoods.in --apply
"""
from __future__ import annotations
import argparse
import logging
import sys
import time
from pathlib import Path
from typing import List, Optional
_BACKEND_DIR = Path(__file__).resolve().parent.parent
if str(_BACKEND_DIR) not in sys.path:
sys.path.insert(0, str(_BACKEND_DIR))
from app.services import brand_store # noqa: E402
from app.services.brand_registry import ( # noqa: E402
BRAND_STORE_DOMAINS,
get_brand_store_domain,
)
logger = logging.getLogger("backfill_brand_stores")
def _summarise(brand: str, rows: List[dict]) -> None:
sized = sum(1 for r in rows if r.get("size"))
barcoded = sum(1 for r in rows if r.get("barcode"))
imaged = sum(1 for r in rows if r.get("image_url"))
priced = sum(1 for r in rows if r.get("price") is not None)
logger.info(
" %d products size=%d barcode=%d image=%d price=%d",
len(rows), sized, barcoded, imaged, priced,
)
# Size coverage looks low here for a reason worth stating rather than
# silently accepting: most of these shops put the pack size in the PRODUCT
# NAME, not in a variant. `brand_discovery._resolve_sizes` reads a title's
# own size first, which lifts Double Horse from 14 sized rows to 94 of 99.
if rows and sized < len(rows) // 2:
logger.info(
" (most sizes are in the product name, not a variant - "
"_resolve_sizes recovers those at discovery time)"
)
def main(argv: Optional[List[str]] = None) -> int:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--brand", help="Only this brand.")
parser.add_argument("--all", action="store_true",
help="Every brand in BRAND_STORE_DOMAINS.")
parser.add_argument("--domain",
help="Try this domain instead of the curated one. NOT "
"trusted - it must pass the identity guard.")
parser.add_argument("--apply", action="store_true",
help="Actually fetch. Without this nothing hits the network.")
parser.add_argument("--refresh", action="store_true",
help="Re-fetch even if a cached catalogue exists.")
parser.add_argument("--pause", type=float, default=brand_store.PAUSE_SECONDS)
args = parser.parse_args(argv)
logging.basicConfig(level=logging.INFO, format="%(message)s")
if not args.brand and not args.all:
parser.error("give --brand NAME or --all")
if args.brand:
brands = [args.brand]
else:
brands = sorted(BRAND_STORE_DOMAINS)
ok = missing = failed = 0
for brand in brands:
curated = get_brand_store_domain(brand)
domain = args.domain or curated
trusted = bool(curated) and not args.domain
if not domain:
logger.warning("%-22s no store domain configured - add it to "
"brand_registry.BRAND_STORE_DOMAINS", brand)
missing += 1
continue
label = domain + ("" if trusted else " (untrusted, guard applies)")
if not args.apply:
logger.info("[dry-run] %-20s would read %s", brand, label)
continue
logger.info("%-22s %s", brand, label)
try:
rows = brand_store.fetch_store_catalogue(
brand, domain, live=True, refresh=args.refresh,
trusted_domain=trusted,
)
except Exception as exc: # noqa: BLE001 - one bad site cannot stop the sweep
logger.warning(" failed: %s", exc)
failed += 1
continue
if not rows:
logger.warning(" no catalogue (not a supported storefront, "
"unreachable, or rejected by the identity guard)")
failed += 1
else:
_summarise(brand, rows)
ok += 1
time.sleep(args.pause)
if args.apply:
logger.info("")
logger.info("catalogues read : %d", ok)
logger.info("no catalogue : %d", failed)
if missing:
logger.info("no domain set : %d", missing)
return 0
if __name__ == "__main__":
raise SystemExit(main())