unpopular brand generation

This commit is contained in:
sriram
2026-09-10 17:51:05 +05:30
parent d5a23f6456
commit e1a5962f82
11 changed files with 7629 additions and 3 deletions

View File

@@ -105,7 +105,7 @@ from app.infrastructure.settings import (
BRAND_DISCOVERY_USE_LLM,
BRAND_DISCOVERY_USE_OFF,
)
from app.services import active_brands, ollama_service, retail_presence
from app.services import active_brands, brand_store, ollama_service, retail_presence
from app.services.brand_registry import (
get_fssai_license,
get_known_sub_brands,
@@ -481,6 +481,33 @@ def _from_open_facts(brand: str, *, refresh: bool = False) -> List[Dict[str, Any
return out
def _from_brand_store(brand: str) -> List[Dict[str, Any]]:
"""The brand's own shop catalogue - the strongest source there is.
Open Food Facts is crowd-sourced and, for a regional South Indian brand,
usually empty: Udhaiyam 0 hits, Gopuram 0, Tenali Double Horse 0. Their
storefronts publish the whole catalogue as structured JSON - Double Horse
99 rows, Gopuram 122 - with real names, prices, images and sometimes a
valid EAN-13.
CACHE ONLY. `scripts/backfill_brand_stores.py` does the fetching, exactly
as the retail check is split, so a discovery preview never blocks on
somebody's storefront being slow or down.
Pack sizes are deliberately left as the store stated them, which is often
nothing: most of these shops put the size in the PRODUCT NAME rather than
in a variant ("Roasted Vermicelli 400g"). `_resolve_sizes` already reads a
title's own size first and is the one place that rule should live -
measured, that lifts Double Horse from 14 sized rows to 94 of 99.
"""
try:
rows = brand_store.fetch_store_catalogue(brand, None, live=False)
except Exception as exc: # noqa: BLE001 - a bad cache must not kill discovery
logger.warning("Brand store catalogue unreadable for %r: %s", brand, exc)
return []
return list(rows or [])
# ---------------------------------------------------------------------------
# Source B - Ollama (breadth and prose)
# ---------------------------------------------------------------------------
@@ -669,6 +696,21 @@ def _score(sources: Sequence[str], evidence: Optional[str]) -> float:
"""
has_off = "off" in sources
has_llm = "llm" in sources
has_store = "store" in sources
# THE BRAND'S OWN SHOP OUTRANKS OPEN FOOD FACTS.
#
# OFF is crowd-sourced and, for these brands, mostly absent or wrong - its
# two "Naga" rows are a British pickle. A manufacturer's own storefront is
# first-party: it is the definitive list of what the company sells, at the
# sizes and prices it sells them, with its own photographs.
#
# Scored above `off` rather than merged into it so the preview can say
# which source a row came from, and so a future disagreement between the
# two is visible rather than silently resolved.
if has_store and (has_off or has_llm):
return 1.0
if has_store:
return 0.95
if has_off and has_llm:
return 1.0
if has_off:
@@ -866,6 +908,7 @@ def discover_brand_products(
deadline_seconds: float = BRAND_DISCOVERY_DEADLINE_SECONDS,
use_openfacts: bool = BRAND_DISCOVERY_USE_OFF,
use_llm: bool = BRAND_DISCOVERY_USE_LLM,
use_store: bool = True,
require_evidence: bool = True,
refresh_corpus: bool = False,
) -> DiscoveryResult:
@@ -898,6 +941,24 @@ def discover_brand_products(
if key:
off_keys.setdefault(key, candidate)
# The brand's own shop. Read from cache only - see `_from_brand_store`.
store_candidates = _from_brand_store(brand) if use_store else []
store_keys = {}
for candidate in store_candidates:
key = _normalise_title(brand, candidate.get("title") or "")
if key:
store_keys.setdefault(key, candidate)
if store_candidates:
logger.info("%s: %d products from the brand's own shop", brand, len(store_candidates))
elif use_store and not off_candidates:
warnings.append(
"Neither Open Food Facts nor a brand storefront has anything for "
"this brand. Every row below rests on the language model alone - "
"review them individually. If this brand has its own online shop, "
"adding it to brand_registry.BRAND_STORE_DOMAINS and running "
"scripts/backfill_brand_stores.py will populate a real catalogue."
)
llm_candidates: List[Dict[str, Any]] = []
if use_llm:
@@ -920,7 +981,10 @@ def discover_brand_products(
# and adopts the size of the one it joins.
merged: List[Dict[str, Any]] = []
keys: List[str] = []
for candidate in off_candidates + llm_candidates:
# STORE FIRST, then Open Food Facts, then the model. The merge fills a
# blank from whatever comes later, so order IS precedence: the brand's own
# shop outranks a crowd-sourced database, which outranks a guess.
for candidate in store_candidates + off_candidates + llm_candidates:
title = candidate["title"]
key = _normalise_title(brand, title)
target = None
@@ -1056,7 +1120,8 @@ def discover_brand_products(
)
if product is None:
continue
if require_evidence and "off" not in product.sources and not product.evidence:
if (require_evidence and "off" not in product.sources
and "store" not in product.sources and not product.evidence):
dropped += 1
continue
products.append(product)

View File

@@ -261,8 +261,70 @@ BRAND_ALIASES = {
# that spells it without the s. Removing it re-opens the bug.
"haldiram": "haldirams",
"haldiram's": "haldirams",
# --- Regional South Indian brands -------------------------------------
# Not in Open Food Facts (Udhaiyam 0 hits, Gopuram 0, Tenali Double Horse
# 0), so their catalogues come from the brands' own storefronts - see
# BRAND_STORE_DOMAINS below and app/services/brand_store.py.
"udhayam": "udhaiyam",
"udhaiyam dhall": "udhaiyam",
"uthayam": "udhaiyam",
"gopuram products": "gopuram",
#
# TENALI DOUBLE HORSE IS A DIFFERENT COMPANY FROM DOUBLE HORSE.
#
# Double Horse is Manjilas, in Kerala; Tenali Double Horse is an Andhra
# rice and rava brand. They share a name and nothing else.
#
# This entry is LOAD-BEARING and must stay a DIRECT one. `_contains_word`
# matches whole words in either direction, and "double horse" is a whole
# word sequence inside "tenali double horse" - so the moment any
# "double horse" key exists, the fallback loop folds the Andhra brand into
# the Kerala one and both companies' products land in a single table.
# `resolve_parent_brand` checks BRAND_ALIASES directly before it ever
# reaches that loop, which is the only reason this works.
#
# BOTH names need a direct self-entry, because `_contains_word` is applied
# in BOTH directions. Registering only the longer one moved the collision
# rather than fixing it - measured: with only "tenali double horse" present,
# resolve_parent_brand("Double Horse") returned "tenali double horse",
# because "double horse" is a whole word sequence inside the alias key and
# the loop matches `_contains_word(alias, key)` too. Kerala's brand was
# swallowed by Andhra's instead of the other way round.
#
# A direct hit short-circuits before the loop, so each of these entries
# protects the OTHER brand. Remove either one and the two merge.
"tenali double horse": "tenali double horse",
"tenali doublehorse": "tenali double horse",
"double horse": "double horse",
"doublehorse": "double horse",
}
# A brand's own online shop, verified by a person.
#
# NOT resolved by search, on purpose. `retail_presence.resolve_brand_domain`
# answers "Naga" with `cityofnagacebu.gov.ph` - the Philippine city - and a
# wrong domain here does not cost one bad row, it imports a hundred of another
# company's real products under this brand's name. `brand_store` treats a
# domain from this map as already checked and applies its heuristic identity
# guard only to domains that came from search.
#
# Add a brand here only after opening the site and confirming it is theirs.
BRAND_STORE_DOMAINS: dict[str, str] = {
"aachi": "aachifoods.com",
"anil": "shop.theanilgroup.com",
"double horse": "doublehorse.in",
"gopuram": "gopuramproducts.com",
"udhaiyam": "udhaiyamdhall.com",
# Naga and Tenali Double Horse are deliberately ABSENT: their official
# sites have not been confirmed. An absent brand simply has no store
# catalogue, which is the safe outcome; a guessed one is not.
}
def get_brand_store_domain(brand: str) -> Optional[str]:
"""The verified storefront for a brand, or None if we do not have one."""
return BRAND_STORE_DOMAINS.get(resolve_parent_brand(brand).lower().strip())
DEFAULT_ALIASES = BRAND_ALIASES

571
app/services/brand_store.py Normal file
View File

@@ -0,0 +1,571 @@
"""
Read a brand's real catalogue from the brand's own shop.
WHY THIS FILE EXISTS
--------------------
`off_bulk.fetch_brand_corpus()` is the only thing in this codebase that can
ENUMERATE a brand's products. Everything else - `retail_presence`,
`sku_service`, `manufacturer_site` - is a per-product VERIFIER: it needs a
product name as input and answers yes or no. So for a brand Open Food Facts has
never heard of there is nothing to verify, and the catalogue is whatever
`qwen2.5:1.5b` invents.
Measured 2026-09-10, Open Food Facts hits: Udhaiyam 0, Tenali Double Horse 0,
Gopuram 0, Double Horse 14, and Naga 2 - where both "Naga" rows are a UK /
Bangladeshi pickle brand, not the Tamil Nadu one. Regional South Indian brands
are simply not in that database.
They are, however, on their own shop, and modern storefronts publish their
whole catalogue as structured JSON:
doublehorse.in Shopify 91 products in one request
gopuramproducts.com WooCommerce 100+
aachifoods.com Shopify 123 (Open Food Facts has 61)
shop.theanilgroup.com Shopify 52 (Open Food Facts has 8)
and each record carries what the pipeline needs, from the manufacturer itself:
title Soan Papdi Ghee
product_type Snacks vendor Aachifoods
variants[0] title='200g' price=76.00 sku='8904209319340' grams=215
images[0] https://cdn.shopify.com/.../ghee-soan-papdi.webp
That SKU passes `validators.validate_barcode` - a real EAN-13 on the Indian 890
GS1 prefix. Name, category, pack size, price, images and a barcode, none of it
guessed.
THE RISK THIS MODULE CARRIES, AND THE GUARD ON IT
--------------------------------------------------
Every other source here verifies one product at a time, so a mistake costs one
bad row. This one ASSERTS a hundred products at once, so pointing it at the
wrong site imports a hundred bad rows under a real brand's name - and they would
look impeccable, because they are a real catalogue, just somebody else's.
That is not hypothetical. `retail_presence.resolve_brand_domain("Naga")`
returns `cityofnagacebu.gov.ph`, the website of Naga City in the Philippines.
So: the domain comes from a human-curated map (`brand_registry.
BRAND_STORE_DOMAINS`), and the catalogue is additionally checked against the
brand before any of it is accepted - see `_store_identity_ok`. A catalogue that
fails is rejected WHOLE. Half a foreign catalogue is not better than all of it.
STRUCTURED ENDPOINTS ONLY
-------------------------
Shopify's `products.json`, WooCommerce's Store API, and failing those a product
sitemap plus each page's JSON-LD. No HTML scraping and no Playwright: these are
documented, stable, paginated interfaces, and a brand that offers none of them
is better reported as "no store catalogue" than guessed at.
"""
from __future__ import annotations
import json
import logging
import os
import re
import time
from pathlib import Path
from typing import Any, Dict, List, Optional, Sequence
import requests
from app.services.category_units import COUNT_UNITS, VOLUME_UNITS, WEIGHT_UNITS, parse_unit
from app.services.enrichment.barcode.matching import brand_matches
from app.services.enrichment.barcode.retry import with_retry
from app.services.enrichment.barcode.validators import validate_barcode
logger = logging.getLogger(__name__)
# The same identifying User-Agent off_bulk uses, deliberately NOT the
# browser-spoofing strings elsewhere in this codebase. These are small
# companies' own websites rather than marketplaces, and a crawler reading them
# should say who it is and how to be reached.
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
# Same figure off_bulk uses between pages.
PAUSE_SECONDS = 2.0
_BACKEND_DIR = Path(__file__).resolve().parent.parent.parent
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "brand_store"
TIMEOUT_SECONDS = 25
SHOPIFY_PAGE_SIZE = 250
WOO_PAGE_SIZE = 100
MAX_PAGES = 12 # 3000 Shopify products; no FMCG brand here is close
# Units a pack size may legitimately carry - the same set brand_discovery uses,
# imported from category_units rather than from brand_discovery, because
# brand_discovery imports THIS module and the reverse would close a cycle.
_SIZE_UNITS = WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS
# Shopify's placeholder when a product has no real variants.
_NO_VARIANT = {"default title", "default", ""}
# ---------------------------------------------------------------------------
# HTTP
# ---------------------------------------------------------------------------
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
def _get(url: str, params: Optional[Dict[str, Any]] = None) -> requests.Response:
return requests.get(
url, params=params or {},
headers={"User-Agent": USER_AGENT, "Accept": "application/json, */*"},
timeout=TIMEOUT_SECONDS,
)
def _json_or_none(resp: requests.Response) -> Optional[Any]:
"""Parse a JSON body, tolerating a UTF-8 BOM.
WooCommerce really does serve one: gopuramproducts.com's Store API response
begins with EF BB BF, and `resp.json()` raises
"Unexpected UTF-8 BOM (decode using utf-8-sig)". `brand_sync._read_catalog`
already reads seed files as utf-8-sig for the same reason.
"""
if resp.status_code != 200:
return None
content_type = (resp.headers.get("content-type") or "").lower()
if "json" not in content_type:
return None
try:
return json.loads(resp.content.decode("utf-8-sig"))
except Exception as e: # noqa: BLE001 - a malformed body is "no catalogue"
logger.debug("Unparseable JSON from %s: %s", resp.url, e)
return None
# ---------------------------------------------------------------------------
# Sizes
# ---------------------------------------------------------------------------
def canonical_size(raw: Optional[str]) -> Optional[str]:
"""A pack size in the one form the pipeline hashes consistently.
Mirrors `brand_discovery._canonical_size` exactly - "200 g" and "200g" are
the same pack, but `build_image_id` slugifies them differently and would
make two permanent rows. Returns None for anything that is not a size: a
bare count, "Default Title", a colour.
"""
if not raw or not str(raw).strip():
return None
value, unit = parse_unit(str(raw))
if value is None or not unit or unit not in _SIZE_UNITS:
return None
number = str(int(value)) if float(value).is_integer() else str(value)
return f"{number}{unit}"
# ---------------------------------------------------------------------------
# Identity - the guard that matters
# ---------------------------------------------------------------------------
def _store_identity_ok(brand: str, domain: str, vendors: Sequence[str],
aliases: Optional[Sequence[str]] = None) -> bool:
"""Does this catalogue actually belong to `brand`?
Accepts on either signal, because neither alone is reliable:
- the DOMAIN names the brand (`doublehorse.in` for Double Horse), or
- the catalogue's own `vendor` field names it (Shopify's "Aachifoods").
Some stores leave `vendor` as the shop's theme name or blank, so requiring
it would reject good catalogues; and a brand can trade on a domain that
does not contain its name, so requiring that would too.
What this DOES stop is the case it was written for: a catalogue fetched
from a site belonging to neither the brand's domain nor its name, which is
how `cityofnagacebu.gov.ph` would otherwise have become Naga's product
list.
"""
aliases = list(aliases or [])
stem = (domain or "").split(".")[0].lower()
domain_words = set(re.split(r"[^a-z0-9]+", stem))
brand_words = [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if len(w) > 2]
if brand_words and all(w in domain_words for w in brand_words):
return True
# PREFIX, NOT SUBSTRING. A brand's own domain is the brand followed by a
# qualifier - "gopuramproducts", "udhaiyamdhall", "doublehorse". A domain
# that merely CONTAINS the brand somewhere in the middle is a different
# organisation that happens to share a word:
#
# "gopuramproducts".startswith("gopuram") -> True, correct
# "cityofnagacebu".startswith("naga") -> False, correct
# "cityofnagacebu" contains "naga" -> True, THE BUG
#
# This is the whole reason `_pick_brand_domain` handed back the website of
# Naga City in the Philippines as a Tamil Nadu food brand's catalogue.
if brand_words and stem.startswith("".join(brand_words)):
return True
for vendor in vendors:
if vendor and brand_matches(vendor, brand, aliases):
return True
return False
# ---------------------------------------------------------------------------
# Shopify
# ---------------------------------------------------------------------------
def fetch_shopify(domain: str) -> Optional[List[Dict[str, Any]]]:
"""Raw Shopify products, or None when this is not a Shopify store."""
out: List[Dict[str, Any]] = []
for page in range(1, MAX_PAGES + 1):
try:
resp = _get(f"https://{domain}/products.json",
{"limit": SHOPIFY_PAGE_SIZE, "page": page})
except Exception as e: # noqa: BLE001 - unreachable is not a verdict
logger.debug("Shopify fetch failed for %s p%d: %s", domain, page, e)
return out or None
data = _json_or_none(resp)
if not isinstance(data, dict) or "products" not in data:
return out or None
products = data.get("products") or []
if not products:
break
out.extend(products)
if len(products) < SHOPIFY_PAGE_SIZE:
break
time.sleep(PAUSE_SECONDS)
return out or None
def _shopify_candidates(product: Dict[str, Any]) -> List[Dict[str, Any]]:
title = str(product.get("title") or "").strip()
if not title:
return []
category = str(product.get("product_type") or "").strip() or None
images = [
str(i.get("src")) for i in (product.get("images") or [])
if isinstance(i, dict) and i.get("src")
]
variants = product.get("variants") or []
rows: List[Dict[str, Any]] = []
for variant in variants:
if not isinstance(variant, dict):
continue
raw_size = variant.get("title")
if str(raw_size or "").strip().lower() in _NO_VARIANT:
# No real variant axis. The size, if any, is in the product name,
# and `_resolve_sizes` reads a title's own size first anyway.
raw_size = variant.get("option1")
size = canonical_size(raw_size)
# The SKU is a barcode only when it IS one. Shopify shops put anything
# here - Aachi puts a real EAN-13 ("8904209319340"), Gopuram leaves it
# blank, others use an internal code. `validate_barcode` decides, so a
# shop's internal numbering never reaches the barcode column.
sku = str(variant.get("sku") or "").strip()
barcode = validate_barcode(sku) if sku else None
if not barcode:
raw_barcode = str(variant.get("barcode") or "").strip()
barcode = validate_barcode(raw_barcode) if raw_barcode else None
rows.append({
"title": title,
"category": category,
"size": size,
"barcode": barcode,
"price": _decimal_or_none(variant.get("price")),
"image_url": images[0] if images else None,
"image_urls": images,
"vendor": str(product.get("vendor") or "").strip() or None,
"source": "store",
})
if not rows:
rows.append({
"title": title, "category": category, "size": None, "barcode": None,
"price": None, "image_url": images[0] if images else None,
"image_urls": images,
"vendor": str(product.get("vendor") or "").strip() or None,
"source": "store",
})
return rows
# ---------------------------------------------------------------------------
# WooCommerce Store API
# ---------------------------------------------------------------------------
def fetch_woocommerce(domain: str) -> Optional[List[Dict[str, Any]]]:
"""Raw WooCommerce Store API products, or None when unavailable.
udhaiyamdhall.com is WooCommerce but answers this endpoint with 403, which
is why the sitemap tier below exists.
"""
out: List[Dict[str, Any]] = []
for page in range(1, MAX_PAGES + 1):
try:
resp = _get(f"https://{domain}/wp-json/wc/store/products",
{"per_page": WOO_PAGE_SIZE, "page": page})
except Exception as e: # noqa: BLE001
logger.debug("Woo fetch failed for %s p%d: %s", domain, page, e)
return out or None
data = _json_or_none(resp)
if not isinstance(data, list):
return out or None
if not data:
break
out.extend(data)
if len(data) < WOO_PAGE_SIZE:
break
time.sleep(PAUSE_SECONDS)
return out or None
def _woo_price(prices: Dict[str, Any]) -> Optional[float]:
"""WooCommerce reports MINOR UNITS.
gopuramproducts.com returns `price: "5500"` with `currency_minor_unit: 2`,
which is Rs55.00 and not Rs5,500. Read straight, every price on the site is
a hundred times too large - and `product_validator.validate_price_range`
would then reject the row as an implausible price, so the failure would
surface as "this brand's products are all wrong" rather than as a units bug.
"""
raw = prices.get("price")
if raw is None or str(raw).strip() == "":
return None
try:
minor = int(prices.get("currency_minor_unit", 2))
return float(raw) / (10 ** minor)
except Exception: # noqa: BLE001
return None
def _woo_candidates(product: Dict[str, Any]) -> List[Dict[str, Any]]:
title = str(product.get("name") or "").strip()
if not title:
return []
categories = [
str(c.get("name")).strip() for c in (product.get("categories") or [])
if isinstance(c, dict) and c.get("name")
]
images = [
str(i.get("src")) for i in (product.get("images") or [])
if isinstance(i, dict) and i.get("src")
]
sku = str(product.get("sku") or "").strip()
return [{
"title": title,
"category": categories[0] if categories else None,
"size": canonical_size(title),
"barcode": validate_barcode(sku) if sku else None,
"price": _woo_price(product.get("prices") or {}),
"image_url": images[0] if images else None,
"image_urls": images,
"vendor": None,
"source": "store",
}]
def _decimal_or_none(raw: Any) -> Optional[float]:
if raw is None or str(raw).strip() == "":
return None
try:
return float(raw)
except Exception: # noqa: BLE001
return None
# ---------------------------------------------------------------------------
# Tier 3 - product sitemap, for sites that publish no catalogue API
# ---------------------------------------------------------------------------
# udhaiyamdhall.com is WooCommerce but answers the Store API with 403, and its
# product pages carry only a BreadcrumbList in JSON-LD - no Product node, no
# og: tags. What it does have is `wp-sitemap-posts-product-1.xml` listing 32
# products, each page with an <h1> and a gallery image.
#
# So this tier yields NAMES AND IMAGES, AND NOTHING ELSE. No price, no pack
# size, no barcode - because the site does not state them, and a catalogue
# source that guesses is the exact problem this whole area exists to fix. The
# names are real and the photographs are the brand's own, which solves
# enumeration and solves the image problem; sizes have to come from somewhere
# that actually knows them.
_SITEMAP_CANDIDATES = ("/sitemap.xml", "/wp-sitemap.xml", "/sitemap_index.xml")
MAX_SITEMAP_PAGES = 200
def _sitemap_locs(xml: str) -> List[str]:
return re.findall(r"<loc>\s*(.*?)\s*</loc>", xml or "", re.IGNORECASE | re.DOTALL)
def fetch_sitemap(domain: str) -> Optional[List[Dict[str, Any]]]:
"""Product names and images from a site's product sitemap, or None."""
try:
from bs4 import BeautifulSoup
except ImportError: # pragma: no cover - declared in requirements.txt
logger.debug("Sitemap tier unavailable: beautifulsoup4 not installed")
return None
product_urls: List[str] = []
for path in _SITEMAP_CANDIDATES:
try:
resp = _get(f"https://{domain}{path}")
except Exception: # noqa: BLE001
continue
if resp.status_code != 200 or "xml" not in (resp.headers.get("content-type") or ""):
continue
locs = _sitemap_locs(resp.text)
# A sitemap index points at sub-sitemaps; take only the product one.
for loc in locs:
if "product" not in loc.lower() or "categor" in loc.lower():
continue
if loc.lower().endswith(".xml"):
try:
sub = _get(loc)
except Exception: # noqa: BLE001
continue
if sub.status_code == 200:
product_urls.extend(
u for u in _sitemap_locs(sub.text) if "/product/" in u.lower()
)
elif "/product/" in loc.lower():
product_urls.append(loc)
if product_urls:
break
product_urls = list(dict.fromkeys(product_urls))[:MAX_SITEMAP_PAGES]
if not product_urls:
return None
out: List[Dict[str, Any]] = []
for url in product_urls:
try:
page = _get(url)
except Exception: # noqa: BLE001 - one dead page cannot stop the sweep
continue
if page.status_code != 200:
continue
soup = BeautifulSoup(page.text, "lxml")
heading = soup.find("h1")
title = heading.get_text(strip=True) if heading else ""
if not title:
continue
image = None
node = soup.select_one(".woocommerce-product-gallery img, .wp-post-image, "
"meta[property='og:image']")
if node is not None:
image = node.get("src") or node.get("data-src") or node.get("content")
out.append({
"title": title,
"category": None,
# Left None deliberately - this site states neither, and inventing
# them is the failure mode, not the fallback.
"size": None,
"barcode": None,
"price": None,
"image_url": image,
"image_urls": [image] if image else [],
"vendor": None,
"source": "store",
})
time.sleep(1.0)
return out or None
# ---------------------------------------------------------------------------
# Public entry point
# ---------------------------------------------------------------------------
def _cache_path(brand: str) -> Path:
slug = re.sub(r"[^a-z0-9]+", "_", (brand or "").lower()).strip("_")
return CACHE_DIR / f"{slug}.json"
def _write_cache(path: Path, payload: Dict[str, Any]) -> None:
"""Atomic write, the same .tmp + os.replace off_bulk uses - a half-written
catalogue read by the next run is worse than no cache."""
try:
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(".tmp")
tmp.write_text(json.dumps(payload, indent=2), encoding="utf-8")
os.replace(tmp, path)
except Exception as e: # noqa: BLE001 - caching is best effort
logger.debug("Could not cache brand store catalogue at %s: %s", path, e)
def read_cached(brand: str) -> Optional[List[Dict[str, Any]]]:
"""The cached catalogue for a brand, or None. Never raises."""
path = _cache_path(brand)
if not path.exists():
return None
try:
data = json.loads(path.read_text(encoding="utf-8-sig"))
return list(data.get("products") or [])
except Exception as e: # noqa: BLE001
logger.debug("Unreadable brand store cache %s: %s", path, e)
return None
def fetch_store_catalogue(brand: str, domain: Optional[str], *,
live: bool = False, refresh: bool = False,
trusted_domain: bool = False,
aliases: Optional[Sequence[str]] = None,
) -> Optional[List[Dict[str, Any]]]:
"""The brand's own catalogue as candidate dicts, or None if there isn't one.
`live` defaults to FALSE, matching `retail_presence.check_listing`: the
runtime path reads the cache and the backfill script does the fetching, so
a discovery preview never blocks on somebody's storefront being slow.
None means "no catalogue" - not reachable, not a supported platform, or
rejected by the identity guard. An empty list is not returned; a store with
zero products is indistinguishable from no store and is reported the same.
"""
if not refresh:
cached = read_cached(brand)
if cached is not None:
return cached or None
if not live or not domain:
return None
raw = fetch_shopify(domain)
platform = "shopify"
to_candidates = _shopify_candidates
if raw is None:
raw = fetch_woocommerce(domain)
platform = "woocommerce"
to_candidates = _woo_candidates
if raw is None:
# Last tier: names and images off the product sitemap. See fetch_sitemap
# for why it yields nothing else.
raw = fetch_sitemap(domain)
platform = "sitemap"
to_candidates = lambda row: [row] # noqa: E731 - already candidates
if raw is None:
logger.info("No structured catalogue at %s for %s", domain, brand)
return None
candidates: List[Dict[str, Any]] = []
for product in raw:
if isinstance(product, dict):
candidates.extend(to_candidates(product))
vendors = sorted({c.get("vendor") for c in candidates if c.get("vendor")})
# A CURATED DOMAIN IS ALREADY VERIFIED, BY A PERSON.
#
# The heuristic below cannot recognise every legitimate shape a brand's
# domain takes - "theanilgroup.com" is Anil's, and neither starts with
# "anil" nor contains it as a whole token - so applying it to a
# hand-checked mapping would reject good catalogues for looking unusual.
# It runs only on domains that came from automatic resolution, which is
# the path that produced cityofnagacebu.gov.ph.
if not trusted_domain and not _store_identity_ok(brand, domain, vendors, aliases):
# REJECTED WHOLE, on purpose. This is the failure that would otherwise
# put a hundred of somebody else's products under this brand's name,
# each of them a real product and none of them theirs.
logger.warning(
"Rejecting catalogue at %s for brand %r: neither the domain nor "
"its vendors (%s) identify this brand",
domain, brand, ", ".join(vendors) or "none stated",
)
return None
for c in candidates:
c.pop("vendor", None)
logger.info("%s: %d products from %s (%s)", brand, len(candidates), domain, platform)
_write_cache(_cache_path(brand), {
"brand": brand,
"domain": domain,
"platform": platform,
"fetched_at": time.time(),
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
"products": candidates,
})
return candidates or None