unpopular brand generation
This commit is contained in:
@@ -105,7 +105,7 @@ from app.infrastructure.settings import (
|
||||
BRAND_DISCOVERY_USE_LLM,
|
||||
BRAND_DISCOVERY_USE_OFF,
|
||||
)
|
||||
from app.services import active_brands, ollama_service, retail_presence
|
||||
from app.services import active_brands, brand_store, ollama_service, retail_presence
|
||||
from app.services.brand_registry import (
|
||||
get_fssai_license,
|
||||
get_known_sub_brands,
|
||||
@@ -481,6 +481,33 @@ def _from_open_facts(brand: str, *, refresh: bool = False) -> List[Dict[str, Any
|
||||
return out
|
||||
|
||||
|
||||
def _from_brand_store(brand: str) -> List[Dict[str, Any]]:
|
||||
"""The brand's own shop catalogue - the strongest source there is.
|
||||
|
||||
Open Food Facts is crowd-sourced and, for a regional South Indian brand,
|
||||
usually empty: Udhaiyam 0 hits, Gopuram 0, Tenali Double Horse 0. Their
|
||||
storefronts publish the whole catalogue as structured JSON - Double Horse
|
||||
99 rows, Gopuram 122 - with real names, prices, images and sometimes a
|
||||
valid EAN-13.
|
||||
|
||||
CACHE ONLY. `scripts/backfill_brand_stores.py` does the fetching, exactly
|
||||
as the retail check is split, so a discovery preview never blocks on
|
||||
somebody's storefront being slow or down.
|
||||
|
||||
Pack sizes are deliberately left as the store stated them, which is often
|
||||
nothing: most of these shops put the size in the PRODUCT NAME rather than
|
||||
in a variant ("Roasted Vermicelli 400g"). `_resolve_sizes` already reads a
|
||||
title's own size first and is the one place that rule should live -
|
||||
measured, that lifts Double Horse from 14 sized rows to 94 of 99.
|
||||
"""
|
||||
try:
|
||||
rows = brand_store.fetch_store_catalogue(brand, None, live=False)
|
||||
except Exception as exc: # noqa: BLE001 - a bad cache must not kill discovery
|
||||
logger.warning("Brand store catalogue unreadable for %r: %s", brand, exc)
|
||||
return []
|
||||
return list(rows or [])
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Source B - Ollama (breadth and prose)
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -669,6 +696,21 @@ def _score(sources: Sequence[str], evidence: Optional[str]) -> float:
|
||||
"""
|
||||
has_off = "off" in sources
|
||||
has_llm = "llm" in sources
|
||||
has_store = "store" in sources
|
||||
# THE BRAND'S OWN SHOP OUTRANKS OPEN FOOD FACTS.
|
||||
#
|
||||
# OFF is crowd-sourced and, for these brands, mostly absent or wrong - its
|
||||
# two "Naga" rows are a British pickle. A manufacturer's own storefront is
|
||||
# first-party: it is the definitive list of what the company sells, at the
|
||||
# sizes and prices it sells them, with its own photographs.
|
||||
#
|
||||
# Scored above `off` rather than merged into it so the preview can say
|
||||
# which source a row came from, and so a future disagreement between the
|
||||
# two is visible rather than silently resolved.
|
||||
if has_store and (has_off or has_llm):
|
||||
return 1.0
|
||||
if has_store:
|
||||
return 0.95
|
||||
if has_off and has_llm:
|
||||
return 1.0
|
||||
if has_off:
|
||||
@@ -866,6 +908,7 @@ def discover_brand_products(
|
||||
deadline_seconds: float = BRAND_DISCOVERY_DEADLINE_SECONDS,
|
||||
use_openfacts: bool = BRAND_DISCOVERY_USE_OFF,
|
||||
use_llm: bool = BRAND_DISCOVERY_USE_LLM,
|
||||
use_store: bool = True,
|
||||
require_evidence: bool = True,
|
||||
refresh_corpus: bool = False,
|
||||
) -> DiscoveryResult:
|
||||
@@ -898,6 +941,24 @@ def discover_brand_products(
|
||||
if key:
|
||||
off_keys.setdefault(key, candidate)
|
||||
|
||||
# The brand's own shop. Read from cache only - see `_from_brand_store`.
|
||||
store_candidates = _from_brand_store(brand) if use_store else []
|
||||
store_keys = {}
|
||||
for candidate in store_candidates:
|
||||
key = _normalise_title(brand, candidate.get("title") or "")
|
||||
if key:
|
||||
store_keys.setdefault(key, candidate)
|
||||
if store_candidates:
|
||||
logger.info("%s: %d products from the brand's own shop", brand, len(store_candidates))
|
||||
elif use_store and not off_candidates:
|
||||
warnings.append(
|
||||
"Neither Open Food Facts nor a brand storefront has anything for "
|
||||
"this brand. Every row below rests on the language model alone - "
|
||||
"review them individually. If this brand has its own online shop, "
|
||||
"adding it to brand_registry.BRAND_STORE_DOMAINS and running "
|
||||
"scripts/backfill_brand_stores.py will populate a real catalogue."
|
||||
)
|
||||
|
||||
|
||||
llm_candidates: List[Dict[str, Any]] = []
|
||||
if use_llm:
|
||||
@@ -920,7 +981,10 @@ def discover_brand_products(
|
||||
# and adopts the size of the one it joins.
|
||||
merged: List[Dict[str, Any]] = []
|
||||
keys: List[str] = []
|
||||
for candidate in off_candidates + llm_candidates:
|
||||
# STORE FIRST, then Open Food Facts, then the model. The merge fills a
|
||||
# blank from whatever comes later, so order IS precedence: the brand's own
|
||||
# shop outranks a crowd-sourced database, which outranks a guess.
|
||||
for candidate in store_candidates + off_candidates + llm_candidates:
|
||||
title = candidate["title"]
|
||||
key = _normalise_title(brand, title)
|
||||
target = None
|
||||
@@ -1056,7 +1120,8 @@ def discover_brand_products(
|
||||
)
|
||||
if product is None:
|
||||
continue
|
||||
if require_evidence and "off" not in product.sources and not product.evidence:
|
||||
if (require_evidence and "off" not in product.sources
|
||||
and "store" not in product.sources and not product.evidence):
|
||||
dropped += 1
|
||||
continue
|
||||
products.append(product)
|
||||
|
||||
@@ -261,8 +261,70 @@ BRAND_ALIASES = {
|
||||
# that spells it without the s. Removing it re-opens the bug.
|
||||
"haldiram": "haldirams",
|
||||
"haldiram's": "haldirams",
|
||||
# --- Regional South Indian brands -------------------------------------
|
||||
# Not in Open Food Facts (Udhaiyam 0 hits, Gopuram 0, Tenali Double Horse
|
||||
# 0), so their catalogues come from the brands' own storefronts - see
|
||||
# BRAND_STORE_DOMAINS below and app/services/brand_store.py.
|
||||
"udhayam": "udhaiyam",
|
||||
"udhaiyam dhall": "udhaiyam",
|
||||
"uthayam": "udhaiyam",
|
||||
"gopuram products": "gopuram",
|
||||
#
|
||||
# TENALI DOUBLE HORSE IS A DIFFERENT COMPANY FROM DOUBLE HORSE.
|
||||
#
|
||||
# Double Horse is Manjilas, in Kerala; Tenali Double Horse is an Andhra
|
||||
# rice and rava brand. They share a name and nothing else.
|
||||
#
|
||||
# This entry is LOAD-BEARING and must stay a DIRECT one. `_contains_word`
|
||||
# matches whole words in either direction, and "double horse" is a whole
|
||||
# word sequence inside "tenali double horse" - so the moment any
|
||||
# "double horse" key exists, the fallback loop folds the Andhra brand into
|
||||
# the Kerala one and both companies' products land in a single table.
|
||||
# `resolve_parent_brand` checks BRAND_ALIASES directly before it ever
|
||||
# reaches that loop, which is the only reason this works.
|
||||
#
|
||||
# BOTH names need a direct self-entry, because `_contains_word` is applied
|
||||
# in BOTH directions. Registering only the longer one moved the collision
|
||||
# rather than fixing it - measured: with only "tenali double horse" present,
|
||||
# resolve_parent_brand("Double Horse") returned "tenali double horse",
|
||||
# because "double horse" is a whole word sequence inside the alias key and
|
||||
# the loop matches `_contains_word(alias, key)` too. Kerala's brand was
|
||||
# swallowed by Andhra's instead of the other way round.
|
||||
#
|
||||
# A direct hit short-circuits before the loop, so each of these entries
|
||||
# protects the OTHER brand. Remove either one and the two merge.
|
||||
"tenali double horse": "tenali double horse",
|
||||
"tenali doublehorse": "tenali double horse",
|
||||
"double horse": "double horse",
|
||||
"doublehorse": "double horse",
|
||||
}
|
||||
|
||||
# A brand's own online shop, verified by a person.
|
||||
#
|
||||
# NOT resolved by search, on purpose. `retail_presence.resolve_brand_domain`
|
||||
# answers "Naga" with `cityofnagacebu.gov.ph` - the Philippine city - and a
|
||||
# wrong domain here does not cost one bad row, it imports a hundred of another
|
||||
# company's real products under this brand's name. `brand_store` treats a
|
||||
# domain from this map as already checked and applies its heuristic identity
|
||||
# guard only to domains that came from search.
|
||||
#
|
||||
# Add a brand here only after opening the site and confirming it is theirs.
|
||||
BRAND_STORE_DOMAINS: dict[str, str] = {
|
||||
"aachi": "aachifoods.com",
|
||||
"anil": "shop.theanilgroup.com",
|
||||
"double horse": "doublehorse.in",
|
||||
"gopuram": "gopuramproducts.com",
|
||||
"udhaiyam": "udhaiyamdhall.com",
|
||||
# Naga and Tenali Double Horse are deliberately ABSENT: their official
|
||||
# sites have not been confirmed. An absent brand simply has no store
|
||||
# catalogue, which is the safe outcome; a guessed one is not.
|
||||
}
|
||||
|
||||
|
||||
def get_brand_store_domain(brand: str) -> Optional[str]:
|
||||
"""The verified storefront for a brand, or None if we do not have one."""
|
||||
return BRAND_STORE_DOMAINS.get(resolve_parent_brand(brand).lower().strip())
|
||||
|
||||
DEFAULT_ALIASES = BRAND_ALIASES
|
||||
|
||||
|
||||
|
||||
571
app/services/brand_store.py
Normal file
571
app/services/brand_store.py
Normal file
@@ -0,0 +1,571 @@
|
||||
"""
|
||||
Read a brand's real catalogue from the brand's own shop.
|
||||
|
||||
WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
`off_bulk.fetch_brand_corpus()` is the only thing in this codebase that can
|
||||
ENUMERATE a brand's products. Everything else - `retail_presence`,
|
||||
`sku_service`, `manufacturer_site` - is a per-product VERIFIER: it needs a
|
||||
product name as input and answers yes or no. So for a brand Open Food Facts has
|
||||
never heard of there is nothing to verify, and the catalogue is whatever
|
||||
`qwen2.5:1.5b` invents.
|
||||
|
||||
Measured 2026-09-10, Open Food Facts hits: Udhaiyam 0, Tenali Double Horse 0,
|
||||
Gopuram 0, Double Horse 14, and Naga 2 - where both "Naga" rows are a UK /
|
||||
Bangladeshi pickle brand, not the Tamil Nadu one. Regional South Indian brands
|
||||
are simply not in that database.
|
||||
|
||||
They are, however, on their own shop, and modern storefronts publish their
|
||||
whole catalogue as structured JSON:
|
||||
|
||||
doublehorse.in Shopify 91 products in one request
|
||||
gopuramproducts.com WooCommerce 100+
|
||||
aachifoods.com Shopify 123 (Open Food Facts has 61)
|
||||
shop.theanilgroup.com Shopify 52 (Open Food Facts has 8)
|
||||
|
||||
and each record carries what the pipeline needs, from the manufacturer itself:
|
||||
|
||||
title Soan Papdi Ghee
|
||||
product_type Snacks vendor Aachifoods
|
||||
variants[0] title='200g' price=76.00 sku='8904209319340' grams=215
|
||||
images[0] https://cdn.shopify.com/.../ghee-soan-papdi.webp
|
||||
|
||||
That SKU passes `validators.validate_barcode` - a real EAN-13 on the Indian 890
|
||||
GS1 prefix. Name, category, pack size, price, images and a barcode, none of it
|
||||
guessed.
|
||||
|
||||
THE RISK THIS MODULE CARRIES, AND THE GUARD ON IT
|
||||
--------------------------------------------------
|
||||
Every other source here verifies one product at a time, so a mistake costs one
|
||||
bad row. This one ASSERTS a hundred products at once, so pointing it at the
|
||||
wrong site imports a hundred bad rows under a real brand's name - and they would
|
||||
look impeccable, because they are a real catalogue, just somebody else's.
|
||||
|
||||
That is not hypothetical. `retail_presence.resolve_brand_domain("Naga")`
|
||||
returns `cityofnagacebu.gov.ph`, the website of Naga City in the Philippines.
|
||||
|
||||
So: the domain comes from a human-curated map (`brand_registry.
|
||||
BRAND_STORE_DOMAINS`), and the catalogue is additionally checked against the
|
||||
brand before any of it is accepted - see `_store_identity_ok`. A catalogue that
|
||||
fails is rejected WHOLE. Half a foreign catalogue is not better than all of it.
|
||||
|
||||
STRUCTURED ENDPOINTS ONLY
|
||||
-------------------------
|
||||
Shopify's `products.json`, WooCommerce's Store API, and failing those a product
|
||||
sitemap plus each page's JSON-LD. No HTML scraping and no Playwright: these are
|
||||
documented, stable, paginated interfaces, and a brand that offers none of them
|
||||
is better reported as "no store catalogue" than guessed at.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Sequence
|
||||
|
||||
import requests
|
||||
|
||||
from app.services.category_units import COUNT_UNITS, VOLUME_UNITS, WEIGHT_UNITS, parse_unit
|
||||
from app.services.enrichment.barcode.matching import brand_matches
|
||||
from app.services.enrichment.barcode.retry import with_retry
|
||||
from app.services.enrichment.barcode.validators import validate_barcode
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# The same identifying User-Agent off_bulk uses, deliberately NOT the
|
||||
# browser-spoofing strings elsewhere in this codebase. These are small
|
||||
# companies' own websites rather than marketplaces, and a crawler reading them
|
||||
# should say who it is and how to be reached.
|
||||
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
|
||||
|
||||
# Same figure off_bulk uses between pages.
|
||||
PAUSE_SECONDS = 2.0
|
||||
|
||||
_BACKEND_DIR = Path(__file__).resolve().parent.parent.parent
|
||||
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "brand_store"
|
||||
|
||||
TIMEOUT_SECONDS = 25
|
||||
SHOPIFY_PAGE_SIZE = 250
|
||||
WOO_PAGE_SIZE = 100
|
||||
MAX_PAGES = 12 # 3000 Shopify products; no FMCG brand here is close
|
||||
|
||||
# Units a pack size may legitimately carry - the same set brand_discovery uses,
|
||||
# imported from category_units rather than from brand_discovery, because
|
||||
# brand_discovery imports THIS module and the reverse would close a cycle.
|
||||
_SIZE_UNITS = WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS
|
||||
|
||||
# Shopify's placeholder when a product has no real variants.
|
||||
_NO_VARIANT = {"default title", "default", ""}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# HTTP
|
||||
# ---------------------------------------------------------------------------
|
||||
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
|
||||
def _get(url: str, params: Optional[Dict[str, Any]] = None) -> requests.Response:
|
||||
return requests.get(
|
||||
url, params=params or {},
|
||||
headers={"User-Agent": USER_AGENT, "Accept": "application/json, */*"},
|
||||
timeout=TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
def _json_or_none(resp: requests.Response) -> Optional[Any]:
|
||||
"""Parse a JSON body, tolerating a UTF-8 BOM.
|
||||
|
||||
WooCommerce really does serve one: gopuramproducts.com's Store API response
|
||||
begins with EF BB BF, and `resp.json()` raises
|
||||
"Unexpected UTF-8 BOM (decode using utf-8-sig)". `brand_sync._read_catalog`
|
||||
already reads seed files as utf-8-sig for the same reason.
|
||||
"""
|
||||
if resp.status_code != 200:
|
||||
return None
|
||||
content_type = (resp.headers.get("content-type") or "").lower()
|
||||
if "json" not in content_type:
|
||||
return None
|
||||
try:
|
||||
return json.loads(resp.content.decode("utf-8-sig"))
|
||||
except Exception as e: # noqa: BLE001 - a malformed body is "no catalogue"
|
||||
logger.debug("Unparseable JSON from %s: %s", resp.url, e)
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Sizes
|
||||
# ---------------------------------------------------------------------------
|
||||
def canonical_size(raw: Optional[str]) -> Optional[str]:
|
||||
"""A pack size in the one form the pipeline hashes consistently.
|
||||
|
||||
Mirrors `brand_discovery._canonical_size` exactly - "200 g" and "200g" are
|
||||
the same pack, but `build_image_id` slugifies them differently and would
|
||||
make two permanent rows. Returns None for anything that is not a size: a
|
||||
bare count, "Default Title", a colour.
|
||||
"""
|
||||
if not raw or not str(raw).strip():
|
||||
return None
|
||||
value, unit = parse_unit(str(raw))
|
||||
if value is None or not unit or unit not in _SIZE_UNITS:
|
||||
return None
|
||||
number = str(int(value)) if float(value).is_integer() else str(value)
|
||||
return f"{number}{unit}"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Identity - the guard that matters
|
||||
# ---------------------------------------------------------------------------
|
||||
def _store_identity_ok(brand: str, domain: str, vendors: Sequence[str],
|
||||
aliases: Optional[Sequence[str]] = None) -> bool:
|
||||
"""Does this catalogue actually belong to `brand`?
|
||||
|
||||
Accepts on either signal, because neither alone is reliable:
|
||||
|
||||
- the DOMAIN names the brand (`doublehorse.in` for Double Horse), or
|
||||
- the catalogue's own `vendor` field names it (Shopify's "Aachifoods").
|
||||
|
||||
Some stores leave `vendor` as the shop's theme name or blank, so requiring
|
||||
it would reject good catalogues; and a brand can trade on a domain that
|
||||
does not contain its name, so requiring that would too.
|
||||
|
||||
What this DOES stop is the case it was written for: a catalogue fetched
|
||||
from a site belonging to neither the brand's domain nor its name, which is
|
||||
how `cityofnagacebu.gov.ph` would otherwise have become Naga's product
|
||||
list.
|
||||
"""
|
||||
aliases = list(aliases or [])
|
||||
stem = (domain or "").split(".")[0].lower()
|
||||
domain_words = set(re.split(r"[^a-z0-9]+", stem))
|
||||
brand_words = [w for w in re.split(r"[^a-z0-9]+", (brand or "").lower()) if len(w) > 2]
|
||||
if brand_words and all(w in domain_words for w in brand_words):
|
||||
return True
|
||||
|
||||
# PREFIX, NOT SUBSTRING. A brand's own domain is the brand followed by a
|
||||
# qualifier - "gopuramproducts", "udhaiyamdhall", "doublehorse". A domain
|
||||
# that merely CONTAINS the brand somewhere in the middle is a different
|
||||
# organisation that happens to share a word:
|
||||
#
|
||||
# "gopuramproducts".startswith("gopuram") -> True, correct
|
||||
# "cityofnagacebu".startswith("naga") -> False, correct
|
||||
# "cityofnagacebu" contains "naga" -> True, THE BUG
|
||||
#
|
||||
# This is the whole reason `_pick_brand_domain` handed back the website of
|
||||
# Naga City in the Philippines as a Tamil Nadu food brand's catalogue.
|
||||
if brand_words and stem.startswith("".join(brand_words)):
|
||||
return True
|
||||
for vendor in vendors:
|
||||
if vendor and brand_matches(vendor, brand, aliases):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Shopify
|
||||
# ---------------------------------------------------------------------------
|
||||
def fetch_shopify(domain: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Raw Shopify products, or None when this is not a Shopify store."""
|
||||
out: List[Dict[str, Any]] = []
|
||||
for page in range(1, MAX_PAGES + 1):
|
||||
try:
|
||||
resp = _get(f"https://{domain}/products.json",
|
||||
{"limit": SHOPIFY_PAGE_SIZE, "page": page})
|
||||
except Exception as e: # noqa: BLE001 - unreachable is not a verdict
|
||||
logger.debug("Shopify fetch failed for %s p%d: %s", domain, page, e)
|
||||
return out or None
|
||||
data = _json_or_none(resp)
|
||||
if not isinstance(data, dict) or "products" not in data:
|
||||
return out or None
|
||||
products = data.get("products") or []
|
||||
if not products:
|
||||
break
|
||||
out.extend(products)
|
||||
if len(products) < SHOPIFY_PAGE_SIZE:
|
||||
break
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
return out or None
|
||||
|
||||
|
||||
def _shopify_candidates(product: Dict[str, Any]) -> List[Dict[str, Any]]:
|
||||
title = str(product.get("title") or "").strip()
|
||||
if not title:
|
||||
return []
|
||||
category = str(product.get("product_type") or "").strip() or None
|
||||
images = [
|
||||
str(i.get("src")) for i in (product.get("images") or [])
|
||||
if isinstance(i, dict) and i.get("src")
|
||||
]
|
||||
variants = product.get("variants") or []
|
||||
|
||||
rows: List[Dict[str, Any]] = []
|
||||
for variant in variants:
|
||||
if not isinstance(variant, dict):
|
||||
continue
|
||||
raw_size = variant.get("title")
|
||||
if str(raw_size or "").strip().lower() in _NO_VARIANT:
|
||||
# No real variant axis. The size, if any, is in the product name,
|
||||
# and `_resolve_sizes` reads a title's own size first anyway.
|
||||
raw_size = variant.get("option1")
|
||||
size = canonical_size(raw_size)
|
||||
|
||||
# The SKU is a barcode only when it IS one. Shopify shops put anything
|
||||
# here - Aachi puts a real EAN-13 ("8904209319340"), Gopuram leaves it
|
||||
# blank, others use an internal code. `validate_barcode` decides, so a
|
||||
# shop's internal numbering never reaches the barcode column.
|
||||
sku = str(variant.get("sku") or "").strip()
|
||||
barcode = validate_barcode(sku) if sku else None
|
||||
if not barcode:
|
||||
raw_barcode = str(variant.get("barcode") or "").strip()
|
||||
barcode = validate_barcode(raw_barcode) if raw_barcode else None
|
||||
|
||||
rows.append({
|
||||
"title": title,
|
||||
"category": category,
|
||||
"size": size,
|
||||
"barcode": barcode,
|
||||
"price": _decimal_or_none(variant.get("price")),
|
||||
"image_url": images[0] if images else None,
|
||||
"image_urls": images,
|
||||
"vendor": str(product.get("vendor") or "").strip() or None,
|
||||
"source": "store",
|
||||
})
|
||||
if not rows:
|
||||
rows.append({
|
||||
"title": title, "category": category, "size": None, "barcode": None,
|
||||
"price": None, "image_url": images[0] if images else None,
|
||||
"image_urls": images,
|
||||
"vendor": str(product.get("vendor") or "").strip() or None,
|
||||
"source": "store",
|
||||
})
|
||||
return rows
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# WooCommerce Store API
|
||||
# ---------------------------------------------------------------------------
|
||||
def fetch_woocommerce(domain: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Raw WooCommerce Store API products, or None when unavailable.
|
||||
|
||||
udhaiyamdhall.com is WooCommerce but answers this endpoint with 403, which
|
||||
is why the sitemap tier below exists.
|
||||
"""
|
||||
out: List[Dict[str, Any]] = []
|
||||
for page in range(1, MAX_PAGES + 1):
|
||||
try:
|
||||
resp = _get(f"https://{domain}/wp-json/wc/store/products",
|
||||
{"per_page": WOO_PAGE_SIZE, "page": page})
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("Woo fetch failed for %s p%d: %s", domain, page, e)
|
||||
return out or None
|
||||
data = _json_or_none(resp)
|
||||
if not isinstance(data, list):
|
||||
return out or None
|
||||
if not data:
|
||||
break
|
||||
out.extend(data)
|
||||
if len(data) < WOO_PAGE_SIZE:
|
||||
break
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
return out or None
|
||||
|
||||
|
||||
def _woo_price(prices: Dict[str, Any]) -> Optional[float]:
|
||||
"""WooCommerce reports MINOR UNITS.
|
||||
|
||||
gopuramproducts.com returns `price: "5500"` with `currency_minor_unit: 2`,
|
||||
which is Rs55.00 and not Rs5,500. Read straight, every price on the site is
|
||||
a hundred times too large - and `product_validator.validate_price_range`
|
||||
would then reject the row as an implausible price, so the failure would
|
||||
surface as "this brand's products are all wrong" rather than as a units bug.
|
||||
"""
|
||||
raw = prices.get("price")
|
||||
if raw is None or str(raw).strip() == "":
|
||||
return None
|
||||
try:
|
||||
minor = int(prices.get("currency_minor_unit", 2))
|
||||
return float(raw) / (10 ** minor)
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
def _woo_candidates(product: Dict[str, Any]) -> List[Dict[str, Any]]:
|
||||
title = str(product.get("name") or "").strip()
|
||||
if not title:
|
||||
return []
|
||||
categories = [
|
||||
str(c.get("name")).strip() for c in (product.get("categories") or [])
|
||||
if isinstance(c, dict) and c.get("name")
|
||||
]
|
||||
images = [
|
||||
str(i.get("src")) for i in (product.get("images") or [])
|
||||
if isinstance(i, dict) and i.get("src")
|
||||
]
|
||||
sku = str(product.get("sku") or "").strip()
|
||||
return [{
|
||||
"title": title,
|
||||
"category": categories[0] if categories else None,
|
||||
"size": canonical_size(title),
|
||||
"barcode": validate_barcode(sku) if sku else None,
|
||||
"price": _woo_price(product.get("prices") or {}),
|
||||
"image_url": images[0] if images else None,
|
||||
"image_urls": images,
|
||||
"vendor": None,
|
||||
"source": "store",
|
||||
}]
|
||||
|
||||
|
||||
def _decimal_or_none(raw: Any) -> Optional[float]:
|
||||
if raw is None or str(raw).strip() == "":
|
||||
return None
|
||||
try:
|
||||
return float(raw)
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tier 3 - product sitemap, for sites that publish no catalogue API
|
||||
# ---------------------------------------------------------------------------
|
||||
# udhaiyamdhall.com is WooCommerce but answers the Store API with 403, and its
|
||||
# product pages carry only a BreadcrumbList in JSON-LD - no Product node, no
|
||||
# og: tags. What it does have is `wp-sitemap-posts-product-1.xml` listing 32
|
||||
# products, each page with an <h1> and a gallery image.
|
||||
#
|
||||
# So this tier yields NAMES AND IMAGES, AND NOTHING ELSE. No price, no pack
|
||||
# size, no barcode - because the site does not state them, and a catalogue
|
||||
# source that guesses is the exact problem this whole area exists to fix. The
|
||||
# names are real and the photographs are the brand's own, which solves
|
||||
# enumeration and solves the image problem; sizes have to come from somewhere
|
||||
# that actually knows them.
|
||||
_SITEMAP_CANDIDATES = ("/sitemap.xml", "/wp-sitemap.xml", "/sitemap_index.xml")
|
||||
MAX_SITEMAP_PAGES = 200
|
||||
|
||||
|
||||
def _sitemap_locs(xml: str) -> List[str]:
|
||||
return re.findall(r"<loc>\s*(.*?)\s*</loc>", xml or "", re.IGNORECASE | re.DOTALL)
|
||||
|
||||
|
||||
def fetch_sitemap(domain: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Product names and images from a site's product sitemap, or None."""
|
||||
try:
|
||||
from bs4 import BeautifulSoup
|
||||
except ImportError: # pragma: no cover - declared in requirements.txt
|
||||
logger.debug("Sitemap tier unavailable: beautifulsoup4 not installed")
|
||||
return None
|
||||
|
||||
product_urls: List[str] = []
|
||||
for path in _SITEMAP_CANDIDATES:
|
||||
try:
|
||||
resp = _get(f"https://{domain}{path}")
|
||||
except Exception: # noqa: BLE001
|
||||
continue
|
||||
if resp.status_code != 200 or "xml" not in (resp.headers.get("content-type") or ""):
|
||||
continue
|
||||
locs = _sitemap_locs(resp.text)
|
||||
# A sitemap index points at sub-sitemaps; take only the product one.
|
||||
for loc in locs:
|
||||
if "product" not in loc.lower() or "categor" in loc.lower():
|
||||
continue
|
||||
if loc.lower().endswith(".xml"):
|
||||
try:
|
||||
sub = _get(loc)
|
||||
except Exception: # noqa: BLE001
|
||||
continue
|
||||
if sub.status_code == 200:
|
||||
product_urls.extend(
|
||||
u for u in _sitemap_locs(sub.text) if "/product/" in u.lower()
|
||||
)
|
||||
elif "/product/" in loc.lower():
|
||||
product_urls.append(loc)
|
||||
if product_urls:
|
||||
break
|
||||
|
||||
product_urls = list(dict.fromkeys(product_urls))[:MAX_SITEMAP_PAGES]
|
||||
if not product_urls:
|
||||
return None
|
||||
|
||||
out: List[Dict[str, Any]] = []
|
||||
for url in product_urls:
|
||||
try:
|
||||
page = _get(url)
|
||||
except Exception: # noqa: BLE001 - one dead page cannot stop the sweep
|
||||
continue
|
||||
if page.status_code != 200:
|
||||
continue
|
||||
soup = BeautifulSoup(page.text, "lxml")
|
||||
heading = soup.find("h1")
|
||||
title = heading.get_text(strip=True) if heading else ""
|
||||
if not title:
|
||||
continue
|
||||
image = None
|
||||
node = soup.select_one(".woocommerce-product-gallery img, .wp-post-image, "
|
||||
"meta[property='og:image']")
|
||||
if node is not None:
|
||||
image = node.get("src") or node.get("data-src") or node.get("content")
|
||||
out.append({
|
||||
"title": title,
|
||||
"category": None,
|
||||
# Left None deliberately - this site states neither, and inventing
|
||||
# them is the failure mode, not the fallback.
|
||||
"size": None,
|
||||
"barcode": None,
|
||||
"price": None,
|
||||
"image_url": image,
|
||||
"image_urls": [image] if image else [],
|
||||
"vendor": None,
|
||||
"source": "store",
|
||||
})
|
||||
time.sleep(1.0)
|
||||
return out or None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Public entry point
|
||||
# ---------------------------------------------------------------------------
|
||||
def _cache_path(brand: str) -> Path:
|
||||
slug = re.sub(r"[^a-z0-9]+", "_", (brand or "").lower()).strip("_")
|
||||
return CACHE_DIR / f"{slug}.json"
|
||||
|
||||
|
||||
def _write_cache(path: Path, payload: Dict[str, Any]) -> None:
|
||||
"""Atomic write, the same .tmp + os.replace off_bulk uses - a half-written
|
||||
catalogue read by the next run is worse than no cache."""
|
||||
try:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = path.with_suffix(".tmp")
|
||||
tmp.write_text(json.dumps(payload, indent=2), encoding="utf-8")
|
||||
os.replace(tmp, path)
|
||||
except Exception as e: # noqa: BLE001 - caching is best effort
|
||||
logger.debug("Could not cache brand store catalogue at %s: %s", path, e)
|
||||
|
||||
|
||||
def read_cached(brand: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""The cached catalogue for a brand, or None. Never raises."""
|
||||
path = _cache_path(brand)
|
||||
if not path.exists():
|
||||
return None
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8-sig"))
|
||||
return list(data.get("products") or [])
|
||||
except Exception as e: # noqa: BLE001
|
||||
logger.debug("Unreadable brand store cache %s: %s", path, e)
|
||||
return None
|
||||
|
||||
|
||||
def fetch_store_catalogue(brand: str, domain: Optional[str], *,
|
||||
live: bool = False, refresh: bool = False,
|
||||
trusted_domain: bool = False,
|
||||
aliases: Optional[Sequence[str]] = None,
|
||||
) -> Optional[List[Dict[str, Any]]]:
|
||||
"""The brand's own catalogue as candidate dicts, or None if there isn't one.
|
||||
|
||||
`live` defaults to FALSE, matching `retail_presence.check_listing`: the
|
||||
runtime path reads the cache and the backfill script does the fetching, so
|
||||
a discovery preview never blocks on somebody's storefront being slow.
|
||||
|
||||
None means "no catalogue" - not reachable, not a supported platform, or
|
||||
rejected by the identity guard. An empty list is not returned; a store with
|
||||
zero products is indistinguishable from no store and is reported the same.
|
||||
"""
|
||||
if not refresh:
|
||||
cached = read_cached(brand)
|
||||
if cached is not None:
|
||||
return cached or None
|
||||
if not live or not domain:
|
||||
return None
|
||||
|
||||
raw = fetch_shopify(domain)
|
||||
platform = "shopify"
|
||||
to_candidates = _shopify_candidates
|
||||
if raw is None:
|
||||
raw = fetch_woocommerce(domain)
|
||||
platform = "woocommerce"
|
||||
to_candidates = _woo_candidates
|
||||
if raw is None:
|
||||
# Last tier: names and images off the product sitemap. See fetch_sitemap
|
||||
# for why it yields nothing else.
|
||||
raw = fetch_sitemap(domain)
|
||||
platform = "sitemap"
|
||||
to_candidates = lambda row: [row] # noqa: E731 - already candidates
|
||||
if raw is None:
|
||||
logger.info("No structured catalogue at %s for %s", domain, brand)
|
||||
return None
|
||||
|
||||
candidates: List[Dict[str, Any]] = []
|
||||
for product in raw:
|
||||
if isinstance(product, dict):
|
||||
candidates.extend(to_candidates(product))
|
||||
|
||||
vendors = sorted({c.get("vendor") for c in candidates if c.get("vendor")})
|
||||
# A CURATED DOMAIN IS ALREADY VERIFIED, BY A PERSON.
|
||||
#
|
||||
# The heuristic below cannot recognise every legitimate shape a brand's
|
||||
# domain takes - "theanilgroup.com" is Anil's, and neither starts with
|
||||
# "anil" nor contains it as a whole token - so applying it to a
|
||||
# hand-checked mapping would reject good catalogues for looking unusual.
|
||||
# It runs only on domains that came from automatic resolution, which is
|
||||
# the path that produced cityofnagacebu.gov.ph.
|
||||
if not trusted_domain and not _store_identity_ok(brand, domain, vendors, aliases):
|
||||
# REJECTED WHOLE, on purpose. This is the failure that would otherwise
|
||||
# put a hundred of somebody else's products under this brand's name,
|
||||
# each of them a real product and none of them theirs.
|
||||
logger.warning(
|
||||
"Rejecting catalogue at %s for brand %r: neither the domain nor "
|
||||
"its vendors (%s) identify this brand",
|
||||
domain, brand, ", ".join(vendors) or "none stated",
|
||||
)
|
||||
return None
|
||||
|
||||
for c in candidates:
|
||||
c.pop("vendor", None)
|
||||
|
||||
logger.info("%s: %d products from %s (%s)", brand, len(candidates), domain, platform)
|
||||
_write_cache(_cache_path(brand), {
|
||||
"brand": brand,
|
||||
"domain": domain,
|
||||
"platform": platform,
|
||||
"fetched_at": time.time(),
|
||||
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
|
||||
"products": candidates,
|
||||
})
|
||||
return candidates or None
|
||||
Reference in New Issue
Block a user