Brand valid image generation

This commit is contained in:
sriram
2026-09-11 15:47:56 +05:30
parent e1a5962f82
commit ce4fa70dee
31 changed files with 5586 additions and 1495 deletions

View File

@@ -17,14 +17,20 @@ now the only admin way in. What made it the survivor is the unit of work: five
files are one batch with one id, so the question a colleague actually asks -
"did the drop land?" - has one answer rather than five.
WHY THE NETWORK STAGES DEFAULT OFF HERE
---------------------------------------
Turning both on for a single file is a considered trade. At twenty files it is
thousands of outbound requests and, for the image stage, a Playwright subprocess
that can burn three minutes on its own - on a single-vCPU container that is also
serving the API. So a batch opts IN to those stages; it does not opt out.
`USE_OLLAMA` is false in production anyway, which makes `use_llm` a no-op there
and the honest default obvious.
WHY IMAGES DEFAULT OFF HERE AND THE LLM DEFAULTS ON
---------------------------------------------------
Image search for a single file is a considered trade. At twenty files it is
thousands of outbound requests and a Playwright subprocess that can burn three
minutes on its own - on a single-vCPU container that is also serving the API.
So a batch opts IN to that stage; it does not opt out.
`use_llm` is different and defaults ON: it gates only the description written
in stage 2 for rows the sheet left blank, which is the one field a shopper
reads and a store almost never supplies. It is cheap to leave on because
`ollama_service._ensure_client` caches its reachability probe per batch and
`store_catalog_pipeline.LlmBreaker` stops calling after three consecutive
misses, so with `USE_OLLAMA` false (production today) the cost is one warning
per file and the rows fall back to a factual template.
Note that these defaults bind THIS router only. The open upload endpoint runs
itself and takes its two flags from `UPLOAD_AUTORUN_FETCH_IMAGES` and
@@ -145,7 +151,7 @@ async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
dependencies=[Depends(require_admin)])
async def ingest_catalog_batch(
files: List[UploadFile] = File(...),
use_llm: bool = False,
use_llm: bool = True,
fetch_images: bool = False,
) -> BatchOut:
"""Stage the files, queue the batch, and return an id to poll.
@@ -325,7 +331,8 @@ class InboxStartRequest(InboxSelection):
# Chosen HERE, not by the sender - see the note on the POST handler in
# uploads.py. These commit the host to outbound work, so the decision
# belongs to the person who can see what the machine is already doing.
use_llm: bool = False
# The LLM is on by default for the reason in the module docstring.
use_llm: bool = True
fetch_images: bool = False
# Who runs it. "inprocess" is this container's worker thread and is the
# default, so an existing client that never sends the field is unaffected.

View File

@@ -124,11 +124,14 @@ def _normalize_header(raw: Any) -> str:
_EXACT_HEADERS: Dict[str, str] = {
# brand
# brand. "productbrand" is what a camel-cased ProductBrand header becomes
# after _normalize_header, and is listed so the mapping does not depend on
# the keyword rules' ordering.
"brand": "brand", "brand name": "brand", "brands": "brand",
"company": "brand", "manufacturer": "brand", "company name": "brand",
"product brand": "brand", "productbrand": "brand",
# product name
"product": "product_name", "product name": "product_name",
"product": "product_name", "product name": "product_name", "productname": "product_name",
"product name variant": "product_name", "product variant": "product_name",
"variant": "product_name", "name": "product_name", "item": "product_name",
"item name": "product_name", "product title": "product_name",
@@ -138,18 +141,27 @@ _EXACT_HEADERS: Dict[str, str] = {
# category
"category": "category", "categories": "category", "cat": "category",
"product category": "category", "segment": "category",
# price
# price. Two columns, two meanings: `selling_price` is what the shop
# charges - a retail / sale / selling price, or a bare "Price" - and
# `final_selling_price` is the tax-inclusive ceiling (MRP, "final price").
# A sheet that carries only one of them still fills both, because
# vector_store.upsert_brand_products copies selling -> final when final is
# blank. Before this split every price header landed in final_selling_price
# and selling_price was NULL for every uploaded row.
"price range": "price_range", "range": "price_range", "mrp range": "price_range",
"price": "final_selling_price", "final price": "final_selling_price",
"final price rs": "final_selling_price", "final selling price": "final_selling_price",
"selling price": "final_selling_price", "mrp": "final_selling_price",
"rate": "final_selling_price", "amount": "final_selling_price",
"cost": "final_selling_price", "unit price": "final_selling_price",
"selling price": "selling_price", "sellingprice": "selling_price",
"retail price": "selling_price", "retailprice": "selling_price",
"sale price": "selling_price", "sp": "selling_price",
"price": "selling_price", "unit price": "selling_price", "rate": "selling_price",
"final price": "final_selling_price", "final price rs": "final_selling_price",
"final selling price": "final_selling_price", "mrp": "final_selling_price",
"amount": "final_selling_price", "cost": "final_selling_price",
# identifiers
"barcode": "barcode", "barcode gtin ean": "barcode", "bar code": "barcode",
"gtin": "barcode", "ean": "barcode", "upc": "barcode", "ean13": "barcode",
"hsn": "hsn_code", "hsn code": "hsn_code", "hsn sac": "hsn_code",
"sku": "product_sku", "product sku": "product_sku", "sku code": "product_sku",
"sku": "product_sku", "product sku": "product_sku", "productsku": "product_sku",
"sku code": "product_sku",
"fssai": "fssai_license", "fssai license": "fssai_license",
"fssai license number": "fssai_license", "fssai number": "fssai_license",
# text / media
@@ -173,7 +185,13 @@ _KEYWORD_RULES: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
("barcode", ("barcode", "bar code", "gtin", "ean", "upc")),
("product_sku", ("sku",)),
("price_range", ("price range", "range")),
("final_selling_price", ("final price", "selling price", "price", "mrp", "rate", "cost")),
# Order is load-bearing: "final selling price" contains "selling price",
# and every price header contains "price", so the final-price forms must
# be claimed before the generic ones. "sp" is deliberately exact-only - as
# a substring it is inside "spice", "display" and "sponge".
("final_selling_price", ("final price", "final selling price", "mrp")),
("selling_price", ("selling price", "retail price", "sale price", "price", "rate")),
("final_selling_price", ("cost",)),
("image_url", ("image", "photo", "picture", "url", "link")),
("description", ("description", "desc", "detail")),
("category", ("category", "segment")),
@@ -346,6 +364,7 @@ def row_to_request(row: Dict[str, Any], mapping: _ColumnMapping) -> AddProductRe
product_sku=_text(row, mapping, "product_sku"),
hsn_code=_text(row, mapping, "hsn_code"),
final_selling_price=_number(row, mapping, "final_selling_price"),
selling_price=_number(row, mapping, "selling_price"),
barcode=_text(row, mapping, "barcode"),
image_url=_text(row, mapping, "image_url"),
)
@@ -528,8 +547,9 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
price_range = req.price_range
if not price_range:
if req.final_selling_price:
price_range = f"₹{req.final_selling_price}"
sheet_price = req.final_selling_price or req.selling_price
if sheet_price:
price_range = f"₹{sheet_price}"
elif sample_existing.get("price_range"):
price_range = sample_existing.get("price_range")
else:

View File

@@ -91,7 +91,8 @@ def generate_product_highlights(product: Dict[str, Any], brand: str) -> List[str
highlights.append(f"Available in {size} - {price}")
elif size:
highlights.append(f"Available in {size}")
elif isinstance(v, str) and v.strip():
elif isinstance(v, str) and v.strip() and v.strip().lower() != "standard":
# "Standard" is the pipeline's no-size sentinel, not a pack size.
highlights.append(f"Available in {v}")
# Quality indicators from description

View File

@@ -47,7 +47,7 @@ import asyncio
import logging
import re
from dataclasses import dataclass, field
from typing import Any, Callable, Dict, List, Optional, Tuple
from typing import Any, Callable, Dict, List, Optional, Sequence, Tuple
from app.infrastructure.settings import (
ENABLE_BARCODE_LOOKUP,
@@ -132,8 +132,16 @@ _ENRICHABLE = (
"selling_price", "image_url", "image_urls", "size_variants",
)
# The pack size a store sheet usually carries inside the product name itself -
# "India Gate Basmati Rice 1kg", "Fortune Sunflower Oil 1L", "Amul Milk 1
# Litre". The match is kept as written in the name (case and spacing), so the
# size a shopper sees on the card is the size the store typed. Longer unit
# spellings precede their prefixes because alternation is first-match-wins.
_SIZE_IN_TITLE = re.compile(
r"\b(\d+(?:[.,]\d+)?)\s*(kg|kgs|g|gm|gms|gram|grams|ml|l|ltr|litre|liters?|pack|pcs|n)\b",
r"\b(\d+(?:[.,]\d+)?)\s*"
r"(kgs|kg|gms|gm|grams|gram|mg|g|"
r"mls|ml|litres|litre|liters|liter|ltrs|ltr|lt|l|"
r"pieces|piece|pcs|pc|nos|pack|n)\b",
re.I,
)
@@ -276,24 +284,71 @@ def stage_1_brand_and_fssai(row: Dict[str, Any]) -> Dict[str, Any]:
# ---------------------------------------------------------------------------
# Stage 2 - Row intake (replaces AI discovery)
# ---------------------------------------------------------------------------
def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True) -> Dict[str, Any]:
@dataclass
class LlmBreaker:
"""Stop asking the LLM once it has stopped answering.
`ollama_service._generate` swallows every exception and returns "", so a
slow or dead Ollama never raises - it just costs up to
OLLAMA_TIMEOUT_SECONDS x retries per row and yields nothing. On a
2000-row sheet that is days. Three misses in a row and the rest of the
batch goes straight to the factual fallback description; one success
resets the count, so a single hiccup does not switch the LLM off.
"""
limit: int = 3
consecutive_misses: int = 0
tripped: bool = False
def record(self, ok: bool) -> None:
if ok:
self.consecutive_misses = 0
return
self.consecutive_misses += 1
if self.consecutive_misses >= self.limit:
self.tripped = True
# The generated description is stored in a TEXT column, but a 1.5b model does
# not reliably honour the length it is asked for and the catalog card shows a
# short paragraph, not an essay.
_MAX_LLM_DESCRIPTION_CHARS = 300
def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True,
breaker: Optional[LlmBreaker] = None) -> Dict[str, Any]:
"""The spreadsheet row *is* the product. Only fill a missing description.
The LLM call is best-effort and optional: with Ollama unreachable the row
keeps an empty description and later stages still work.
keeps an empty description here and `_to_storage_row` supplies a factual
template at storage time, so later stages still work.
DESCRIPTION ONLY. This stage used to copy the LLM's `size_variants` into a
row that had none, which is how "Naga Maida" acquired a 500g/1kg/5kg it
was never sold in. A pack size is a fact about the physical product and
comes from the sheet or the product name (stage 4), never from a guess.
"""
if not _blank(row.get("description")) or not use_llm:
return row
if breaker is not None and breaker.tripped:
return row
ok = False
try:
from app.services.ollama_service import fetch_product_details
details = fetch_product_details(row.get("brand", ""), row.get("product_name", ""))
sheet_sizes = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
title_size = _SIZE_IN_TITLE.search(row.get("product_name") or "")
size = sheet_sizes[0] if sheet_sizes else (title_size.group(0).strip() if title_size else None)
details = fetch_product_details(
row.get("brand", ""), row.get("product_name", ""),
category=row.get("category") or None, size=size,
)
if details and details.get("description"):
row["description"] = str(details["description"]).strip()
if _blank(row.get("size_variants")) and details and details.get("size_variants"):
row["size_variants"] = list(details["size_variants"])
row["description"] = str(details["description"]).strip()[:_MAX_LLM_DESCRIPTION_CHARS]
ok = True
except Exception as exc: # noqa: BLE001 - enrichment is best-effort
logger.debug("LLM enrichment skipped for %r: %s", row.get("product_name"), exc)
if breaker is not None:
breaker.record(ok)
return row
@@ -442,30 +497,32 @@ def _sizes_for(row: Dict[str, Any]) -> List[str]:
if match:
return [match.group(0).strip()]
# A COMMODITY WITH NO WEIGHT COLUMN GETS ONE UNSIZED ROW, NOT THREE MADE-UP
# ONES. default_size_variants() invents a plausible set - 100g/250g/500g -
# and for packaged goods that is a reasonable guess at what a brand sells.
# For loose produce it is not: a shop sells apples by the kilo at whatever
# the customer asks for, so "Apple 250g" is a product that does not exist.
# NO SIZE ANYWHERE MEANS ONE UNSIZED ROW, NOT THREE MADE-UP ONES - for
# every brand. This used to fall through to default_size_variants() for
# branded rows, which invents a plausible set (1kg/5kg/25kg for rice) and
# so turned "India Gate Basmati Rice" into three products that the sheet
# never listed and the shop may never have stocked. It was also the
# mechanism behind the catalogue drift the integrator reported: the
# invented set is keyed on the resolved category, so the same product
# ingested twice with the category resolved differently produced two
# disjoint size sets, two sets of image_ids, and a re-scrape that looked
# like the old rows were deleted. A pack size is recorded only when the
# sheet or the product name states it.
#
# It is also the mechanism behind the catalogue drift the integrator
# reported: the invented set is keyed on the resolved category, so the same
# product ingested twice with the category resolved differently produces
# two disjoint size sets, two sets of image_ids, and a re-scrape that looks
# like the old rows were deleted. Keeping produce out of that from the
# start is cheaper than repairing it later.
# "Standard" and not "": an empty size fails validate_size ("size/pack is
# missing") and stage 4 would drop the row, which is the rejection this
# whole area exists to prevent. "Standard" is also what _to_storage_row
# already substitutes for a blank size and what the seeded base list uses,
# so an uploaded "Apple" deduplicates onto the seeded "Apple" instead of
# creating a second row.
if row.get("brand") == OWN_PRODUCTS_BRAND:
return ["Standard"]
return list(price_estimator.default_size_variants(
row.get("category") or "", row.get("product_name") or ""
))
# missing") and stage 10 would penalise the row for a fact nobody had. It is
# the pipeline's internal no-size sentinel, understood by the validator,
# the price estimator and the SKU resolver. What reaches the table depends
# on the brand - see _to_storage_row: Own Products keeps "Standard" so an
# uploaded "Apple" deduplicates onto the seeded "Apple" (same image_id);
# a branded row stores an empty size_variants and an unsuffixed name.
if row.get("brand") != OWN_PRODUCTS_BRAND:
# Worth a line in the job report for a packaged good; for loose
# produce an absent size is the normal case and would only be noise.
row.setdefault("_notes", []).append(
"no pack size in the sheet or the product name; stored without one"
)
return ["Standard"]
def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], List[str]]:
@@ -519,8 +576,14 @@ def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
if not _blank(row.get("price_range")):
return row
if not _blank(row.get("final_selling_price")):
price = float(row["final_selling_price"])
# Either price the sheet gave is a real price for this pack; the final
# (tax-inclusive) one is preferred when both are present because the band
# is what the card shows a shopper.
sheet_price = row.get("final_selling_price")
if _blank(sheet_price):
sheet_price = row.get("selling_price")
if not _blank(sheet_price):
price = float(sheet_price)
lo = int(round(price * (1 - _SHEET_PRICE_BAND)))
hi = int(round(price * (1 + _SHEET_PRICE_BAND)))
elif row.get("brand") == OWN_PRODUCTS_BRAND:
@@ -537,7 +600,84 @@ def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
# ---------------------------------------------------------------------------
# Stage 6 - Images
# ---------------------------------------------------------------------------
def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, Any]:
class ImageSearchCache:
"""One image search per PRODUCT, shared by every pack size of it.
Stage 4 turns one source row into one row per size, and this stage used to
search once per resulting row. The query was the same each time - the size
is not in it - but the providers behind `find_all_image_urls` are live and
not deterministic, so "Dabur Honey" asked four times came back four ways:
the 225g row drew an Amazon photo of Dabur Honey, the 1kg row drew nothing
usable. Same product, four different images, three of them wrong or blank.
Keyed on `image_corroboration.search_key`, which strips a trailing pack
size from the name, so sheet rows that carry the size inside the name
("India Gate Basmati Rice 1kg" / "... 5kg") share as well. The cached value
is the RANKED candidate list; each size still runs its own gate, which
ignores size tokens and so reaches the same verdict for every sibling.
`existing` maps a search key to an already-stored eligible primary for
the same product under the same brand, so a re-ingest that finds nothing
new can inherit the photo a sibling size already has - after the gate.
"""
def __init__(self, existing_rows: Optional[Sequence[Dict[str, Any]]] = None) -> None:
self.candidates: Dict[tuple, List[str]] = {}
self.existing: Dict[tuple, str] = {}
for prior in existing_rows or ():
url = prior.get("image_url")
if not url or not str(url).startswith("http"):
continue
key = image_corroboration.search_key(prior.get("product_name") or "",
prior.get("brand") or "")
self.existing.setdefault(key, str(url))
def _adopt_sibling_image(row: Dict[str, Any], brand: str, cache: ImageSearchCache) -> bool:
"""Reuse the eligible primary of another pack size of the same product.
Only ever through the gate: the sibling's URL is corroborated against THIS
row's title before it is adopted, so a wrong image stored under one size
cannot propagate to the rest. A sibling is the same product in a different
pack, which is what `search_key` means, and the same photo on every pack is
what a retailer shows too.
"""
key = image_corroboration.search_key(row.get("product_name") or "", brand)
url = cache.existing.get(key)
if not url:
return False
choice = image_corroboration.choose_primary([url], row.get("product_name") or "", brand)
if choice.primary is None:
return False
row["image_url"] = choice.primary
row["image_urls"] = [choice.primary]
row.setdefault("_notes", []).append("image adopted from another pack size of this product")
return True
def _image_cache_for(rows: Sequence[Dict[str, Any]]) -> ImageSearchCache:
"""The run's shared image cache, primed with what each brand already holds.
Reads each destination brand once (stage 11 reads it again for the merge;
two reads of a small table beat threading the rows through five stages).
A read failure primes nothing - the search still runs.
"""
existing: List[Dict[str, Any]] = []
for brand in sorted({r.get("brand") for r in rows if r.get("brand")}):
if brand == OWN_PRODUCTS_BRAND:
continue
try:
for prior in get_products_by_brand(brand) or []:
prior = dict(prior)
prior.setdefault("brand", brand)
existing.append(prior)
except Exception as exc: # noqa: BLE001 - priming is an optimisation
logger.warning("Could not read existing images for %s: %s", brand, exc)
return ImageSearchCache(existing)
def stage_6_images(row: Dict[str, Any], *, enabled: bool = True,
cache: Optional[ImageSearchCache] = None) -> Dict[str, Any]:
if not _blank(row.get("image_urls")) or not enabled:
return row
try:
@@ -580,33 +720,61 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
# drops the "-plant -tree -fish -botanical" Wikimedia exclusion block
# that fights every produce query. Without it this stage recreates the
# exact defect the Own Products image repair exists to undo.
candidates = find_all_image_urls(
product_name, brand=brand or None, max_results=24, produce=is_produce
)
if candidates:
best = catalog_engine._select_best_images(
candidates, product_name, brand, max_images=10
# A pack size of this product that is already in the table may carry
# the photo. Take it - through the gate - before spending a search:
# the same product in a different pack is the same photo, every size
# then agrees, and a re-ingest costs no network. See _adopt_sibling_image.
if cache is not None and not is_produce and _adopt_sibling_image(row, brand, cache):
return row
# One search per product, not per pack size - see ImageSearchCache.
# The query is the size-less name: a photo's filename or alt text
# almost never carries the pack size, and a size token in the query
# only steers the engines towards pages LISTING that size, which for
# "Honey 1kg" was every other brand's 1kg honey.
key = image_corroboration.search_key(product_name, brand)
base_name = image_corroboration.SIZE_TAIL.sub("", product_name).strip() or product_name
if cache is not None and key in cache.candidates:
best = cache.candidates[key]
else:
candidates = find_all_image_urls(
base_name, brand=brand or None, max_results=24, produce=is_produce
)
if best:
# `_select_best_images` ORDERS candidates; it does not judge
# whether any of them is this product. Its scoring awards points
# for the brand appearing in the URL, so for a brand that is
# also a personal name it actively rewarded the wrong photo -
# `Anil_Kapoor_2019.jpg` outscored everything and became
# `image_url`. choose_primary decides what may be promoted.
#
# The list is still stored in full. A product whose only
# candidates are uncorroborated keeps them for review; it just
# does not get one of them presented as fact.
choice = image_corroboration.choose_primary(
best, product_name, brand or ""
best = catalog_engine._select_best_images(
candidates, base_name, brand, max_images=10
) if candidates else []
if cache is not None:
cache.candidates[key] = list(best)
if best:
# `_select_best_images` ORDERS candidates; it does not judge
# whether any of them is this product. Its scoring awards points
# for the brand appearing in the URL, so for a brand that is
# also a personal name it actively rewarded the wrong photo -
# `Anil_Kapoor_2019.jpg` outscored everything and became
# `image_url`. choose_primary decides what may be promoted.
#
# ONLY ELIGIBLE CANDIDATES ARE STORED. The rejects used to be kept
# in `image_urls` "for review" - but the product card falls back to
# `image_urls[0]` when `image_url` is empty and the modal shows the
# list as a gallery, so the review pile was what the shopper saw:
# four foreign honeys for "Dabur Honey 1kg", a mosquito repellent
# for "Dabur Honey 500g". A reject is logged, not displayed.
choice = image_corroboration.choose_primary(
best, product_name, brand or ""
)
row["image_urls"] = list(choice.eligible)
row["image_url"] = choice.primary
if choice.rejected:
row.setdefault("_notes", []).append(
f"{len(choice.rejected)} image candidate(s) rejected: "
"they do not name this product"
)
row["image_urls"] = list(choice.ordered)
row["image_url"] = choice.primary
if choice.primary is None:
row.setdefault("_notes", []).append(
f"no primary image: {choice.reason}"
)
if choice.primary is None:
row.setdefault("_notes", []).append(
f"no primary image: {choice.reason}"
)
except Exception as exc: # noqa: BLE001 - an image is not worth the row
# `warning`, not `debug`: at the default log level a debug line is
# invisible, so a row that silently lost its images looked identical to
@@ -731,18 +899,50 @@ def build_image_id(brand: str, product_name: str, size: str) -> str:
return f"{_sanitize_name(brand)}_{slug}"
def _fallback_description(brand: str, name: str, size: str, category: str) -> str:
"""What the card says when neither the sheet nor the LLM said anything.
Factual by construction - only facts the row already holds: the product,
its pack size, its category, its maker. Deliberately NOT
`catalog_engine.generate_detailed_description`, which pads any product
into eight hundred characters of "premium quality" and "exceptional
taste" that nobody verified. An honest short line beats a confident essay.
"""
parts = [name.strip()]
if size and size.lower() not in name.lower():
parts.append(f"{size} pack")
if category and category.lower() not in ("general", "uncategorized"):
parts.append(category)
if brand:
# resolve_parent_brand hands back the alias KEY ("amul"), which is a
# lookup token rather than a name. Title-case only that form; a brand
# written with its own casing ("India Gate", "P&G") is left alone.
parts.append(f"by {brand.title() if brand.islower() else brand}")
return ", ".join(p for p in parts if p) + "."
def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"""Project an enriched row onto the brand-table columns."""
name = row.get("product_name") or row.get("title") or ""
size = row.get("size") or "Standard"
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
category = row.get("category") or "General"
own = row.get("brand") == OWN_PRODUCTS_BRAND
# "Standard" is the pipeline's internal no-size sentinel (see _sizes_for).
# Own Products keeps it in the table: the seeded base list uses it and an
# uploaded "Apple" has to build the same image_id as the seeded "Apple".
# A branded row has no such counterpart, so the sentinel is dropped here
# and the row is stored the way the sheet described it - "India Gate
# Basmati Rice", size_variants [], image_id without a size suffix - rather
# than as "India Gate Basmati Rice Standard".
raw_size = row.get("size") or "Standard"
size = raw_size if (own or raw_size != "Standard") else ""
display = name if (not size or size.lower() in name.lower()) else f"{name} {size}".strip()
category = row.get("category") or "General"
# "Apple 500g from Own Products." is a sentence nobody wrote and nobody
# wants, and "Own Products" is a bucket rather than a maker, so the
# template reads as a false provenance claim. A commodity keeps whatever
# description the sheet gave, including none.
description = row.get("description") or (None if own else f"{display} from {row.get('brand')}.")
description = row.get("description") or (
None if own else _fallback_description(row.get("brand") or "", display, size, category)
)
# The embedding text. Interpolating a null description and the bucket name
# would embed the literal "Own Products ... None", so a commodity is
# described to the vector index by what actually identifies it.
@@ -759,7 +959,7 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"image_url": row.get("image_url"),
"image_urls": list(row.get("image_urls") or []),
"price_range": row.get("price_range"),
"size_variants": [size],
"size_variants": [size] if size else [],
"providers": list(row.get("providers") or []),
"fssai_license": row.get("fssai_license"),
"product_sku": row.get("product_sku"),
@@ -1001,9 +1201,16 @@ def run_pipeline(
progress(1, STAGE_NAMES[0], 0, len(prepared))
prepared = [stage_1_brand_and_fssai(r) for r in prepared]
breaker = LlmBreaker() if use_llm else None
for index, row in enumerate(prepared, start=1):
stage_2_row_intake(row, use_llm=use_llm)
stage_2_row_intake(row, use_llm=use_llm, breaker=breaker)
progress(2, STAGE_NAMES[1], index, len(prepared))
if breaker is not None and breaker.tripped:
result.warnings.append(
f"LLM descriptions stopped after {breaker.limit} consecutive failures "
"(Ollama unreachable or too slow); the remaining rows use the factual "
"template description."
)
progress(3, STAGE_NAMES[2], 0, len(prepared))
prepared = [stage_3_title_category(r) for r in prepared]
@@ -1031,8 +1238,9 @@ def run_pipeline(
for index, row in enumerate(exploded, start=1):
stage_5_pricing(row)
progress(5, STAGE_NAMES[4], index, len(exploded))
image_cache = _image_cache_for(exploded) if fetch_images else None
for index, row in enumerate(exploded, start=1):
stage_6_images(row, enabled=fetch_images)
stage_6_images(row, enabled=fetch_images, cache=image_cache)
progress(6, STAGE_NAMES[5], index, len(exploded))
for index, row in enumerate(exploded, start=1):
stage_7_sku(row)

View File

@@ -107,6 +107,7 @@ from app.infrastructure.settings import (
)
from app.services import active_brands, brand_store, ollama_service, retail_presence
from app.services.brand_registry import (
get_brand_store_domain,
get_fssai_license,
get_known_sub_brands,
resolve_parent_brand,
@@ -449,19 +450,29 @@ def _existing_title_index(brand: str) -> Dict[str, str]:
# ---------------------------------------------------------------------------
# Source A - Open Food Facts (ground truth)
# ---------------------------------------------------------------------------
class OpenFactsUnavailable(RuntimeError):
"""Open Food Facts could not be consulted - as opposed to consulted and
found empty. `discover_brand_products` turns this into a warning that says
so, because "OFF has nothing for this brand" is a claim about the brand
and "OFF was down" is not."""
def _from_open_facts(brand: str, *, refresh: bool = False) -> List[Dict[str, Any]]:
"""Real products for `brand`, with a real GTIN and often a real pack size.
`off_bulk.fetch_brand_corpus` is already disk-cached, retry-wrapped and
paced at 2s between pages, and `scripts/backfill_barcodes_from_off.py`
already imports it from outside the enrichment package, so reaching for it
here is an established pattern rather than a new one.
`off_bulk.fetch_brand_corpus_result` is disk-cached, retry-wrapped and
paced at 6s between 100-row pages. Raises `OpenFactsUnavailable` when the fetch
did not complete and produced nothing; anything else is returned as-is,
an empty list meaning OFF genuinely has no products under this brand tag.
"""
try:
hits = off_bulk.fetch_brand_corpus(brand, refresh=refresh)
result = off_bulk.fetch_brand_corpus_result(brand, refresh=refresh)
except Exception as exc: # noqa: BLE001 - a dead OFF must not kill discovery
logger.warning("Open Food Facts lookup failed for %r: %s", brand, exc)
return []
raise OpenFactsUnavailable(str(exc) or exc.__class__.__name__) from exc
if result.error and not result.hits:
raise OpenFactsUnavailable(result.error)
hits = result.hits
out: List[Dict[str, Any]] = []
for hit in hits or ():
@@ -759,10 +770,10 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str,
which is what makes `image_id` reproducible - the name is half of it.
The LLM's list is a guess and comes last. Only when all three are empty is
the cell left blank, which hands the decision to stage 4's
`default_size_variants` - and that invents 100g/250g/500g, which
`_sizes_for`'s own comment identifies as the mechanism behind observed
catalogue drift. Leaving it blank is the last resort, not the default.
the cell left blank; stage 4 then stores ONE unsized row (it no longer
invents 100g/250g/500g - see `_sizes_for`, which names that invention as
the mechanism behind observed catalogue drift). Blank is honest, but a
real size is what makes the row a distinct pack, so it is worth finding.
"""
in_title = _TITLE_SIZE_RE.search(title or "")
if in_title:
@@ -850,25 +861,21 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
# semantic search.
#
# A blank description is honest and already handled: `_to_storage_row`
# substitutes "<name> <size> from <brand>." exactly as it does for a store
# sheet that left the column empty.
# substitutes a short factual line (name, size, category, brand) exactly
# as it does for a store sheet that left the column empty.
description = (candidate.get("description") or "").strip()
# A BARCODE IDENTIFIES ONE PACK, SO IT ONLY SURVIVES A KNOWN PACK SIZE.
# With no real size, stage 4 falls back to `default_size_variants` and
# invents 100g/250g/500g - and because every column is copied into each
# exploded variant, one real GTIN would be stamped onto three packs, two of
# which do not exist. That is a worse error than a blank barcode: it is
# wrong data that looks authoritative, and `nutrition_by_barcode` would
# happily resolve all three to the same product.
# A BARCODE IDENTIFIES ONE PACK. That used to mean dropping it whenever no
# pack size was known, because stage 4 then invented 100g/250g/500g and
# every column is copied into each exploded variant - one real GTIN
# stamped onto three packs, two of which did not exist. Stage 4 now stores
# a single unsized row when it has no size, and one GTIN on one row is
# exactly what a GTIN means, so the barcode survives; the row is flagged
# so a reviewer knows the pack size is still unknown.
barcode = candidate.get("barcode")
notes: List[str] = []
if barcode and not sizes:
notes.append(
"barcode dropped: no real pack size is known, and stage 4 will "
"invent several - a GTIN names one pack, not three"
)
barcode = None
notes.append("pack size unknown: the barcode names one pack, but its size was not found")
shape = {"title": title, "category": category_hint, "size_variants": sizes,
"description": description}
@@ -929,12 +936,25 @@ def discover_brand_products(
registry_terms = _registry_terms(brand)
fssai = get_fssai_license(brand)
off_candidates = _from_open_facts(brand, refresh=refresh_corpus) if use_openfacts else []
if use_openfacts and not off_candidates:
warnings.append(
"Open Food Facts returned nothing for this brand. Every row below "
"rests on the language model alone - review them individually."
)
# Three distinct outcomes, three distinct warnings. "OFF has nothing under
# this brand tag" is a fact about the brand; "OFF could not be reached" is
# a fact about the network and must not be reported as the former - that
# is how Naga (2 real rows on OFF) was being shown as unknown to it.
off_candidates: List[Dict[str, Any]] = []
if use_openfacts:
try:
off_candidates = _from_open_facts(brand, refresh=refresh_corpus)
except OpenFactsUnavailable as exc:
warnings.append(
f"Open Food Facts could not be reached ({exc}). Nothing was "
"cached, so the next preview will try again; the rows below "
"come from the brand's shop and the language model only."
)
else:
if not off_candidates:
warnings.append(
"Open Food Facts has no products tagged with this brand."
)
off_keys = {}
for candidate in off_candidates:
key = _normalise_title(brand, candidate["title"])
@@ -951,12 +971,26 @@ def discover_brand_products(
if store_candidates:
logger.info("%s: %d products from the brand's own shop", brand, len(store_candidates))
elif use_store and not off_candidates:
# Say what was actually checked. An unregistered storefront is not a
# storefront that "has nothing" - nobody has looked.
store_domain = get_brand_store_domain(brand)
if store_domain:
warnings.append(
f"{store_domain} is registered as this brand's shop but its "
"catalogue has not been fetched - run "
f"scripts/backfill_brand_stores.py --brand \"{brand}\" to load it."
)
else:
warnings.append(
"No verified storefront is registered for this brand. If it "
"has its own online shop, add it to "
"brand_registry.BRAND_STORE_DOMAINS and run "
"scripts/backfill_brand_stores.py to populate a real catalogue."
)
if not off_candidates and not store_candidates:
warnings.append(
"Neither Open Food Facts nor a brand storefront has anything for "
"this brand. Every row below rests on the language model alone - "
"review them individually. If this brand has its own online shop, "
"adding it to brand_registry.BRAND_STORE_DOMAINS and running "
"scripts/backfill_brand_stores.py will populate a real catalogue."
"Every row below rests on the language model alone - review them "
"individually."
)

View File

@@ -11,9 +11,12 @@ never heard of there is nothing to verify, and the catalogue is whatever
`qwen2.5:1.5b` invents.
Measured 2026-09-10, Open Food Facts hits: Udhaiyam 0, Tenali Double Horse 0,
Gopuram 0, Double Horse 14, and Naga 2 - where both "Naga" rows are a UK /
Bangladeshi pickle brand, not the Tamil Nadu one. Regional South Indian brands
are simply not in that database.
Gopuram 0, Double Horse 14. (An earlier note here said Naga's two rows were a
UK / Bangladeshi pickle brand - that was the old free-text `brands:Naga`
Search-a-licious query matching "Mr Naga" and "Bombay Naga Jhal". Under the
exact `brands_tags=naga` filter the v2 API returns two real Naga Limited rows,
Sooji 500 g and Maida 500 g, both on the 890 GS1 prefix - see off_bulk.py.)
Regional South Indian brands are still thin in that database.
They are, however, on their own shop, and modern storefronts publish their
whole catalogue as structured JSON:

View File

@@ -10,34 +10,50 @@ resulting cascade "misses far more often than it hits". That module is wired
into the live enrichment pipeline and is deliberately NOT touched by this file.
This module inverts the problem: fetch a brand's **entire** OFF catalogue in one
or two requests from the Search-a-licious endpoint, cache it on disk, then match
every one of our products against that corpus offline. A whole-catalogue
backfill costs ~5 HTTP requests instead of ~600, and re-tuning the similarity
threshold costs zero network because the corpus is cached.
or two requests from the v2 search API, cache it on disk, then match every one
of our products against that corpus offline. A whole-catalogue backfill costs
~5 HTTP requests instead of ~600, and re-tuning the similarity threshold costs
zero network because the corpus is cached.
Nothing here is imported by the running app - the only consumer is
`scripts/backfill_barcodes_from_off.py`. Everything except `fetch_brand_corpus`
is a pure function so it can be unit-tested without network or database.
`fetch_brand_corpus` is the one network function here and is shared by the
backfill scripts, `product_grounding`, `post_ingest_barcodes` and
`brand_discovery`. Everything else is a pure function so it can be unit-tested
without network or database.
ENDPOINT NOTES (verified empirically, 2026-09)
GET https://search.openfoodfacts.org/search
?q=brands:amul AND countries_tags:"en:india"
ENDPOINT NOTES (verified empirically, 2026-09-11)
GET https://world.openfoodfacts.org/api/v2/search
?brands_tags=amul
&countries_tags=en:india
&fields=code,product_name,quantity,...
&page_size=250
* The double quotes around "en:india" are REQUIRED. Without them the query
parses as a bare term and silently returns count=0 rather than erroring.
* page_size up to 1000 is accepted; 250 keeps responses small.
&page_size=100&page=1
* `brands_tags` is an EXACT match on the brand's tag slug ("naga", not
"Naga"), which is what a brand catalogue needs. The earlier Search-a-licious
query `q=brands:Naga` was a free-text match: it returned "Mr Naga" and
"Bombay Naga Jhal" - other companies - and MISSED the real Naga rows,
because Search-a-licious reads a separate Elasticsearch index that was
stale for them (barcode 8906011830068 was still filed brandless under
Kuwait). The v2 API reads the live product database. Measured on the same
day: Amul 216 products here vs 140 there; Naga 2 vs 0.
* `page_count` in this API is the number of products ON THIS PAGE, not the
number of pages, and `page_size` in the RESPONSE is what the server
actually applied (a larger request is capped to 100 without complaint).
Paginate from `count` / that served `page_size`.
* OFF rate-limits all search endpoints to 10 requests/minute per IP.
PAUSE_SECONDS keeps a multi-page brand under that.
* `world.openfoodfacts.org` intermittently serves an HTML "Page temporarily
unavailable" page with a 200 status, so every response is content-type
checked before parsing.
unavailable" page with a 200 status, and plain 503s under load, so every
response is status- and content-type-checked before parsing, and a failed
fetch is never written to the cache as an empty corpus.
"""
from __future__ import annotations
import json
import logging
import math
import os
import re
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
@@ -63,21 +79,50 @@ from app.services.quantity_utils import quantities_match
logger = logging.getLogger(__name__)
SEARCH_URL = "https://search.openfoodfacts.org/search"
SEARCH_URL = "https://world.openfoodfacts.org/api/v2/search"
# OFF's usage policy requires a contactable custom User-Agent; requests sent
# with the default python-requests agent are treated as anonymous crawling.
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
FIELDS = "code,product_name,product_name_en,brands,quantity,countries_tags"
PAGE_SIZE = 250
MAX_PAGES = 10 # 2500 products per brand is far beyond any real brand
PAUSE_SECONDS = 2.0 # search endpoints are the strictly-limited ones
PAGE_SIZE = 100 # the v2 API caps this at 100 and silently serves that
MAX_PAGES = 25 # 2500 products per brand is far beyond any real brand
PAUSE_SECONDS = 6.0 # 10 search requests/minute is OFF's published limit
# Bumped when the cache file's meaning changes. Files written before this key
# existed came from the Search-a-licious endpoint; an EMPTY one of those is
# more likely to be that endpoint's stale index (or a swallowed 503) than a
# real absence, so it is re-fetched. A non-empty one is real data and is kept.
CACHE_SCHEMA = 2
# backend/app/services/enrichment/barcode/sources/off_bulk.py -> backend/
_BACKEND_DIR = Path(__file__).resolve().parents[5]
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "off_brand_corpus"
class OffUnavailable(requests.exceptions.ConnectionError):
"""OFF answered but could not serve (429 / 5xx).
Subclasses ConnectionError so `retry.with_retry` treats it exactly like a
dropped socket - a few seconds later the same request usually succeeds -
without widening the retry set for every other barcode source.
"""
@dataclass
class CorpusFetch:
"""What `fetch_brand_corpus_result` learned.
`error` is None when OFF answered every page, even if it answered with
nothing - that is a real "this brand is not on Open Food Facts". When
`error` is set the fetch did not complete and NOTHING was cached, so a
caller can say "could not be reached" instead of "has nothing".
"""
hits: List[Dict[str, Any]] = field(default_factory=list)
error: Optional[str] = None
from_cache: bool = False
# Pack sizes embedded in a product name ("Marie Gold 250g", "Butter 1L").
# Mirrors the unit list in scripts/backfill_nutrition_from_barcodes.py, widened
# with the count-based units this catalog also uses.
@@ -122,84 +167,149 @@ def _cache_path(slug: str, cache_dir: Optional[Path] = None) -> Path:
return (cache_dir or CACHE_DIR) / f"{slug}.json"
def brand_tag_slug(brand: str) -> str:
"""The brand as Open Food Facts tags it: lowercase, runs of anything that
is not a letter or digit collapsed to one hyphen. "Hindustan Unilever" ->
"hindustan-unilever", "P&G" -> "p-g", "Naga" -> "naga". OFF matches
`brands_tags` case-insensitively, but sending the slug form is what its
own site does and avoids depending on that."""
return re.sub(r"[^a-z0-9]+", "-", (brand or "").lower()).strip("-")
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
def _get_page(brand: str, country: Optional[str], page: int) -> Dict[str, Any]:
"""One page of OFF search results. Raises on transport errors (retried by
the decorator); returns an empty result dict for any response that is not
parseable JSON, which is how the HTML "temporarily unavailable" page and
any future error page are absorbed without killing the run."""
query = f"brands:{brand}"
def _get_page(brand: str, country: Optional[str], page: int) -> Optional[Dict[str, Any]]:
"""One page of OFF v2 search results.
Raises on transport errors and on 429 / 5xx (both retried by the
decorator); returns None for any other response that is not parseable
JSON - an HTML "temporarily unavailable" page, a 4xx, a truncated body.
None means "this page FAILED", which the caller must keep distinct from a
page that parsed fine and simply held no products.
"""
params: Dict[str, Any] = {
"brands_tags": brand_tag_slug(brand),
"fields": FIELDS,
"page_size": PAGE_SIZE,
"page": page,
}
if country:
# The quotes are load-bearing - see the module docstring.
query += f' AND countries_tags:"en:{country}"'
params["countries_tags"] = f"en:{country}"
resp = requests.get(
SEARCH_URL,
params={"q": query, "fields": FIELDS, "page_size": PAGE_SIZE, "page": page},
params=params,
headers={"User-Agent": USER_AGENT, "Accept": "application/json"},
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
if resp.status_code == 429 or resp.status_code >= 500:
raise OffUnavailable(f"HTTP {resp.status_code} from Open Food Facts")
if resp.status_code != 200:
logger.warning("OFF search returned HTTP %s for brand %r page %s",
resp.status_code, brand, page)
return {}
return None
if "json" not in (resp.headers.get("content-type") or "").lower():
logger.warning("OFF search returned non-JSON (%s) for brand %r - "
"the service is probably serving an error page",
resp.headers.get("content-type"), brand)
return {}
return None
try:
payload = resp.json()
except ValueError as e:
logger.warning("OFF search returned unparseable JSON for brand %r: %s", brand, e)
return {}
return payload if isinstance(payload, dict) else {}
return None
return payload if isinstance(payload, dict) else None
def fetch_brand_corpus(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
"""Every OFF product for `brand`, from disk cache unless `refresh`.
def _read_cache(path: Path, brand: str) -> Optional[List[Dict[str, Any]]]:
"""Cached hits, or None when the cache must not be trusted: unreadable, or
an empty file from before CACHE_SCHEMA existed (see that constant)."""
try:
cached = json.loads(path.read_text(encoding="utf-8"))
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
return None
hits = cached.get("hits") or []
if not hits and cached.get("schema") != CACHE_SCHEMA:
logger.info(" Ignoring empty pre-v%s OFF cache for %r - re-fetching",
CACHE_SCHEMA, brand)
return None
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
brand, len(hits), cached.get("fetched_at_human", "?"))
return hits
def fetch_brand_corpus_result(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> CorpusFetch:
"""Every OFF product for `brand`, from disk cache unless `refresh`, plus
whether the fetch actually completed.
Hits with no usable product name are dropped here rather than at match time
(6 of 146 Amul hits, 1 of 233 Britannia hits) - a nameless hit can never
clear a name-similarity threshold, so carrying it forward only inflates the
corpus. Returns [] rather than raising when OFF is unreachable, so one bad
brand does not abort a multi-brand backfill.
corpus.
A fetch that fails part-way returns what it got with `error` set and
writes NO cache file. Caching a failure as `hits: []` is how "Open Food
Facts has nothing for this brand" was being asserted for brands OFF had
simply been too busy to answer about.
"""
country = BARCODE_COUNTRY_TAG if country is None else (country or None)
slug = re.sub(r"[^a-z0-9]+", "_", brand.lower()).strip("_")
path = _cache_path(slug, cache_dir)
if not refresh and path.exists():
try:
cached = json.loads(path.read_text(encoding="utf-8"))
hits = cached.get("hits") or []
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
brand, len(hits), cached.get("fetched_at_human", "?"))
return hits
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
cached = _read_cache(path, brand)
if cached is not None:
return CorpusFetch(hits=cached, from_cache=True)
hits: List[Dict[str, Any]] = []
error: Optional[str] = None
page = 1
while page <= MAX_PAGES:
total_pages = 1
while page <= min(total_pages, MAX_PAGES):
if page > 1:
time.sleep(PAUSE_SECONDS)
payload = _get_page(brand, country, page)
batch = payload.get("hits") or []
try:
payload = _get_page(brand, country, page)
except requests.exceptions.RequestException as e:
# Retries exhausted (OffUnavailable is a ConnectionError too).
error = str(e) or e.__class__.__name__
break
if payload is None:
error = "Open Food Facts returned an unusable response"
break
batch = payload.get("products") or []
hits.extend(h for h in batch if isinstance(h, dict) and (h.get("product_name") or "").strip())
page_count = payload.get("page_count") or 0
if page >= page_count or not batch:
if page == 1:
# `page_count` is products-on-this-page in the v2 API; the real
# page total is `count` over the page size the SERVER applied -
# it silently caps the requested size (250 asked, 100 served for
# Amul), so dividing by PAGE_SIZE under-pages.
count = payload.get("count") or 0
served = payload.get("page_size") or len(batch) or PAGE_SIZE
try:
total_pages = max(1, math.ceil(int(count) / int(served)))
except (TypeError, ValueError, ZeroDivisionError):
total_pages = 1
if not batch:
break
page += 1
if error:
logger.warning(" OFF corpus for %r: fetch failed after %d usable product(s) - %s "
"(nothing cached)", brand, len(hits), error)
return CorpusFetch(hits=hits, error=error)
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(".json.tmp")
tmp.write_text(json.dumps({
"schema": CACHE_SCHEMA,
"endpoint": SEARCH_URL,
"brand": brand,
"brand_tag": brand_tag_slug(brand),
"country": country,
"fetched_at": time.time(),
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
@@ -208,7 +318,20 @@ def fetch_brand_corpus(brand: str,
os.replace(tmp, path)
logger.info(" OFF corpus for %r: %d usable product(s) fetched", brand, len(hits))
return hits
return CorpusFetch(hits=hits)
def fetch_brand_corpus(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
"""`fetch_brand_corpus_result(...).hits` - the list-only form every
backfill and ingestion caller uses. Returns [] rather than raising when
OFF is unreachable, so one bad brand does not abort a multi-brand run;
callers that need to tell "unreachable" from "empty" use the result form.
"""
return fetch_brand_corpus_result(brand, country=country, refresh=refresh,
cache_dir=cache_dir).hits
# ---------------------------------------------------------------------------

View File

@@ -243,6 +243,16 @@ def _round2(value: Optional[float]) -> Optional[float]:
return round(value, 2)
def _held_number(value: object) -> Optional[float]:
"""A price the row already carries, or None for blank / unparseable."""
if value is None or (isinstance(value, str) and not value.strip()):
return None
try:
return float(value)
except (TypeError, ValueError):
return None
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
"""Compute the pricing fields added by this stage for one catalog row.
@@ -253,8 +263,24 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
derivable from the data this pipeline generates, so they stay None
(exactly as the existing enriched exports store them).
A PRICE THE ROW ALREADY HOLDS WINS. A store sheet that says "Retail Price
155" has stated a fact; the band derived from it is Rs143-167, and this
stage used to read the band's ceiling back as "selling_price = 167" and
hand that to `EnrichmentStage.apply`, which overwrites a held value with
a different one (it only refuses to BLANK one). Held prices are therefore
returned as None here so `apply` keeps them, the base for the tax is the
held selling price, then the held final price, and only then the band.
"""
selling_price = _extract_selling_price(product.get("price_range"))
held_selling = _held_number(product.get("selling_price"))
held_final = _held_number(product.get("final_selling_price"))
if held_selling is not None:
selling_price = held_selling
elif held_final is not None:
selling_price = held_final
else:
selling_price = _extract_selling_price(product.get("price_range"))
gst = product.get("gst_percent")
tax_amount = None
final_selling_price = None
@@ -263,10 +289,10 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
final_selling_price = _round2(selling_price + tax_amount)
return {
"selling_price": _round2(selling_price),
"selling_price": None if held_selling is not None else _round2(selling_price),
"cost_price": None,
"tax_amount": tax_amount,
"final_selling_price": final_selling_price,
"final_selling_price": None if held_final is not None else final_selling_price,
"profit_before_tax": None,
"profit_after_tax": None,
}

View File

@@ -76,7 +76,7 @@ logger = logging.getLogger(__name__)
# Matches scripts/backfill_barcodes_from_off.py, which measured it.
DEFAULT_MIN_SIMILARITY = 0.88
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
BARCODE_SOURCE = "openfoodfacts_bulk (world.openfoodfacts.org/api/v2)"
def _rows_needing_a_barcode(cur, table: str) -> List[Dict[str, Any]]:

View File

@@ -63,7 +63,7 @@ import sqlite3
import threading
import time
from contextlib import closing
from dataclasses import dataclass
from dataclasses import dataclass, field
from pathlib import Path
from typing import Iterable, List, Optional, Sequence
from urllib.parse import urlparse
@@ -175,6 +175,23 @@ def distinctive_tokens(product_name: str, brand: str) -> List[str]:
return out
def _token_in(token: str, lowered_url: str) -> bool:
"""Does the URL carry this word, allowing for the plural on either side?
The catalogue names "Aachi Appalams" and "Aachi Pickles"; the photo is
`Aachi-Appalam-100-g-1.webp`. A whole-token substring test rejected the
correct image for the plural and then promoted an opaque Amazon URL over
it. The singular is tried as well - only for tokens long enough that
stripping the "s" leaves a real word ("gems" -> "gem" is fine; "kgs" never
gets here, size tokens are removed upstream).
"""
if token in lowered_url:
return True
if len(token) > 4 and token.endswith("s") and token[:-1] in lowered_url:
return True
return False
def search_key(product_name: str, brand: str) -> tuple:
"""Collapse a trailing pack size so sizes of one product share a lookup.
@@ -213,7 +230,7 @@ def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
distinctive = distinctive_tokens(product_name, brand)
if distinctive:
hit = next((t for t in distinctive if t in lowered), None)
hit = next((t for t in distinctive if _token_in(t, lowered)), None)
if hit:
return Corroboration(True, f"url names {hit!r}")
return Corroboration(
@@ -272,6 +289,26 @@ _lock = threading.Lock()
_initialized = False
_CACHE_TTL_SECONDS = 30 * 24 * 3600
# Open*Facts allows 100 product reads a minute per IP. A brand ingestion or a
# re-gate asks about every distinct barcode it meets, and the Dabur run issued
# about a hundred in two minutes - the tail was throttled, and a throttled
# lookup fails open, which is how a Kellogg's honey got past the gate for a
# moment. Pacing the calls keeps the gate answering instead of guessing.
_OFF_MIN_INTERVAL_SECONDS = 0.65
_OFF_LOOKUP_ATTEMPTS = 2
_OFF_RETRY_SLEEP_SECONDS = 3.0
_off_last_call = 0.0
_off_pace_lock = threading.Lock()
def _pace_off_lookup() -> None:
global _off_last_call
with _off_pace_lock:
wait = _OFF_MIN_INTERVAL_SECONDS - (time.monotonic() - _off_last_call)
if wait > 0:
time.sleep(wait)
_off_last_call = time.monotonic()
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
@@ -366,29 +403,58 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
if not tokens:
return True
key = f"{api_host}:{barcode}:{(brand or '').lower()}"
# "v2:" - entries written before the fail-open verdicts stopped being
# cached are ignored rather than trusted; they age out with the TTL.
key = f"v2:{api_host}:{barcode}:{(brand or '').lower()}"
cached = _cache_get(key)
if cached is not None:
return cached
verdict = True
try:
resp = requests.get(
f"https://{api_host}/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=timeout,
headers={"User-Agent": _BROWSER_UA},
)
# ONLY A REAL ANSWER IS CACHED. The fail-open True for a lookup that did
# not happen - a 429 or 503 from Open*Facts, a non-JSON body - used to be
# written to the cache too, for thirty days. A Dabur ingestion issued a
# hundred lookups in two minutes, the tail of them were throttled, and
# Kellogg's "Miel Pops", a Toblerone and a Nature Valley bar were filed as
# Dabur products; "Dabur Honey 1kg" then showed the Kellogg's honey. A
# throttled lookup still fails open for THIS call, but the next call asks
# again.
payload = None
for attempt in range(_OFF_LOOKUP_ATTEMPTS):
_pace_off_lookup()
try:
resp = requests.get(
f"https://{api_host}/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=timeout,
headers={"User-Agent": _BROWSER_UA},
)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
return True
if resp.ok:
product = (resp.json() or {}).get("product") or {}
haystack = (
f"{product.get('brands') or ''} {product.get('product_name') or ''}"
).lower()
if haystack.strip():
verdict = any(t in haystack for t in tokens)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
try:
payload = resp.json() or {}
except ValueError:
return True
break
if resp.status_code == 429 or resp.status_code >= 500:
# Throttled or unwell. One paced retry is cheap and turns most of
# these into a real answer; a second failure fails open, uncached.
time.sleep(_OFF_RETRY_SLEEP_SECONDS * (attempt + 1))
continue
return True
if payload is None:
return True
product = payload.get("product") or {}
if not product and payload.get("status") == 0:
# Open*Facts positively says: no such barcode. Nothing to compare
# against, and asking again will not change that - cache the open verdict.
_cache_set(key, True)
return True
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
if not haystack.strip():
return True
verdict = any(t in haystack for t in tokens)
_cache_set(key, verdict)
return verdict
@@ -398,11 +464,25 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
# ---------------------------------------------------------------------------
@dataclass
class PrimaryChoice:
"""The outcome of choosing a product's `image_url` from its candidates."""
"""The outcome of choosing a product's `image_url` from its candidates.
`ordered` is every candidate, best first. `eligible` is the subset that
may be SHOWN - tiers 1 and 2 - and is what the store pipeline persists as
`image_urls`. The two differ for a reason that was learned the hard way:
tier 3 was kept in `image_urls` "for review, never promoted", but the
product card falls back to `image_urls[0]` whenever `image_url` is empty
and the product modal shows the whole list as a gallery. So for "Dabur
Honey 1kg" the gate correctly withheld the primary - every candidate was
another company's honey - and the UI displayed those very honeys anyway.
A candidate the gate would not promote must not be stored where the UI
will promote it.
"""
primary: Optional[str]
ordered: List[str]
reason: str
eligible: List[str] = field(default_factory=list)
rejected: List[str] = field(default_factory=list)
def choose_primary(
@@ -424,7 +504,10 @@ def choose_primary(
catalog_engine's comment protects, where dropping uncorroborated
candidates would leave real products with no image at all.
3. The URL could have named the product and did not - a human-authored
Commons filename, say. Kept in `image_urls`, never promoted.
Commons filename, say - or an Open*Facts photo whose barcode belongs
to another brand. Returned in `ordered` and `rejected`, never in
`eligible`, and the store pipeline does not persist it (see
PrimaryChoice for why "kept for review" was not safe).
Nothing eligible means `primary is None`. A blank image renders as the
brand monogram, which is honest; another company's product is not, and it
@@ -466,12 +549,16 @@ def choose_primary(
tier3.sort(key=looks_like_person_photo)
ordered = tier1 + tier2 + tier3
eligible = tier1 + tier2
if tier1:
return PrimaryChoice(tier1[0], ordered, "url names the product")
return PrimaryChoice(tier1[0], ordered, "url names the product",
eligible=eligible, rejected=tier3)
if tier2:
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host")
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host",
eligible=eligible, rejected=tier3)
return PrimaryChoice(
None,
ordered,
"no candidate names this product; primary withheld rather than guessed",
eligible=eligible, rejected=tier3,
)

View File

@@ -114,11 +114,44 @@ def _query_openfacts(query: str, max_results: int) -> list:
return []
def _openfacts_product_is_brand(product: dict, brand: Optional[str]) -> bool:
"""Does this Open*Facts record belong to `brand`?
Open*Facts' free-text search matches on the NAME, not the brand: asked for
"Dabur Honey 1kg" it answered with a UK, a French, a Swiss and a Spanish
honey, every one a real front-of-pack photo of somebody else's product.
The record says who made it - `brands` / `brands_tags` - and that field
used to be thrown away here, leaving a per-URL API round trip downstream
(`image_corroboration.openfacts_product_matches_brand`) as the only thing
standing between those photos and the catalogue. Check it at the source.
Fails OPEN on a record with no brand at all: an unlabelled record is not
evidence of another brand, and the downstream gate still runs.
"""
if not brand:
return True
from app.services.image_corroboration import brand_tokens
tokens = brand_tokens(brand)
if not tokens:
return True
tags = product.get("brands_tags") or []
haystack = " ".join([str(product.get("brands") or "")] + [str(t) for t in tags]).lower()
if not haystack.strip():
return True
return any(t in haystack for t in tokens)
def find_images_openfacts(title: str, brand: Optional[str] = None, max_results: int = 20) -> list:
"""Query the Open *Facts family of open product databases for real
product photos. No API key required. Falls back from a brand+title
query to a title-only query if the combined query is too specific to
match anything (small/regional brand name variants are a common case)."""
match anything (small/regional brand name variants are a common case).
Only records whose own `brands` field names our brand are used - see
`_openfacts_product_is_brand`. The title-only fallback makes this filter
load-bearing: "Honey 1kg" matches every honey on the site.
"""
if not USE_OPEN_FACTS:
return []
@@ -126,12 +159,12 @@ def find_images_openfacts(title: str, brand: Optional[str] = None, max_results:
if not query:
return []
products = _query_openfacts(query, max_results)
products = [p for p in _query_openfacts(query, max_results) if _openfacts_product_is_brand(p, brand)]
if not products and brand and title:
# Combined "brand + title" query found nothing - retry with just
# the title, since Open*Facts' free-text search is exact-ish and
# brand naming conventions vary (e.g. "Dettol" vs "Reckitt Dettol").
products = _query_openfacts(title, max_results)
products = [p for p in _query_openfacts(title, max_results) if _openfacts_product_is_brand(p, brand)]
urls: List[str] = []
for product in products:

View File

@@ -332,16 +332,29 @@ def fetch_brand_catalog_exhaustive(brand: str, max_products: int = 300) -> Dict[
return {"brand": brand, "products": products_out}
def fetch_product_details(brand: str, product_title: str) -> Dict[str, Any] | None:
def fetch_product_details(brand: str, product_title: str,
category: str | None = None,
size: str | None = None) -> Dict[str, Any] | None:
"""Get details for a single product: description, image_url, pricing fields.
Returns a dict with keys: description, image_url, size_variants, price_ranges, price_range, provider_examples.
`category` and `size` are optional context. A 1.5b model asked about
"Naga Maida" alone will happily describe a curry; told it is a 500g pack
in Flours & Grains it describes refined wheat flour. Both are facts the
caller already holds, so they cost nothing to pass.
"""
if not _ensure_client():
return None
context = f"Brand: {brand}\nProduct: {product_title}\n"
if category:
context += f"Category: {category}\n"
if size:
context += f"Pack size: {size}\n"
user_prompt = (
f"Brand: {brand}\nProduct: {product_title}\n"
"Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
"description must be <=160 chars, concise and factual."
context
+ "Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
"description: 1-2 sentences, <=220 chars, factual - what the product is, "
"its form and its typical use. No marketing adjectives, no claims you cannot know."
)
text = _generate(SYSTEM_PROMPT, user_prompt)
if not text:

View File

@@ -5,7 +5,7 @@ WHY THIS FILE EXISTS
--------------------
Open Food Facts answers "does this product exist" for food, and answers it
well. It does not answer it for anything else: `off_bulk` queries only
`search.openfoodfacts.org`, so a toothpaste or a detergent has no
Open Food Facts' food database, so a toothpaste or a detergent has no
product-discovery source at all and its entire catalogue is language-model
output. Measured 2026-09-10, India-tagged coverage on the sibling databases is
2-13 products per non-food brand against Britannia's 218 on OFF - real, but