Brand valid image generation

This commit is contained in:
sriram
2026-09-11 15:47:56 +05:30
parent e1a5962f82
commit ce4fa70dee
31 changed files with 5586 additions and 1495 deletions

View File

@@ -17,14 +17,20 @@ now the only admin way in. What made it the survivor is the unit of work: five
files are one batch with one id, so the question a colleague actually asks -
"did the drop land?" - has one answer rather than five.
WHY THE NETWORK STAGES DEFAULT OFF HERE
---------------------------------------
Turning both on for a single file is a considered trade. At twenty files it is
thousands of outbound requests and, for the image stage, a Playwright subprocess
that can burn three minutes on its own - on a single-vCPU container that is also
serving the API. So a batch opts IN to those stages; it does not opt out.
`USE_OLLAMA` is false in production anyway, which makes `use_llm` a no-op there
and the honest default obvious.
WHY IMAGES DEFAULT OFF HERE AND THE LLM DEFAULTS ON
---------------------------------------------------
Image search for a single file is a considered trade. At twenty files it is
thousands of outbound requests and a Playwright subprocess that can burn three
minutes on its own - on a single-vCPU container that is also serving the API.
So a batch opts IN to that stage; it does not opt out.
`use_llm` is different and defaults ON: it gates only the description written
in stage 2 for rows the sheet left blank, which is the one field a shopper
reads and a store almost never supplies. It is cheap to leave on because
`ollama_service._ensure_client` caches its reachability probe per batch and
`store_catalog_pipeline.LlmBreaker` stops calling after three consecutive
misses, so with `USE_OLLAMA` false (production today) the cost is one warning
per file and the rows fall back to a factual template.
Note that these defaults bind THIS router only. The open upload endpoint runs
itself and takes its two flags from `UPLOAD_AUTORUN_FETCH_IMAGES` and
@@ -145,7 +151,7 @@ async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
dependencies=[Depends(require_admin)])
async def ingest_catalog_batch(
files: List[UploadFile] = File(...),
use_llm: bool = False,
use_llm: bool = True,
fetch_images: bool = False,
) -> BatchOut:
"""Stage the files, queue the batch, and return an id to poll.
@@ -325,7 +331,8 @@ class InboxStartRequest(InboxSelection):
# Chosen HERE, not by the sender - see the note on the POST handler in
# uploads.py. These commit the host to outbound work, so the decision
# belongs to the person who can see what the machine is already doing.
use_llm: bool = False
# The LLM is on by default for the reason in the module docstring.
use_llm: bool = True
fetch_images: bool = False
# Who runs it. "inprocess" is this container's worker thread and is the
# default, so an existing client that never sends the field is unaffected.

View File

@@ -124,11 +124,14 @@ def _normalize_header(raw: Any) -> str:
_EXACT_HEADERS: Dict[str, str] = {
# brand
# brand. "productbrand" is what a camel-cased ProductBrand header becomes
# after _normalize_header, and is listed so the mapping does not depend on
# the keyword rules' ordering.
"brand": "brand", "brand name": "brand", "brands": "brand",
"company": "brand", "manufacturer": "brand", "company name": "brand",
"product brand": "brand", "productbrand": "brand",
# product name
"product": "product_name", "product name": "product_name",
"product": "product_name", "product name": "product_name", "productname": "product_name",
"product name variant": "product_name", "product variant": "product_name",
"variant": "product_name", "name": "product_name", "item": "product_name",
"item name": "product_name", "product title": "product_name",
@@ -138,18 +141,27 @@ _EXACT_HEADERS: Dict[str, str] = {
# category
"category": "category", "categories": "category", "cat": "category",
"product category": "category", "segment": "category",
# price
# price. Two columns, two meanings: `selling_price` is what the shop
# charges - a retail / sale / selling price, or a bare "Price" - and
# `final_selling_price` is the tax-inclusive ceiling (MRP, "final price").
# A sheet that carries only one of them still fills both, because
# vector_store.upsert_brand_products copies selling -> final when final is
# blank. Before this split every price header landed in final_selling_price
# and selling_price was NULL for every uploaded row.
"price range": "price_range", "range": "price_range", "mrp range": "price_range",
"price": "final_selling_price", "final price": "final_selling_price",
"final price rs": "final_selling_price", "final selling price": "final_selling_price",
"selling price": "final_selling_price", "mrp": "final_selling_price",
"rate": "final_selling_price", "amount": "final_selling_price",
"cost": "final_selling_price", "unit price": "final_selling_price",
"selling price": "selling_price", "sellingprice": "selling_price",
"retail price": "selling_price", "retailprice": "selling_price",
"sale price": "selling_price", "sp": "selling_price",
"price": "selling_price", "unit price": "selling_price", "rate": "selling_price",
"final price": "final_selling_price", "final price rs": "final_selling_price",
"final selling price": "final_selling_price", "mrp": "final_selling_price",
"amount": "final_selling_price", "cost": "final_selling_price",
# identifiers
"barcode": "barcode", "barcode gtin ean": "barcode", "bar code": "barcode",
"gtin": "barcode", "ean": "barcode", "upc": "barcode", "ean13": "barcode",
"hsn": "hsn_code", "hsn code": "hsn_code", "hsn sac": "hsn_code",
"sku": "product_sku", "product sku": "product_sku", "sku code": "product_sku",
"sku": "product_sku", "product sku": "product_sku", "productsku": "product_sku",
"sku code": "product_sku",
"fssai": "fssai_license", "fssai license": "fssai_license",
"fssai license number": "fssai_license", "fssai number": "fssai_license",
# text / media
@@ -173,7 +185,13 @@ _KEYWORD_RULES: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
("barcode", ("barcode", "bar code", "gtin", "ean", "upc")),
("product_sku", ("sku",)),
("price_range", ("price range", "range")),
("final_selling_price", ("final price", "selling price", "price", "mrp", "rate", "cost")),
# Order is load-bearing: "final selling price" contains "selling price",
# and every price header contains "price", so the final-price forms must
# be claimed before the generic ones. "sp" is deliberately exact-only - as
# a substring it is inside "spice", "display" and "sponge".
("final_selling_price", ("final price", "final selling price", "mrp")),
("selling_price", ("selling price", "retail price", "sale price", "price", "rate")),
("final_selling_price", ("cost",)),
("image_url", ("image", "photo", "picture", "url", "link")),
("description", ("description", "desc", "detail")),
("category", ("category", "segment")),
@@ -346,6 +364,7 @@ def row_to_request(row: Dict[str, Any], mapping: _ColumnMapping) -> AddProductRe
product_sku=_text(row, mapping, "product_sku"),
hsn_code=_text(row, mapping, "hsn_code"),
final_selling_price=_number(row, mapping, "final_selling_price"),
selling_price=_number(row, mapping, "selling_price"),
barcode=_text(row, mapping, "barcode"),
image_url=_text(row, mapping, "image_url"),
)
@@ -528,8 +547,9 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
price_range = req.price_range
if not price_range:
if req.final_selling_price:
price_range = f"₹{req.final_selling_price}"
sheet_price = req.final_selling_price or req.selling_price
if sheet_price:
price_range = f"₹{sheet_price}"
elif sample_existing.get("price_range"):
price_range = sample_existing.get("price_range")
else:

View File

@@ -91,7 +91,8 @@ def generate_product_highlights(product: Dict[str, Any], brand: str) -> List[str
highlights.append(f"Available in {size} - {price}")
elif size:
highlights.append(f"Available in {size}")
elif isinstance(v, str) and v.strip():
elif isinstance(v, str) and v.strip() and v.strip().lower() != "standard":
# "Standard" is the pipeline's no-size sentinel, not a pack size.
highlights.append(f"Available in {v}")
# Quality indicators from description

View File

@@ -47,7 +47,7 @@ import asyncio
import logging
import re
from dataclasses import dataclass, field
from typing import Any, Callable, Dict, List, Optional, Tuple
from typing import Any, Callable, Dict, List, Optional, Sequence, Tuple
from app.infrastructure.settings import (
ENABLE_BARCODE_LOOKUP,
@@ -132,8 +132,16 @@ _ENRICHABLE = (
"selling_price", "image_url", "image_urls", "size_variants",
)
# The pack size a store sheet usually carries inside the product name itself -
# "India Gate Basmati Rice 1kg", "Fortune Sunflower Oil 1L", "Amul Milk 1
# Litre". The match is kept as written in the name (case and spacing), so the
# size a shopper sees on the card is the size the store typed. Longer unit
# spellings precede their prefixes because alternation is first-match-wins.
_SIZE_IN_TITLE = re.compile(
r"\b(\d+(?:[.,]\d+)?)\s*(kg|kgs|g|gm|gms|gram|grams|ml|l|ltr|litre|liters?|pack|pcs|n)\b",
r"\b(\d+(?:[.,]\d+)?)\s*"
r"(kgs|kg|gms|gm|grams|gram|mg|g|"
r"mls|ml|litres|litre|liters|liter|ltrs|ltr|lt|l|"
r"pieces|piece|pcs|pc|nos|pack|n)\b",
re.I,
)
@@ -276,24 +284,71 @@ def stage_1_brand_and_fssai(row: Dict[str, Any]) -> Dict[str, Any]:
# ---------------------------------------------------------------------------
# Stage 2 - Row intake (replaces AI discovery)
# ---------------------------------------------------------------------------
def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True) -> Dict[str, Any]:
@dataclass
class LlmBreaker:
"""Stop asking the LLM once it has stopped answering.
`ollama_service._generate` swallows every exception and returns "", so a
slow or dead Ollama never raises - it just costs up to
OLLAMA_TIMEOUT_SECONDS x retries per row and yields nothing. On a
2000-row sheet that is days. Three misses in a row and the rest of the
batch goes straight to the factual fallback description; one success
resets the count, so a single hiccup does not switch the LLM off.
"""
limit: int = 3
consecutive_misses: int = 0
tripped: bool = False
def record(self, ok: bool) -> None:
if ok:
self.consecutive_misses = 0
return
self.consecutive_misses += 1
if self.consecutive_misses >= self.limit:
self.tripped = True
# The generated description is stored in a TEXT column, but a 1.5b model does
# not reliably honour the length it is asked for and the catalog card shows a
# short paragraph, not an essay.
_MAX_LLM_DESCRIPTION_CHARS = 300
def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True,
breaker: Optional[LlmBreaker] = None) -> Dict[str, Any]:
"""The spreadsheet row *is* the product. Only fill a missing description.
The LLM call is best-effort and optional: with Ollama unreachable the row
keeps an empty description and later stages still work.
keeps an empty description here and `_to_storage_row` supplies a factual
template at storage time, so later stages still work.
DESCRIPTION ONLY. This stage used to copy the LLM's `size_variants` into a
row that had none, which is how "Naga Maida" acquired a 500g/1kg/5kg it
was never sold in. A pack size is a fact about the physical product and
comes from the sheet or the product name (stage 4), never from a guess.
"""
if not _blank(row.get("description")) or not use_llm:
return row
if breaker is not None and breaker.tripped:
return row
ok = False
try:
from app.services.ollama_service import fetch_product_details
details = fetch_product_details(row.get("brand", ""), row.get("product_name", ""))
sheet_sizes = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
title_size = _SIZE_IN_TITLE.search(row.get("product_name") or "")
size = sheet_sizes[0] if sheet_sizes else (title_size.group(0).strip() if title_size else None)
details = fetch_product_details(
row.get("brand", ""), row.get("product_name", ""),
category=row.get("category") or None, size=size,
)
if details and details.get("description"):
row["description"] = str(details["description"]).strip()
if _blank(row.get("size_variants")) and details and details.get("size_variants"):
row["size_variants"] = list(details["size_variants"])
row["description"] = str(details["description"]).strip()[:_MAX_LLM_DESCRIPTION_CHARS]
ok = True
except Exception as exc: # noqa: BLE001 - enrichment is best-effort
logger.debug("LLM enrichment skipped for %r: %s", row.get("product_name"), exc)
if breaker is not None:
breaker.record(ok)
return row
@@ -442,30 +497,32 @@ def _sizes_for(row: Dict[str, Any]) -> List[str]:
if match:
return [match.group(0).strip()]
# A COMMODITY WITH NO WEIGHT COLUMN GETS ONE UNSIZED ROW, NOT THREE MADE-UP
# ONES. default_size_variants() invents a plausible set - 100g/250g/500g -
# and for packaged goods that is a reasonable guess at what a brand sells.
# For loose produce it is not: a shop sells apples by the kilo at whatever
# the customer asks for, so "Apple 250g" is a product that does not exist.
# NO SIZE ANYWHERE MEANS ONE UNSIZED ROW, NOT THREE MADE-UP ONES - for
# every brand. This used to fall through to default_size_variants() for
# branded rows, which invents a plausible set (1kg/5kg/25kg for rice) and
# so turned "India Gate Basmati Rice" into three products that the sheet
# never listed and the shop may never have stocked. It was also the
# mechanism behind the catalogue drift the integrator reported: the
# invented set is keyed on the resolved category, so the same product
# ingested twice with the category resolved differently produced two
# disjoint size sets, two sets of image_ids, and a re-scrape that looked
# like the old rows were deleted. A pack size is recorded only when the
# sheet or the product name states it.
#
# It is also the mechanism behind the catalogue drift the integrator
# reported: the invented set is keyed on the resolved category, so the same
# product ingested twice with the category resolved differently produces
# two disjoint size sets, two sets of image_ids, and a re-scrape that looks
# like the old rows were deleted. Keeping produce out of that from the
# start is cheaper than repairing it later.
# "Standard" and not "": an empty size fails validate_size ("size/pack is
# missing") and stage 4 would drop the row, which is the rejection this
# whole area exists to prevent. "Standard" is also what _to_storage_row
# already substitutes for a blank size and what the seeded base list uses,
# so an uploaded "Apple" deduplicates onto the seeded "Apple" instead of
# creating a second row.
if row.get("brand") == OWN_PRODUCTS_BRAND:
return ["Standard"]
return list(price_estimator.default_size_variants(
row.get("category") or "", row.get("product_name") or ""
))
# missing") and stage 10 would penalise the row for a fact nobody had. It is
# the pipeline's internal no-size sentinel, understood by the validator,
# the price estimator and the SKU resolver. What reaches the table depends
# on the brand - see _to_storage_row: Own Products keeps "Standard" so an
# uploaded "Apple" deduplicates onto the seeded "Apple" (same image_id);
# a branded row stores an empty size_variants and an unsuffixed name.
if row.get("brand") != OWN_PRODUCTS_BRAND:
# Worth a line in the job report for a packaged good; for loose
# produce an absent size is the normal case and would only be noise.
row.setdefault("_notes", []).append(
"no pack size in the sheet or the product name; stored without one"
)
return ["Standard"]
def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], List[str]]:
@@ -519,8 +576,14 @@ def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
if not _blank(row.get("price_range")):
return row
if not _blank(row.get("final_selling_price")):
price = float(row["final_selling_price"])
# Either price the sheet gave is a real price for this pack; the final
# (tax-inclusive) one is preferred when both are present because the band
# is what the card shows a shopper.
sheet_price = row.get("final_selling_price")
if _blank(sheet_price):
sheet_price = row.get("selling_price")
if not _blank(sheet_price):
price = float(sheet_price)
lo = int(round(price * (1 - _SHEET_PRICE_BAND)))
hi = int(round(price * (1 + _SHEET_PRICE_BAND)))
elif row.get("brand") == OWN_PRODUCTS_BRAND:
@@ -537,7 +600,84 @@ def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
# ---------------------------------------------------------------------------
# Stage 6 - Images
# ---------------------------------------------------------------------------
def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, Any]:
class ImageSearchCache:
"""One image search per PRODUCT, shared by every pack size of it.
Stage 4 turns one source row into one row per size, and this stage used to
search once per resulting row. The query was the same each time - the size
is not in it - but the providers behind `find_all_image_urls` are live and
not deterministic, so "Dabur Honey" asked four times came back four ways:
the 225g row drew an Amazon photo of Dabur Honey, the 1kg row drew nothing
usable. Same product, four different images, three of them wrong or blank.
Keyed on `image_corroboration.search_key`, which strips a trailing pack
size from the name, so sheet rows that carry the size inside the name
("India Gate Basmati Rice 1kg" / "... 5kg") share as well. The cached value
is the RANKED candidate list; each size still runs its own gate, which
ignores size tokens and so reaches the same verdict for every sibling.
`existing` maps a search key to an already-stored eligible primary for
the same product under the same brand, so a re-ingest that finds nothing
new can inherit the photo a sibling size already has - after the gate.
"""
def __init__(self, existing_rows: Optional[Sequence[Dict[str, Any]]] = None) -> None:
self.candidates: Dict[tuple, List[str]] = {}
self.existing: Dict[tuple, str] = {}
for prior in existing_rows or ():
url = prior.get("image_url")
if not url or not str(url).startswith("http"):
continue
key = image_corroboration.search_key(prior.get("product_name") or "",
prior.get("brand") or "")
self.existing.setdefault(key, str(url))
def _adopt_sibling_image(row: Dict[str, Any], brand: str, cache: ImageSearchCache) -> bool:
"""Reuse the eligible primary of another pack size of the same product.
Only ever through the gate: the sibling's URL is corroborated against THIS
row's title before it is adopted, so a wrong image stored under one size
cannot propagate to the rest. A sibling is the same product in a different
pack, which is what `search_key` means, and the same photo on every pack is
what a retailer shows too.
"""
key = image_corroboration.search_key(row.get("product_name") or "", brand)
url = cache.existing.get(key)
if not url:
return False
choice = image_corroboration.choose_primary([url], row.get("product_name") or "", brand)
if choice.primary is None:
return False
row["image_url"] = choice.primary
row["image_urls"] = [choice.primary]
row.setdefault("_notes", []).append("image adopted from another pack size of this product")
return True
def _image_cache_for(rows: Sequence[Dict[str, Any]]) -> ImageSearchCache:
"""The run's shared image cache, primed with what each brand already holds.
Reads each destination brand once (stage 11 reads it again for the merge;
two reads of a small table beat threading the rows through five stages).
A read failure primes nothing - the search still runs.
"""
existing: List[Dict[str, Any]] = []
for brand in sorted({r.get("brand") for r in rows if r.get("brand")}):
if brand == OWN_PRODUCTS_BRAND:
continue
try:
for prior in get_products_by_brand(brand) or []:
prior = dict(prior)
prior.setdefault("brand", brand)
existing.append(prior)
except Exception as exc: # noqa: BLE001 - priming is an optimisation
logger.warning("Could not read existing images for %s: %s", brand, exc)
return ImageSearchCache(existing)
def stage_6_images(row: Dict[str, Any], *, enabled: bool = True,
cache: Optional[ImageSearchCache] = None) -> Dict[str, Any]:
if not _blank(row.get("image_urls")) or not enabled:
return row
try:
@@ -580,33 +720,61 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
# drops the "-plant -tree -fish -botanical" Wikimedia exclusion block
# that fights every produce query. Without it this stage recreates the
# exact defect the Own Products image repair exists to undo.
candidates = find_all_image_urls(
product_name, brand=brand or None, max_results=24, produce=is_produce
)
if candidates:
best = catalog_engine._select_best_images(
candidates, product_name, brand, max_images=10
# A pack size of this product that is already in the table may carry
# the photo. Take it - through the gate - before spending a search:
# the same product in a different pack is the same photo, every size
# then agrees, and a re-ingest costs no network. See _adopt_sibling_image.
if cache is not None and not is_produce and _adopt_sibling_image(row, brand, cache):
return row
# One search per product, not per pack size - see ImageSearchCache.
# The query is the size-less name: a photo's filename or alt text
# almost never carries the pack size, and a size token in the query
# only steers the engines towards pages LISTING that size, which for
# "Honey 1kg" was every other brand's 1kg honey.
key = image_corroboration.search_key(product_name, brand)
base_name = image_corroboration.SIZE_TAIL.sub("", product_name).strip() or product_name
if cache is not None and key in cache.candidates:
best = cache.candidates[key]
else:
candidates = find_all_image_urls(
base_name, brand=brand or None, max_results=24, produce=is_produce
)
if best:
# `_select_best_images` ORDERS candidates; it does not judge
# whether any of them is this product. Its scoring awards points
# for the brand appearing in the URL, so for a brand that is
# also a personal name it actively rewarded the wrong photo -
# `Anil_Kapoor_2019.jpg` outscored everything and became
# `image_url`. choose_primary decides what may be promoted.
#
# The list is still stored in full. A product whose only
# candidates are uncorroborated keeps them for review; it just
# does not get one of them presented as fact.
choice = image_corroboration.choose_primary(
best, product_name, brand or ""
best = catalog_engine._select_best_images(
candidates, base_name, brand, max_images=10
) if candidates else []
if cache is not None:
cache.candidates[key] = list(best)
if best:
# `_select_best_images` ORDERS candidates; it does not judge
# whether any of them is this product. Its scoring awards points
# for the brand appearing in the URL, so for a brand that is
# also a personal name it actively rewarded the wrong photo -
# `Anil_Kapoor_2019.jpg` outscored everything and became
# `image_url`. choose_primary decides what may be promoted.
#
# ONLY ELIGIBLE CANDIDATES ARE STORED. The rejects used to be kept
# in `image_urls` "for review" - but the product card falls back to
# `image_urls[0]` when `image_url` is empty and the modal shows the
# list as a gallery, so the review pile was what the shopper saw:
# four foreign honeys for "Dabur Honey 1kg", a mosquito repellent
# for "Dabur Honey 500g". A reject is logged, not displayed.
choice = image_corroboration.choose_primary(
best, product_name, brand or ""
)
row["image_urls"] = list(choice.eligible)
row["image_url"] = choice.primary
if choice.rejected:
row.setdefault("_notes", []).append(
f"{len(choice.rejected)} image candidate(s) rejected: "
"they do not name this product"
)
row["image_urls"] = list(choice.ordered)
row["image_url"] = choice.primary
if choice.primary is None:
row.setdefault("_notes", []).append(
f"no primary image: {choice.reason}"
)
if choice.primary is None:
row.setdefault("_notes", []).append(
f"no primary image: {choice.reason}"
)
except Exception as exc: # noqa: BLE001 - an image is not worth the row
# `warning`, not `debug`: at the default log level a debug line is
# invisible, so a row that silently lost its images looked identical to
@@ -731,18 +899,50 @@ def build_image_id(brand: str, product_name: str, size: str) -> str:
return f"{_sanitize_name(brand)}_{slug}"
def _fallback_description(brand: str, name: str, size: str, category: str) -> str:
"""What the card says when neither the sheet nor the LLM said anything.
Factual by construction - only facts the row already holds: the product,
its pack size, its category, its maker. Deliberately NOT
`catalog_engine.generate_detailed_description`, which pads any product
into eight hundred characters of "premium quality" and "exceptional
taste" that nobody verified. An honest short line beats a confident essay.
"""
parts = [name.strip()]
if size and size.lower() not in name.lower():
parts.append(f"{size} pack")
if category and category.lower() not in ("general", "uncategorized"):
parts.append(category)
if brand:
# resolve_parent_brand hands back the alias KEY ("amul"), which is a
# lookup token rather than a name. Title-case only that form; a brand
# written with its own casing ("India Gate", "P&G") is left alone.
parts.append(f"by {brand.title() if brand.islower() else brand}")
return ", ".join(p for p in parts if p) + "."
def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"""Project an enriched row onto the brand-table columns."""
name = row.get("product_name") or row.get("title") or ""
size = row.get("size") or "Standard"
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
category = row.get("category") or "General"
own = row.get("brand") == OWN_PRODUCTS_BRAND
# "Standard" is the pipeline's internal no-size sentinel (see _sizes_for).
# Own Products keeps it in the table: the seeded base list uses it and an
# uploaded "Apple" has to build the same image_id as the seeded "Apple".
# A branded row has no such counterpart, so the sentinel is dropped here
# and the row is stored the way the sheet described it - "India Gate
# Basmati Rice", size_variants [], image_id without a size suffix - rather
# than as "India Gate Basmati Rice Standard".
raw_size = row.get("size") or "Standard"
size = raw_size if (own or raw_size != "Standard") else ""
display = name if (not size or size.lower() in name.lower()) else f"{name} {size}".strip()
category = row.get("category") or "General"
# "Apple 500g from Own Products." is a sentence nobody wrote and nobody
# wants, and "Own Products" is a bucket rather than a maker, so the
# template reads as a false provenance claim. A commodity keeps whatever
# description the sheet gave, including none.
description = row.get("description") or (None if own else f"{display} from {row.get('brand')}.")
description = row.get("description") or (
None if own else _fallback_description(row.get("brand") or "", display, size, category)
)
# The embedding text. Interpolating a null description and the bucket name
# would embed the literal "Own Products ... None", so a commodity is
# described to the vector index by what actually identifies it.
@@ -759,7 +959,7 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
"image_url": row.get("image_url"),
"image_urls": list(row.get("image_urls") or []),
"price_range": row.get("price_range"),
"size_variants": [size],
"size_variants": [size] if size else [],
"providers": list(row.get("providers") or []),
"fssai_license": row.get("fssai_license"),
"product_sku": row.get("product_sku"),
@@ -1001,9 +1201,16 @@ def run_pipeline(
progress(1, STAGE_NAMES[0], 0, len(prepared))
prepared = [stage_1_brand_and_fssai(r) for r in prepared]
breaker = LlmBreaker() if use_llm else None
for index, row in enumerate(prepared, start=1):
stage_2_row_intake(row, use_llm=use_llm)
stage_2_row_intake(row, use_llm=use_llm, breaker=breaker)
progress(2, STAGE_NAMES[1], index, len(prepared))
if breaker is not None and breaker.tripped:
result.warnings.append(
f"LLM descriptions stopped after {breaker.limit} consecutive failures "
"(Ollama unreachable or too slow); the remaining rows use the factual "
"template description."
)
progress(3, STAGE_NAMES[2], 0, len(prepared))
prepared = [stage_3_title_category(r) for r in prepared]
@@ -1031,8 +1238,9 @@ def run_pipeline(
for index, row in enumerate(exploded, start=1):
stage_5_pricing(row)
progress(5, STAGE_NAMES[4], index, len(exploded))
image_cache = _image_cache_for(exploded) if fetch_images else None
for index, row in enumerate(exploded, start=1):
stage_6_images(row, enabled=fetch_images)
stage_6_images(row, enabled=fetch_images, cache=image_cache)
progress(6, STAGE_NAMES[5], index, len(exploded))
for index, row in enumerate(exploded, start=1):
stage_7_sku(row)

View File

@@ -107,6 +107,7 @@ from app.infrastructure.settings import (
)
from app.services import active_brands, brand_store, ollama_service, retail_presence
from app.services.brand_registry import (
get_brand_store_domain,
get_fssai_license,
get_known_sub_brands,
resolve_parent_brand,
@@ -449,19 +450,29 @@ def _existing_title_index(brand: str) -> Dict[str, str]:
# ---------------------------------------------------------------------------
# Source A - Open Food Facts (ground truth)
# ---------------------------------------------------------------------------
class OpenFactsUnavailable(RuntimeError):
"""Open Food Facts could not be consulted - as opposed to consulted and
found empty. `discover_brand_products` turns this into a warning that says
so, because "OFF has nothing for this brand" is a claim about the brand
and "OFF was down" is not."""
def _from_open_facts(brand: str, *, refresh: bool = False) -> List[Dict[str, Any]]:
"""Real products for `brand`, with a real GTIN and often a real pack size.
`off_bulk.fetch_brand_corpus` is already disk-cached, retry-wrapped and
paced at 2s between pages, and `scripts/backfill_barcodes_from_off.py`
already imports it from outside the enrichment package, so reaching for it
here is an established pattern rather than a new one.
`off_bulk.fetch_brand_corpus_result` is disk-cached, retry-wrapped and
paced at 6s between 100-row pages. Raises `OpenFactsUnavailable` when the fetch
did not complete and produced nothing; anything else is returned as-is,
an empty list meaning OFF genuinely has no products under this brand tag.
"""
try:
hits = off_bulk.fetch_brand_corpus(brand, refresh=refresh)
result = off_bulk.fetch_brand_corpus_result(brand, refresh=refresh)
except Exception as exc: # noqa: BLE001 - a dead OFF must not kill discovery
logger.warning("Open Food Facts lookup failed for %r: %s", brand, exc)
return []
raise OpenFactsUnavailable(str(exc) or exc.__class__.__name__) from exc
if result.error and not result.hits:
raise OpenFactsUnavailable(result.error)
hits = result.hits
out: List[Dict[str, Any]] = []
for hit in hits or ():
@@ -759,10 +770,10 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str,
which is what makes `image_id` reproducible - the name is half of it.
The LLM's list is a guess and comes last. Only when all three are empty is
the cell left blank, which hands the decision to stage 4's
`default_size_variants` - and that invents 100g/250g/500g, which
`_sizes_for`'s own comment identifies as the mechanism behind observed
catalogue drift. Leaving it blank is the last resort, not the default.
the cell left blank; stage 4 then stores ONE unsized row (it no longer
invents 100g/250g/500g - see `_sizes_for`, which names that invention as
the mechanism behind observed catalogue drift). Blank is honest, but a
real size is what makes the row a distinct pack, so it is worth finding.
"""
in_title = _TITLE_SIZE_RE.search(title or "")
if in_title:
@@ -850,25 +861,21 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
# semantic search.
#
# A blank description is honest and already handled: `_to_storage_row`
# substitutes "<name> <size> from <brand>." exactly as it does for a store
# sheet that left the column empty.
# substitutes a short factual line (name, size, category, brand) exactly
# as it does for a store sheet that left the column empty.
description = (candidate.get("description") or "").strip()
# A BARCODE IDENTIFIES ONE PACK, SO IT ONLY SURVIVES A KNOWN PACK SIZE.
# With no real size, stage 4 falls back to `default_size_variants` and
# invents 100g/250g/500g - and because every column is copied into each
# exploded variant, one real GTIN would be stamped onto three packs, two of
# which do not exist. That is a worse error than a blank barcode: it is
# wrong data that looks authoritative, and `nutrition_by_barcode` would
# happily resolve all three to the same product.
# A BARCODE IDENTIFIES ONE PACK. That used to mean dropping it whenever no
# pack size was known, because stage 4 then invented 100g/250g/500g and
# every column is copied into each exploded variant - one real GTIN
# stamped onto three packs, two of which did not exist. Stage 4 now stores
# a single unsized row when it has no size, and one GTIN on one row is
# exactly what a GTIN means, so the barcode survives; the row is flagged
# so a reviewer knows the pack size is still unknown.
barcode = candidate.get("barcode")
notes: List[str] = []
if barcode and not sizes:
notes.append(
"barcode dropped: no real pack size is known, and stage 4 will "
"invent several - a GTIN names one pack, not three"
)
barcode = None
notes.append("pack size unknown: the barcode names one pack, but its size was not found")
shape = {"title": title, "category": category_hint, "size_variants": sizes,
"description": description}
@@ -929,12 +936,25 @@ def discover_brand_products(
registry_terms = _registry_terms(brand)
fssai = get_fssai_license(brand)
off_candidates = _from_open_facts(brand, refresh=refresh_corpus) if use_openfacts else []
if use_openfacts and not off_candidates:
warnings.append(
"Open Food Facts returned nothing for this brand. Every row below "
"rests on the language model alone - review them individually."
)
# Three distinct outcomes, three distinct warnings. "OFF has nothing under
# this brand tag" is a fact about the brand; "OFF could not be reached" is
# a fact about the network and must not be reported as the former - that
# is how Naga (2 real rows on OFF) was being shown as unknown to it.
off_candidates: List[Dict[str, Any]] = []
if use_openfacts:
try:
off_candidates = _from_open_facts(brand, refresh=refresh_corpus)
except OpenFactsUnavailable as exc:
warnings.append(
f"Open Food Facts could not be reached ({exc}). Nothing was "
"cached, so the next preview will try again; the rows below "
"come from the brand's shop and the language model only."
)
else:
if not off_candidates:
warnings.append(
"Open Food Facts has no products tagged with this brand."
)
off_keys = {}
for candidate in off_candidates:
key = _normalise_title(brand, candidate["title"])
@@ -951,12 +971,26 @@ def discover_brand_products(
if store_candidates:
logger.info("%s: %d products from the brand's own shop", brand, len(store_candidates))
elif use_store and not off_candidates:
# Say what was actually checked. An unregistered storefront is not a
# storefront that "has nothing" - nobody has looked.
store_domain = get_brand_store_domain(brand)
if store_domain:
warnings.append(
f"{store_domain} is registered as this brand's shop but its "
"catalogue has not been fetched - run "
f"scripts/backfill_brand_stores.py --brand \"{brand}\" to load it."
)
else:
warnings.append(
"No verified storefront is registered for this brand. If it "
"has its own online shop, add it to "
"brand_registry.BRAND_STORE_DOMAINS and run "
"scripts/backfill_brand_stores.py to populate a real catalogue."
)
if not off_candidates and not store_candidates:
warnings.append(
"Neither Open Food Facts nor a brand storefront has anything for "
"this brand. Every row below rests on the language model alone - "
"review them individually. If this brand has its own online shop, "
"adding it to brand_registry.BRAND_STORE_DOMAINS and running "
"scripts/backfill_brand_stores.py will populate a real catalogue."
"Every row below rests on the language model alone - review them "
"individually."
)

View File

@@ -11,9 +11,12 @@ never heard of there is nothing to verify, and the catalogue is whatever
`qwen2.5:1.5b` invents.
Measured 2026-09-10, Open Food Facts hits: Udhaiyam 0, Tenali Double Horse 0,
Gopuram 0, Double Horse 14, and Naga 2 - where both "Naga" rows are a UK /
Bangladeshi pickle brand, not the Tamil Nadu one. Regional South Indian brands
are simply not in that database.
Gopuram 0, Double Horse 14. (An earlier note here said Naga's two rows were a
UK / Bangladeshi pickle brand - that was the old free-text `brands:Naga`
Search-a-licious query matching "Mr Naga" and "Bombay Naga Jhal". Under the
exact `brands_tags=naga` filter the v2 API returns two real Naga Limited rows,
Sooji 500 g and Maida 500 g, both on the 890 GS1 prefix - see off_bulk.py.)
Regional South Indian brands are still thin in that database.
They are, however, on their own shop, and modern storefronts publish their
whole catalogue as structured JSON:

View File

@@ -10,34 +10,50 @@ resulting cascade "misses far more often than it hits". That module is wired
into the live enrichment pipeline and is deliberately NOT touched by this file.
This module inverts the problem: fetch a brand's **entire** OFF catalogue in one
or two requests from the Search-a-licious endpoint, cache it on disk, then match
every one of our products against that corpus offline. A whole-catalogue
backfill costs ~5 HTTP requests instead of ~600, and re-tuning the similarity
threshold costs zero network because the corpus is cached.
or two requests from the v2 search API, cache it on disk, then match every one
of our products against that corpus offline. A whole-catalogue backfill costs
~5 HTTP requests instead of ~600, and re-tuning the similarity threshold costs
zero network because the corpus is cached.
Nothing here is imported by the running app - the only consumer is
`scripts/backfill_barcodes_from_off.py`. Everything except `fetch_brand_corpus`
is a pure function so it can be unit-tested without network or database.
`fetch_brand_corpus` is the one network function here and is shared by the
backfill scripts, `product_grounding`, `post_ingest_barcodes` and
`brand_discovery`. Everything else is a pure function so it can be unit-tested
without network or database.
ENDPOINT NOTES (verified empirically, 2026-09)
GET https://search.openfoodfacts.org/search
?q=brands:amul AND countries_tags:"en:india"
ENDPOINT NOTES (verified empirically, 2026-09-11)
GET https://world.openfoodfacts.org/api/v2/search
?brands_tags=amul
&countries_tags=en:india
&fields=code,product_name,quantity,...
&page_size=250
* The double quotes around "en:india" are REQUIRED. Without them the query
parses as a bare term and silently returns count=0 rather than erroring.
* page_size up to 1000 is accepted; 250 keeps responses small.
&page_size=100&page=1
* `brands_tags` is an EXACT match on the brand's tag slug ("naga", not
"Naga"), which is what a brand catalogue needs. The earlier Search-a-licious
query `q=brands:Naga` was a free-text match: it returned "Mr Naga" and
"Bombay Naga Jhal" - other companies - and MISSED the real Naga rows,
because Search-a-licious reads a separate Elasticsearch index that was
stale for them (barcode 8906011830068 was still filed brandless under
Kuwait). The v2 API reads the live product database. Measured on the same
day: Amul 216 products here vs 140 there; Naga 2 vs 0.
* `page_count` in this API is the number of products ON THIS PAGE, not the
number of pages, and `page_size` in the RESPONSE is what the server
actually applied (a larger request is capped to 100 without complaint).
Paginate from `count` / that served `page_size`.
* OFF rate-limits all search endpoints to 10 requests/minute per IP.
PAUSE_SECONDS keeps a multi-page brand under that.
* `world.openfoodfacts.org` intermittently serves an HTML "Page temporarily
unavailable" page with a 200 status, so every response is content-type
checked before parsing.
unavailable" page with a 200 status, and plain 503s under load, so every
response is status- and content-type-checked before parsing, and a failed
fetch is never written to the cache as an empty corpus.
"""
from __future__ import annotations
import json
import logging
import math
import os
import re
import time
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
@@ -63,21 +79,50 @@ from app.services.quantity_utils import quantities_match
logger = logging.getLogger(__name__)
SEARCH_URL = "https://search.openfoodfacts.org/search"
SEARCH_URL = "https://world.openfoodfacts.org/api/v2/search"
# OFF's usage policy requires a contactable custom User-Agent; requests sent
# with the default python-requests agent are treated as anonymous crawling.
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
FIELDS = "code,product_name,product_name_en,brands,quantity,countries_tags"
PAGE_SIZE = 250
MAX_PAGES = 10 # 2500 products per brand is far beyond any real brand
PAUSE_SECONDS = 2.0 # search endpoints are the strictly-limited ones
PAGE_SIZE = 100 # the v2 API caps this at 100 and silently serves that
MAX_PAGES = 25 # 2500 products per brand is far beyond any real brand
PAUSE_SECONDS = 6.0 # 10 search requests/minute is OFF's published limit
# Bumped when the cache file's meaning changes. Files written before this key
# existed came from the Search-a-licious endpoint; an EMPTY one of those is
# more likely to be that endpoint's stale index (or a swallowed 503) than a
# real absence, so it is re-fetched. A non-empty one is real data and is kept.
CACHE_SCHEMA = 2
# backend/app/services/enrichment/barcode/sources/off_bulk.py -> backend/
_BACKEND_DIR = Path(__file__).resolve().parents[5]
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "off_brand_corpus"
class OffUnavailable(requests.exceptions.ConnectionError):
"""OFF answered but could not serve (429 / 5xx).
Subclasses ConnectionError so `retry.with_retry` treats it exactly like a
dropped socket - a few seconds later the same request usually succeeds -
without widening the retry set for every other barcode source.
"""
@dataclass
class CorpusFetch:
"""What `fetch_brand_corpus_result` learned.
`error` is None when OFF answered every page, even if it answered with
nothing - that is a real "this brand is not on Open Food Facts". When
`error` is set the fetch did not complete and NOTHING was cached, so a
caller can say "could not be reached" instead of "has nothing".
"""
hits: List[Dict[str, Any]] = field(default_factory=list)
error: Optional[str] = None
from_cache: bool = False
# Pack sizes embedded in a product name ("Marie Gold 250g", "Butter 1L").
# Mirrors the unit list in scripts/backfill_nutrition_from_barcodes.py, widened
# with the count-based units this catalog also uses.
@@ -122,84 +167,149 @@ def _cache_path(slug: str, cache_dir: Optional[Path] = None) -> Path:
return (cache_dir or CACHE_DIR) / f"{slug}.json"
def brand_tag_slug(brand: str) -> str:
"""The brand as Open Food Facts tags it: lowercase, runs of anything that
is not a letter or digit collapsed to one hyphen. "Hindustan Unilever" ->
"hindustan-unilever", "P&G" -> "p-g", "Naga" -> "naga". OFF matches
`brands_tags` case-insensitively, but sending the slug form is what its
own site does and avoids depending on that."""
return re.sub(r"[^a-z0-9]+", "-", (brand or "").lower()).strip("-")
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
def _get_page(brand: str, country: Optional[str], page: int) -> Dict[str, Any]:
"""One page of OFF search results. Raises on transport errors (retried by
the decorator); returns an empty result dict for any response that is not
parseable JSON, which is how the HTML "temporarily unavailable" page and
any future error page are absorbed without killing the run."""
query = f"brands:{brand}"
def _get_page(brand: str, country: Optional[str], page: int) -> Optional[Dict[str, Any]]:
"""One page of OFF v2 search results.
Raises on transport errors and on 429 / 5xx (both retried by the
decorator); returns None for any other response that is not parseable
JSON - an HTML "temporarily unavailable" page, a 4xx, a truncated body.
None means "this page FAILED", which the caller must keep distinct from a
page that parsed fine and simply held no products.
"""
params: Dict[str, Any] = {
"brands_tags": brand_tag_slug(brand),
"fields": FIELDS,
"page_size": PAGE_SIZE,
"page": page,
}
if country:
# The quotes are load-bearing - see the module docstring.
query += f' AND countries_tags:"en:{country}"'
params["countries_tags"] = f"en:{country}"
resp = requests.get(
SEARCH_URL,
params={"q": query, "fields": FIELDS, "page_size": PAGE_SIZE, "page": page},
params=params,
headers={"User-Agent": USER_AGENT, "Accept": "application/json"},
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
)
if resp.status_code == 429 or resp.status_code >= 500:
raise OffUnavailable(f"HTTP {resp.status_code} from Open Food Facts")
if resp.status_code != 200:
logger.warning("OFF search returned HTTP %s for brand %r page %s",
resp.status_code, brand, page)
return {}
return None
if "json" not in (resp.headers.get("content-type") or "").lower():
logger.warning("OFF search returned non-JSON (%s) for brand %r - "
"the service is probably serving an error page",
resp.headers.get("content-type"), brand)
return {}
return None
try:
payload = resp.json()
except ValueError as e:
logger.warning("OFF search returned unparseable JSON for brand %r: %s", brand, e)
return {}
return payload if isinstance(payload, dict) else {}
return None
return payload if isinstance(payload, dict) else None
def fetch_brand_corpus(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
"""Every OFF product for `brand`, from disk cache unless `refresh`.
def _read_cache(path: Path, brand: str) -> Optional[List[Dict[str, Any]]]:
"""Cached hits, or None when the cache must not be trusted: unreadable, or
an empty file from before CACHE_SCHEMA existed (see that constant)."""
try:
cached = json.loads(path.read_text(encoding="utf-8"))
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
return None
hits = cached.get("hits") or []
if not hits and cached.get("schema") != CACHE_SCHEMA:
logger.info(" Ignoring empty pre-v%s OFF cache for %r - re-fetching",
CACHE_SCHEMA, brand)
return None
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
brand, len(hits), cached.get("fetched_at_human", "?"))
return hits
def fetch_brand_corpus_result(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> CorpusFetch:
"""Every OFF product for `brand`, from disk cache unless `refresh`, plus
whether the fetch actually completed.
Hits with no usable product name are dropped here rather than at match time
(6 of 146 Amul hits, 1 of 233 Britannia hits) - a nameless hit can never
clear a name-similarity threshold, so carrying it forward only inflates the
corpus. Returns [] rather than raising when OFF is unreachable, so one bad
brand does not abort a multi-brand backfill.
corpus.
A fetch that fails part-way returns what it got with `error` set and
writes NO cache file. Caching a failure as `hits: []` is how "Open Food
Facts has nothing for this brand" was being asserted for brands OFF had
simply been too busy to answer about.
"""
country = BARCODE_COUNTRY_TAG if country is None else (country or None)
slug = re.sub(r"[^a-z0-9]+", "_", brand.lower()).strip("_")
path = _cache_path(slug, cache_dir)
if not refresh and path.exists():
try:
cached = json.loads(path.read_text(encoding="utf-8"))
hits = cached.get("hits") or []
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
brand, len(hits), cached.get("fetched_at_human", "?"))
return hits
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
cached = _read_cache(path, brand)
if cached is not None:
return CorpusFetch(hits=cached, from_cache=True)
hits: List[Dict[str, Any]] = []
error: Optional[str] = None
page = 1
while page <= MAX_PAGES:
total_pages = 1
while page <= min(total_pages, MAX_PAGES):
if page > 1:
time.sleep(PAUSE_SECONDS)
payload = _get_page(brand, country, page)
batch = payload.get("hits") or []
try:
payload = _get_page(brand, country, page)
except requests.exceptions.RequestException as e:
# Retries exhausted (OffUnavailable is a ConnectionError too).
error = str(e) or e.__class__.__name__
break
if payload is None:
error = "Open Food Facts returned an unusable response"
break
batch = payload.get("products") or []
hits.extend(h for h in batch if isinstance(h, dict) and (h.get("product_name") or "").strip())
page_count = payload.get("page_count") or 0
if page >= page_count or not batch:
if page == 1:
# `page_count` is products-on-this-page in the v2 API; the real
# page total is `count` over the page size the SERVER applied -
# it silently caps the requested size (250 asked, 100 served for
# Amul), so dividing by PAGE_SIZE under-pages.
count = payload.get("count") or 0
served = payload.get("page_size") or len(batch) or PAGE_SIZE
try:
total_pages = max(1, math.ceil(int(count) / int(served)))
except (TypeError, ValueError, ZeroDivisionError):
total_pages = 1
if not batch:
break
page += 1
if error:
logger.warning(" OFF corpus for %r: fetch failed after %d usable product(s) - %s "
"(nothing cached)", brand, len(hits), error)
return CorpusFetch(hits=hits, error=error)
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(".json.tmp")
tmp.write_text(json.dumps({
"schema": CACHE_SCHEMA,
"endpoint": SEARCH_URL,
"brand": brand,
"brand_tag": brand_tag_slug(brand),
"country": country,
"fetched_at": time.time(),
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
@@ -208,7 +318,20 @@ def fetch_brand_corpus(brand: str,
os.replace(tmp, path)
logger.info(" OFF corpus for %r: %d usable product(s) fetched", brand, len(hits))
return hits
return CorpusFetch(hits=hits)
def fetch_brand_corpus(brand: str,
country: Optional[str] = None,
refresh: bool = False,
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
"""`fetch_brand_corpus_result(...).hits` - the list-only form every
backfill and ingestion caller uses. Returns [] rather than raising when
OFF is unreachable, so one bad brand does not abort a multi-brand run;
callers that need to tell "unreachable" from "empty" use the result form.
"""
return fetch_brand_corpus_result(brand, country=country, refresh=refresh,
cache_dir=cache_dir).hits
# ---------------------------------------------------------------------------

View File

@@ -243,6 +243,16 @@ def _round2(value: Optional[float]) -> Optional[float]:
return round(value, 2)
def _held_number(value: object) -> Optional[float]:
"""A price the row already carries, or None for blank / unparseable."""
if value is None or (isinstance(value, str) and not value.strip()):
return None
try:
return float(value)
except (TypeError, ValueError):
return None
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
"""Compute the pricing fields added by this stage for one catalog row.
@@ -253,8 +263,24 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
derivable from the data this pipeline generates, so they stay None
(exactly as the existing enriched exports store them).
A PRICE THE ROW ALREADY HOLDS WINS. A store sheet that says "Retail Price
155" has stated a fact; the band derived from it is Rs143-167, and this
stage used to read the band's ceiling back as "selling_price = 167" and
hand that to `EnrichmentStage.apply`, which overwrites a held value with
a different one (it only refuses to BLANK one). Held prices are therefore
returned as None here so `apply` keeps them, the base for the tax is the
held selling price, then the held final price, and only then the band.
"""
selling_price = _extract_selling_price(product.get("price_range"))
held_selling = _held_number(product.get("selling_price"))
held_final = _held_number(product.get("final_selling_price"))
if held_selling is not None:
selling_price = held_selling
elif held_final is not None:
selling_price = held_final
else:
selling_price = _extract_selling_price(product.get("price_range"))
gst = product.get("gst_percent")
tax_amount = None
final_selling_price = None
@@ -263,10 +289,10 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
final_selling_price = _round2(selling_price + tax_amount)
return {
"selling_price": _round2(selling_price),
"selling_price": None if held_selling is not None else _round2(selling_price),
"cost_price": None,
"tax_amount": tax_amount,
"final_selling_price": final_selling_price,
"final_selling_price": None if held_final is not None else final_selling_price,
"profit_before_tax": None,
"profit_after_tax": None,
}

View File

@@ -76,7 +76,7 @@ logger = logging.getLogger(__name__)
# Matches scripts/backfill_barcodes_from_off.py, which measured it.
DEFAULT_MIN_SIMILARITY = 0.88
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
BARCODE_SOURCE = "openfoodfacts_bulk (world.openfoodfacts.org/api/v2)"
def _rows_needing_a_barcode(cur, table: str) -> List[Dict[str, Any]]:

View File

@@ -63,7 +63,7 @@ import sqlite3
import threading
import time
from contextlib import closing
from dataclasses import dataclass
from dataclasses import dataclass, field
from pathlib import Path
from typing import Iterable, List, Optional, Sequence
from urllib.parse import urlparse
@@ -175,6 +175,23 @@ def distinctive_tokens(product_name: str, brand: str) -> List[str]:
return out
def _token_in(token: str, lowered_url: str) -> bool:
"""Does the URL carry this word, allowing for the plural on either side?
The catalogue names "Aachi Appalams" and "Aachi Pickles"; the photo is
`Aachi-Appalam-100-g-1.webp`. A whole-token substring test rejected the
correct image for the plural and then promoted an opaque Amazon URL over
it. The singular is tried as well - only for tokens long enough that
stripping the "s" leaves a real word ("gems" -> "gem" is fine; "kgs" never
gets here, size tokens are removed upstream).
"""
if token in lowered_url:
return True
if len(token) > 4 and token.endswith("s") and token[:-1] in lowered_url:
return True
return False
def search_key(product_name: str, brand: str) -> tuple:
"""Collapse a trailing pack size so sizes of one product share a lookup.
@@ -213,7 +230,7 @@ def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
distinctive = distinctive_tokens(product_name, brand)
if distinctive:
hit = next((t for t in distinctive if t in lowered), None)
hit = next((t for t in distinctive if _token_in(t, lowered)), None)
if hit:
return Corroboration(True, f"url names {hit!r}")
return Corroboration(
@@ -272,6 +289,26 @@ _lock = threading.Lock()
_initialized = False
_CACHE_TTL_SECONDS = 30 * 24 * 3600
# Open*Facts allows 100 product reads a minute per IP. A brand ingestion or a
# re-gate asks about every distinct barcode it meets, and the Dabur run issued
# about a hundred in two minutes - the tail was throttled, and a throttled
# lookup fails open, which is how a Kellogg's honey got past the gate for a
# moment. Pacing the calls keeps the gate answering instead of guessing.
_OFF_MIN_INTERVAL_SECONDS = 0.65
_OFF_LOOKUP_ATTEMPTS = 2
_OFF_RETRY_SLEEP_SECONDS = 3.0
_off_last_call = 0.0
_off_pace_lock = threading.Lock()
def _pace_off_lookup() -> None:
global _off_last_call
with _off_pace_lock:
wait = _OFF_MIN_INTERVAL_SECONDS - (time.monotonic() - _off_last_call)
if wait > 0:
time.sleep(wait)
_off_last_call = time.monotonic()
def _connect() -> sqlite3.Connection:
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
@@ -366,29 +403,58 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
if not tokens:
return True
key = f"{api_host}:{barcode}:{(brand or '').lower()}"
# "v2:" - entries written before the fail-open verdicts stopped being
# cached are ignored rather than trusted; they age out with the TTL.
key = f"v2:{api_host}:{barcode}:{(brand or '').lower()}"
cached = _cache_get(key)
if cached is not None:
return cached
verdict = True
try:
resp = requests.get(
f"https://{api_host}/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=timeout,
headers={"User-Agent": _BROWSER_UA},
)
# ONLY A REAL ANSWER IS CACHED. The fail-open True for a lookup that did
# not happen - a 429 or 503 from Open*Facts, a non-JSON body - used to be
# written to the cache too, for thirty days. A Dabur ingestion issued a
# hundred lookups in two minutes, the tail of them were throttled, and
# Kellogg's "Miel Pops", a Toblerone and a Nature Valley bar were filed as
# Dabur products; "Dabur Honey 1kg" then showed the Kellogg's honey. A
# throttled lookup still fails open for THIS call, but the next call asks
# again.
payload = None
for attempt in range(_OFF_LOOKUP_ATTEMPTS):
_pace_off_lookup()
try:
resp = requests.get(
f"https://{api_host}/api/v2/product/{barcode}.json",
params={"fields": "brands,product_name"},
timeout=timeout,
headers={"User-Agent": _BROWSER_UA},
)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
return True
if resp.ok:
product = (resp.json() or {}).get("product") or {}
haystack = (
f"{product.get('brands') or ''} {product.get('product_name') or ''}"
).lower()
if haystack.strip():
verdict = any(t in haystack for t in tokens)
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
try:
payload = resp.json() or {}
except ValueError:
return True
break
if resp.status_code == 429 or resp.status_code >= 500:
# Throttled or unwell. One paced retry is cheap and turns most of
# these into a real answer; a second failure fails open, uncached.
time.sleep(_OFF_RETRY_SLEEP_SECONDS * (attempt + 1))
continue
return True
if payload is None:
return True
product = payload.get("product") or {}
if not product and payload.get("status") == 0:
# Open*Facts positively says: no such barcode. Nothing to compare
# against, and asking again will not change that - cache the open verdict.
_cache_set(key, True)
return True
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
if not haystack.strip():
return True
verdict = any(t in haystack for t in tokens)
_cache_set(key, verdict)
return verdict
@@ -398,11 +464,25 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
# ---------------------------------------------------------------------------
@dataclass
class PrimaryChoice:
"""The outcome of choosing a product's `image_url` from its candidates."""
"""The outcome of choosing a product's `image_url` from its candidates.
`ordered` is every candidate, best first. `eligible` is the subset that
may be SHOWN - tiers 1 and 2 - and is what the store pipeline persists as
`image_urls`. The two differ for a reason that was learned the hard way:
tier 3 was kept in `image_urls` "for review, never promoted", but the
product card falls back to `image_urls[0]` whenever `image_url` is empty
and the product modal shows the whole list as a gallery. So for "Dabur
Honey 1kg" the gate correctly withheld the primary - every candidate was
another company's honey - and the UI displayed those very honeys anyway.
A candidate the gate would not promote must not be stored where the UI
will promote it.
"""
primary: Optional[str]
ordered: List[str]
reason: str
eligible: List[str] = field(default_factory=list)
rejected: List[str] = field(default_factory=list)
def choose_primary(
@@ -424,7 +504,10 @@ def choose_primary(
catalog_engine's comment protects, where dropping uncorroborated
candidates would leave real products with no image at all.
3. The URL could have named the product and did not - a human-authored
Commons filename, say. Kept in `image_urls`, never promoted.
Commons filename, say - or an Open*Facts photo whose barcode belongs
to another brand. Returned in `ordered` and `rejected`, never in
`eligible`, and the store pipeline does not persist it (see
PrimaryChoice for why "kept for review" was not safe).
Nothing eligible means `primary is None`. A blank image renders as the
brand monogram, which is honest; another company's product is not, and it
@@ -466,12 +549,16 @@ def choose_primary(
tier3.sort(key=looks_like_person_photo)
ordered = tier1 + tier2 + tier3
eligible = tier1 + tier2
if tier1:
return PrimaryChoice(tier1[0], ordered, "url names the product")
return PrimaryChoice(tier1[0], ordered, "url names the product",
eligible=eligible, rejected=tier3)
if tier2:
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host")
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host",
eligible=eligible, rejected=tier3)
return PrimaryChoice(
None,
ordered,
"no candidate names this product; primary withheld rather than guessed",
eligible=eligible, rejected=tier3,
)

View File

@@ -114,11 +114,44 @@ def _query_openfacts(query: str, max_results: int) -> list:
return []
def _openfacts_product_is_brand(product: dict, brand: Optional[str]) -> bool:
"""Does this Open*Facts record belong to `brand`?
Open*Facts' free-text search matches on the NAME, not the brand: asked for
"Dabur Honey 1kg" it answered with a UK, a French, a Swiss and a Spanish
honey, every one a real front-of-pack photo of somebody else's product.
The record says who made it - `brands` / `brands_tags` - and that field
used to be thrown away here, leaving a per-URL API round trip downstream
(`image_corroboration.openfacts_product_matches_brand`) as the only thing
standing between those photos and the catalogue. Check it at the source.
Fails OPEN on a record with no brand at all: an unlabelled record is not
evidence of another brand, and the downstream gate still runs.
"""
if not brand:
return True
from app.services.image_corroboration import brand_tokens
tokens = brand_tokens(brand)
if not tokens:
return True
tags = product.get("brands_tags") or []
haystack = " ".join([str(product.get("brands") or "")] + [str(t) for t in tags]).lower()
if not haystack.strip():
return True
return any(t in haystack for t in tokens)
def find_images_openfacts(title: str, brand: Optional[str] = None, max_results: int = 20) -> list:
"""Query the Open *Facts family of open product databases for real
product photos. No API key required. Falls back from a brand+title
query to a title-only query if the combined query is too specific to
match anything (small/regional brand name variants are a common case)."""
match anything (small/regional brand name variants are a common case).
Only records whose own `brands` field names our brand are used - see
`_openfacts_product_is_brand`. The title-only fallback makes this filter
load-bearing: "Honey 1kg" matches every honey on the site.
"""
if not USE_OPEN_FACTS:
return []
@@ -126,12 +159,12 @@ def find_images_openfacts(title: str, brand: Optional[str] = None, max_results:
if not query:
return []
products = _query_openfacts(query, max_results)
products = [p for p in _query_openfacts(query, max_results) if _openfacts_product_is_brand(p, brand)]
if not products and brand and title:
# Combined "brand + title" query found nothing - retry with just
# the title, since Open*Facts' free-text search is exact-ish and
# brand naming conventions vary (e.g. "Dettol" vs "Reckitt Dettol").
products = _query_openfacts(title, max_results)
products = [p for p in _query_openfacts(title, max_results) if _openfacts_product_is_brand(p, brand)]
urls: List[str] = []
for product in products:

View File

@@ -332,16 +332,29 @@ def fetch_brand_catalog_exhaustive(brand: str, max_products: int = 300) -> Dict[
return {"brand": brand, "products": products_out}
def fetch_product_details(brand: str, product_title: str) -> Dict[str, Any] | None:
def fetch_product_details(brand: str, product_title: str,
category: str | None = None,
size: str | None = None) -> Dict[str, Any] | None:
"""Get details for a single product: description, image_url, pricing fields.
Returns a dict with keys: description, image_url, size_variants, price_ranges, price_range, provider_examples.
`category` and `size` are optional context. A 1.5b model asked about
"Naga Maida" alone will happily describe a curry; told it is a 500g pack
in Flours & Grains it describes refined wheat flour. Both are facts the
caller already holds, so they cost nothing to pass.
"""
if not _ensure_client():
return None
context = f"Brand: {brand}\nProduct: {product_title}\n"
if category:
context += f"Category: {category}\n"
if size:
context += f"Pack size: {size}\n"
user_prompt = (
f"Brand: {brand}\nProduct: {product_title}\n"
"Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
"description must be <=160 chars, concise and factual."
context
+ "Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
"description: 1-2 sentences, <=220 chars, factual - what the product is, "
"its form and its typical use. No marketing adjectives, no claims you cannot know."
)
text = _generate(SYSTEM_PROMPT, user_prompt)
if not text:

View File

@@ -5,7 +5,7 @@ WHY THIS FILE EXISTS
--------------------
Open Food Facts answers "does this product exist" for food, and answers it
well. It does not answer it for anything else: `off_bulk` queries only
`search.openfoodfacts.org`, so a toothpaste or a detergent has no
Open Food Facts' food database, so a toothpaste or a detergent has no
product-discovery source at all and its entire catalogue is language-model
output. Measured 2026-09-10, India-tagged coverage on the sibling databases is
2-13 products per non-food brand against Britannia's 218 on OFF - real, but

File diff suppressed because it is too large Load Diff

Binary file not shown.

File diff suppressed because it is too large Load Diff

31
data/cache/off_brand_corpus/naga.json vendored Normal file
View File

@@ -0,0 +1,31 @@
{
"schema": 2,
"endpoint": "https://world.openfoodfacts.org/api/v2/search",
"brand": "Naga",
"brand_tag": "naga",
"country": "india",
"fetched_at": 1789104574.7944398,
"fetched_at_human": "2026-09-11 10:59:34",
"hits": [
{
"brands": "Naga",
"code": "8906011830068",
"countries_tags": [
"en:india"
],
"product_name": "Sooji",
"product_name_en": "Sooji",
"quantity": "500 g"
},
{
"brands": "Naga",
"code": "8906011831713",
"countries_tags": [
"en:india"
],
"product_name": "Maida",
"product_name_en": "Maida",
"quantity": "500 g"
}
]
}

View File

@@ -0,0 +1,10 @@
{
"schema": 2,
"endpoint": "https://world.openfoodfacts.org/api/v2/search",
"brand": "TestBrand",
"brand_tag": "testbrand",
"country": "india",
"fetched_at": 1789110182.5709507,
"fetched_at_human": "2026-09-11 12:33:02",
"hits": []
}

View File

@@ -98,7 +98,7 @@ from app.services.vector_store import _connect
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("backfill_barcodes_off")
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
BARCODE_SOURCE = "openfoodfacts_bulk (world.openfoodfacts.org/api/v2)"
# The nine keys BarcodeResult.as_product_fields() emits. Named here so the JSON
# writer can prove it touches nothing else.

View File

@@ -30,7 +30,10 @@ USAGE
python scripts/backfill_brand_stores.py --all # dry run
python scripts/backfill_brand_stores.py --all --apply
python scripts/backfill_brand_stores.py --brand Gopuram --apply --refresh
python scripts/backfill_brand_stores.py --brand Naga --domain nagafoods.in --apply
python scripts/backfill_brand_stores.py --brand Aachi --domain aachifoods.com --apply
(Naga has no confirmed storefront to try: nagalimited.com is a parked domain
and nagafoods.in / nagafoods.com do not resolve, checked 2026-09-11.)
"""
from __future__ import annotations

View File

@@ -0,0 +1,212 @@
#!/usr/bin/env python3
"""
Re-run the image gate over a brand's STORED images, and drop what fails it.
WHY THIS EXISTS
---------------
`scripts/repair_brand_images.py` fixes rows whose image URLs are DEAD. This
one fixes rows whose image URLs are ALIVE AND WRONG, which the repair script
refuses to touch by design ("never rewrites a row that already has one working
URL").
The Dabur run of 2026-09-11 left exactly that: "Dabur Honey 1kg / 250g / 50g"
with an empty `image_url` and four foreign honeys in `image_urls` - which the
product card shows as soon as the primary is blank - and "Dabur Honey 500g"
with a Dabur Odomos mosquito repellent as its primary. The ingestion path no
longer stores rejected candidates (`store_catalog_pipeline.stage_6_images`),
but rows written before that keep them, and a re-ingest does not revisit a
row that already has images. Hence this script.
WHAT IT DOES, PER ROW
---------------------
1. Runs `image_corroboration.choose_primary` over every URL the row holds
(primary first). `image_url` becomes the eligible primary, `image_urls`
the eligible list; rejected URLs are dropped.
2. If nothing is eligible, looks for another pack size of the same product
(same `search_key`) whose primary passed, and adopts it - through the gate
for THIS title, exactly as stage 6 does.
3. If still nothing, `image_url` becomes NULL and `image_urls` empty. The card
renders the brand monogram, which is honest; a foreign honey is not.
Own Products is skipped: its images are curated (`produce_reference`) and the
gate was never meant for a bucket without a brand.
USAGE
-----
python -m scripts.regate_brand_images --brands Dabur # dry run
python -m scripts.regate_brand_images --brands Dabur --apply
python -m scripts.regate_brand_images --all # dry run, every brand
`--apply` writes a full backup of each targeted table first, in the same
format `repair_brand_images.py --restore` reads, so the change is reversible:
python -m scripts.repair_brand_images --restore data/brand_image_repair_backup_X.json --apply
The Open*Facts barcode cross-check is a network call per distinct barcode
(cached 30 days in data/cache/image_corroboration.db). Everything else is
offline.
"""
from __future__ import annotations
import argparse
import logging
import sys
from pathlib import Path
from typing import Any, Dict, List, Optional
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
from app.services import image_corroboration as ic # noqa: E402
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
from app.services.vector_store import ( # noqa: E402
_connect,
_sanitize_name,
invalidate_brand_overview_cache,
)
from scripts.repair_brand_images import ( # noqa: E402
_backup,
_brand_tables,
_read_rows,
)
logging.basicConfig(level=logging.INFO, format="%(message)s")
logger = logging.getLogger("regate_brand_images")
def _row_urls(row: Dict[str, Any]) -> List[str]:
"""Primary first, then the rest, deduplicated, http only."""
out: List[str] = []
for url in [row.get("image_url")] + list(row.get("image_urls") or []):
if url and str(url).startswith("http") and url not in out:
out.append(str(url))
return out
def regate_rows(rows: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]:
"""The new (image_url, image_urls) for every row, plus why.
Pure apart from the Open*Facts barcode cross-check inside choose_primary,
so it can be unit-tested with that patched.
"""
verdicts: List[Dict[str, Any]] = []
for row in rows:
urls = _row_urls(row)
name = row.get("product_name") or ""
if not urls:
verdicts.append({**row, "new_url": None, "new_urls": [], "why": "no images stored"})
continue
choice = ic.choose_primary(urls, name, brand)
verdicts.append({
**row,
"new_url": choice.primary,
"new_urls": list(choice.eligible),
"rejected": list(choice.rejected),
"why": choice.reason,
})
# Sibling adoption for rows left with nothing, from rows that kept a primary.
eligible_by_key: Dict[tuple, str] = {}
for v in verdicts:
if v["new_url"]:
eligible_by_key.setdefault(ic.search_key(v.get("product_name") or "", brand), v["new_url"])
for v in verdicts:
if v["new_url"]:
continue
candidate = eligible_by_key.get(ic.search_key(v.get("product_name") or "", brand))
if not candidate:
continue
adopted = ic.choose_primary([candidate], v.get("product_name") or "", brand)
if adopted.primary:
v["new_url"] = adopted.primary
v["new_urls"] = [adopted.primary]
v["why"] = "adopted from another pack size of this product"
return verdicts
def _changed(v: Dict[str, Any]) -> bool:
return (v.get("image_url") or None) != v["new_url"] or list(v.get("image_urls") or []) != v["new_urls"]
def regate(cur, conn, brands: Optional[List[str]], apply: bool) -> Dict[str, int]:
totals = {"rows": 0, "changed": 0, "blanked": 0, "adopted": 0}
snapshot: Dict[str, List[Dict[str, Any]]] = {}
plans: Dict[str, List[Dict[str, Any]]] = {}
for suffix, table in _brand_tables(cur, brands):
if suffix == _sanitize_name(OWN_PRODUCTS_BRAND):
logger.info("%s: skipped (curated images)", table)
continue
rows = _read_rows(cur, table)
if not rows:
continue
snapshot[table] = rows
verdicts = [v for v in regate_rows(rows, suffix) if _changed(v)]
plans[table] = verdicts
totals["rows"] += len(rows)
logger.info("%s: %d row(s), %d to change", table, len(rows), len(verdicts))
for v in verdicts:
arrow = "->"
logger.info(" %-42s %s %s", (v.get("product_name") or "")[:42], arrow,
(v["new_url"] or "(none)")[:80])
logger.info(" was %s (+%d more) %s", (v.get("image_url") or "(none)")[:70],
max(0, len(v.get("image_urls") or []) - 1), v["why"])
for dropped in v.get("rejected") or []:
logger.info(" drop %s", dropped[:90])
totals["changed"] += 1
if v["new_url"] is None:
totals["blanked"] += 1
if v["why"].startswith("adopted"):
totals["adopted"] += 1
if apply and any(plans.values()):
logger.info("Backup written: %s", _backup(snapshot))
for table, verdicts in plans.items():
for v in verdicts:
cur.execute(
f"UPDATE {table} SET image_url = %s, image_urls = %s, "
f"updated_at = CURRENT_TIMESTAMP WHERE image_id = %s",
(v["new_url"], v["new_urls"], v["image_id"]),
)
conn.commit()
try:
invalidate_brand_overview_cache()
except Exception: # noqa: BLE001 - the cache is a convenience
pass
return totals
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
mode = parser.add_mutually_exclusive_group(required=True)
mode.add_argument("--brands", help="comma-separated brands to re-gate")
mode.add_argument("--all", action="store_true", help="every brand table")
parser.add_argument("--apply", action="store_true", help="write the changes (default: dry run)")
args = parser.parse_args()
logger.info("Target database: %s / %s", DB_HOST, DB_NAME)
logger.info("Mode: %s", "APPLY - this writes" if args.apply else "DRY RUN - nothing is written")
conn = _connect()
if conn is None:
logger.error("No database connection.")
return 1
try:
with conn.cursor() as cur:
brands = None if args.all else [b for b in args.brands.split(",") if b.strip()]
totals = regate(cur, conn, brands, args.apply)
finally:
conn.close()
logger.info("")
logger.info("%s: %d row(s) read, %d %s, %d blanked, %d adopted from a sibling size",
"Applied" if args.apply else "Dry run", totals["rows"], totals["changed"],
"changed" if args.apply else "would change", totals["blanked"], totals["adopted"])
if args.apply:
logger.info("The API caches the brand overview; hit GET /api/brands/overview?refresh=true "
"or wait out BRAND_OVERVIEW_TTL_SECONDS.")
return 0
if __name__ == "__main__":
sys.exit(main())

View File

@@ -55,6 +55,11 @@ os.environ.setdefault("USE_PGVECTOR", "true")
os.environ.setdefault("DB_PASSWORD", "test-password-not-real")
os.environ.setdefault("USE_S3", "false")
os.environ.setdefault("USE_GOOGLE_CSE", "false")
# Unconditional: the LLM description stage is on by default for every batch
# and a developer machine often has `ollama serve` running, so without this a
# pipeline test would make real, slow, non-deterministic model calls. Tests
# that exercise the probe itself monkeypatch `ollama_service.USE_OLLAMA`.
os.environ["USE_OLLAMA"] = "false"
# Unconditional, NOT setdefault. The suite pins brand-name behaviour all over
# the place ("any Nestle chocolates?", the suggest ranking fixtures), and a

View File

@@ -5,8 +5,8 @@ boundary *on the pipeline module object*, build real spreadsheets in memory,
and point the batch directory at tmp_path so nothing is written into the repo.
Nothing here loads sentence-transformers or torch, and nothing reaches the
network: every batch runs with use_llm=False and fetch_images=False, which are
also the defaults the endpoint ships.
network: every batch runs with fetch_images=False (the endpoint's default) and
the LLM stage, on by default, finds Ollama disabled and falls back at once.
"""
from __future__ import annotations
@@ -486,7 +486,7 @@ def test_ingest_returns_202_and_a_batch_id(client, admin_headers, store, batch_r
assert response.status_code == 202
body = response.json()
assert body["files_total"] == 2
assert body["use_llm"] is False, "the LLM stage must be opt-in for a batch"
assert body["use_llm"] is True, "LLM descriptions are on by default; a batch opts OUT"
assert body["fetch_images"] is False, "image search must be opt-in for a batch"
assert len(body["files"]) == 2
assert body["files"][0]["total_stages"] == pipeline.TOTAL_STAGES

View File

@@ -21,6 +21,9 @@ from app.api.routers.user_products import map_spreadsheet_columns, row_to_reques
from app.core import store_catalog_pipeline as pipeline
from app.services import brand_discovery as bd
# Captured before the autouse `_no_network` fixture replaces it on the module.
_REAL_FROM_OPEN_FACTS = bd._from_open_facts
# ---------------------------------------------------------------------------
# Fixtures
@@ -220,12 +223,11 @@ def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch):
# ---------------------------------------------------------------------------
# Regression: one GTIN, one pack
# ---------------------------------------------------------------------------
def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
"""A GTIN identifies one pack, and stage 4 invents three when it has none.
Every column is copied into each exploded variant, so a surviving barcode
would be stamped onto two packs that do not exist - wrong data that looks
authoritative.
def test_a_barcode_survives_when_no_pack_size_is_known(monkeypatch):
"""A GTIN identifies one pack, and stage 4 now stores exactly one row for
a product with no size (it used to invent three and stamp the same GTIN
on all of them, which is why the barcode was dropped before). One GTIN on
one unsized row is what a GTIN means; the row is only flagged for review.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", code="8901063017702"),
@@ -234,8 +236,8 @@ def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.size_variants == []
assert product.barcode is None
assert any("barcode dropped" in note for note in product.notes)
assert product.barcode == "8901063017702"
assert any("pack size unknown" in note for note in product.notes)
def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch):
@@ -549,3 +551,113 @@ def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch):
assert row.get(column), f"{column} was left empty"
assert row["barcode"] == "8901063012516"
assert row["fssai_license"] == "10012022000103"
# ---------------------------------------------------------------------------
# What the warnings claim - "unreachable" vs "empty" vs "no storefront"
# ---------------------------------------------------------------------------
# Naga (2026-09-11): Open Food Facts holds two real rows for the brand, yet the
# tab said "Neither Open Food Facts nor a brand storefront has anything for
# this brand". The lookup had failed / been mis-indexed, no storefront was
# registered, and one warning text covered all of it. Each claim now has to
# be true on its own.
def _store_rows(monkeypatch, rows):
monkeypatch.setattr(bd, "_from_brand_store", lambda brand: list(rows))
def test_an_unreachable_off_is_reported_as_unreachable_not_empty(monkeypatch):
def down(brand, refresh=False):
raise bd.OpenFactsUnavailable("HTTP 503 from Open Food Facts")
monkeypatch.setattr(bd, "_from_open_facts", down)
_store_rows(monkeypatch, [])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Naga Sooji", "category": None, "description": None,
"sizes": ["500g"], "providers": [], "source": "llm"},
])
result = bd.discover_brand_products("Naga", require_evidence=False)
reached = [w for w in result.warnings if "could not be reached" in w]
assert len(reached) == 1 and "503" in reached[0]
assert not any("has no products tagged" in w for w in result.warnings)
assert [p.title for p in result.products] == ["Naga Sooji"], "the LLM rows must still come through when OFF is down"
def test_a_genuinely_empty_off_says_so_without_blaming_the_network(monkeypatch):
_store_rows(monkeypatch, [])
result = bd.discover_brand_products("Udhaiyam")
assert any("has no products tagged" in w for w in result.warnings)
assert not any("could not be reached" in w for w in result.warnings)
def test_unregistered_storefront_is_not_described_as_empty(monkeypatch):
_store_rows(monkeypatch, [])
monkeypatch.setattr(bd, "get_brand_store_domain", lambda brand: None)
result = bd.discover_brand_products("Naga")
assert any("No verified storefront is registered" in w for w in result.warnings)
assert not any("Neither Open Food Facts" in w for w in result.warnings)
assert not any("has not been fetched" in w for w in result.warnings)
def test_registered_but_unfetched_storefront_names_the_backfill(monkeypatch):
_store_rows(monkeypatch, [])
monkeypatch.setattr(bd, "get_brand_store_domain", lambda brand: "gopuramproducts.com")
result = bd.discover_brand_products("Gopuram")
hit = [w for w in result.warnings if "has not been fetched" in w]
assert len(hit) == 1
assert "gopuramproducts.com" in hit[0] and "backfill_brand_stores.py" in hit[0]
assert not any("No verified storefront" in w for w in result.warnings)
def test_the_llm_only_caveat_appears_exactly_once(monkeypatch):
_store_rows(monkeypatch, [])
result = bd.discover_brand_products("Naga")
assert sum("rests on the language model alone" in w for w in result.warnings) == 1
def test_no_storefront_warning_when_the_shop_or_off_has_rows(monkeypatch):
_store_rows(monkeypatch, [{"title": "Naga Maida 1kg", "source": "store", "size": "1kg"}])
result = bd.discover_brand_products("Naga")
assert not any("storefront" in w.lower() for w in result.warnings)
assert not any("rests on the language model alone" in w for w in result.warnings)
_store_rows(monkeypatch, [])
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Sooji", code="8906011830068", quantity="500 g"),
])
result = bd.discover_brand_products("Naga")
assert not any("storefront" in w.lower() for w in result.warnings)
assert not any("rests on the language model alone" in w for w in result.warnings)
def test_from_open_facts_raises_when_the_fetch_failed_with_nothing(monkeypatch):
from app.services.enrichment.barcode.sources import off_bulk
monkeypatch.setattr(off_bulk, "fetch_brand_corpus_result",
lambda brand, refresh=False: off_bulk.CorpusFetch(hits=[], error="dns"))
with pytest.raises(bd.OpenFactsUnavailable):
_REAL_FROM_OPEN_FACTS("Naga")
def test_from_open_facts_maps_v2_hits_to_candidates(monkeypatch):
from app.services.enrichment.barcode.sources import off_bulk
monkeypatch.setattr(off_bulk, "fetch_brand_corpus_result",
lambda brand, refresh=False: off_bulk.CorpusFetch(hits=[
{"code": "8906011830068", "product_name": "Sooji",
"product_name_en": "Sooji", "quantity": "500 g"},
{"code": "8906011831713", "product_name": "Maida", "quantity": "500 g"},
]))
rows = _REAL_FROM_OPEN_FACTS("Naga")
assert [(r["title"], r["barcode"], r["source"]) for r in rows] == [
("Sooji", "8906011830068", "off"), ("Maida", "8906011831713", "off"),
]

View File

@@ -174,6 +174,94 @@ def test_a_lookup_failure_never_rejects_an_image(monkeypatch):
assert ic.openfacts_product_matches_brand(url, "Anil") is True
class _Resp:
def __init__(self, ok=True, payload=None):
self.ok = ok
self.status_code = 200 if ok else 503
self._payload = payload or {}
def json(self):
return self._payload
@pytest.fixture
def verdict_cache(tmp_path, monkeypatch):
"""A throwaway sqlite cache, so these tests neither read nor poison the
real one in data/cache."""
monkeypatch.setattr(ic, "_DB_PATH", tmp_path / "verdicts.db")
monkeypatch.setattr(ic, "_initialized", False) # the schema is per file
monkeypatch.setattr(ic.time, "sleep", lambda s: None) # pacing/retry waits
return tmp_path / "verdicts.db"
KELLOGGS_HONEY = "https://images.openfoodfacts.org/images/products/505/931/902/3762/front_fr.32.400.jpg"
def test_a_throttled_lookup_fails_open_but_is_not_remembered(monkeypatch, verdict_cache):
"""The Dabur Honey defect. A hundred lookups in two minutes got the tail
of them a 503; the fail-open True was cached for thirty days, and
Kellogg's "Miel Pops" became a Dabur product for a month."""
throttled = _Resp(ok=False)
throttled.status_code = 503
answers = iter([throttled, throttled, _Resp(payload={"status": 1, "product": {
"brands": "KELLOG'S", "product_name": "Miel Pops"}})])
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: next(answers))
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is True, "throttled on both attempts: fails open now"
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False, "asks again next time"
def test_one_throttled_attempt_is_retried_into_a_real_answer(monkeypatch, verdict_cache):
throttled = _Resp(ok=False)
throttled.status_code = 429
answers = iter([throttled, _Resp(payload={"status": 1, "product": {
"brands": "KELLOG'S", "product_name": "Miel Pops"}})])
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: next(answers))
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False
def test_a_real_verdict_is_cached(monkeypatch, verdict_cache):
calls = []
def get(*a, **k):
calls.append(1)
return _Resp(payload={"status": 1, "product": {"brands": "Toblerone",
"product_name": "Milk Chocolate"}})
monkeypatch.setattr(ic.requests, "get", get)
url = "https://images.openfoodfacts.org/images/products/761/450/001/0013/front_en.362.400.jpg"
assert ic.openfacts_product_matches_brand(url, "Dabur") is False
assert ic.openfacts_product_matches_brand(url, "Dabur") is False
assert len(calls) == 1
def test_an_unknown_barcode_is_cached_as_open(monkeypatch, verdict_cache):
calls = []
def get(*a, **k):
calls.append(1)
return _Resp(payload={"status": 0, "status_verbose": "product not found"})
monkeypatch.setattr(ic.requests, "get", get)
url = "https://images.openfoodfacts.org/images/products/000/000/000/0001/front.jpg"
assert ic.openfacts_product_matches_brand(url, "Dabur") is True
assert ic.openfacts_product_matches_brand(url, "Dabur") is True
assert len(calls) == 1
def test_pre_fix_cache_entries_are_ignored(monkeypatch, verdict_cache):
"""Entries written by the old code (no "v2:" prefix) may be fail-open
Trues; they must not be trusted."""
ic._cache_set("world.openfoodfacts.org:5059319023762:dabur", True)
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: _Resp(payload={
"status": 1, "product": {"brands": "KELLOG'S", "product_name": "Miel Pops"}}))
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False
def test_a_non_openfacts_url_is_not_cross_checked():
assert ic.openfacts_product_matches_brand(REAL_RAVA, "Anil") is True
@@ -193,3 +281,20 @@ def test_pack_sizes_of_one_product_share_a_search_key():
def test_different_products_do_not_share_a_search_key():
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
!= ic.search_key("Anil Samba Rava 500g", "Anil"))
# ---------------------------------------------------------------------------
# Plural titles, singular filenames
# ---------------------------------------------------------------------------
def test_a_plural_title_is_named_by_a_singular_filename():
"""The all-brands re-gate would have dropped the correct
`Aachi-Appalam-100-g-1.webp` from "Aachi Appalams 500g" and promoted an
opaque Amazon URL over it."""
url = "https://thedesifood.com/media/Aachi-Appalam-100-g-1.webp"
assert ic.corroborate(url, "Aachi Appalams 500g", "Aachi").corroborated
assert ic.corroborate("https://x.in/img/aachi-pickle-jar.jpg", "Aachi Pickles 200g", "Aachi").corroborated
def test_the_singular_fallback_does_not_widen_short_tokens():
# "gems" -> "gem" is too short a stem to trust; four letters is the floor.
assert not ic.corroborate("https://x.in/gem-ring.jpg", "Cadbury Gems 20g", "Cadbury").corroborated

View File

@@ -0,0 +1,297 @@
"""The Dabur Honey regression: one product, several pack sizes, one photo.
Brand Discovery for Dabur (2026-09-11) stored, for the same product:
Dabur Honey 225g Amazon photo of Dabur Honey correct
Dabur Honey 1kg image_url empty; image_urls = 4 honeys from the UK,
Dabur Honey 250g France, Switzerland and Spain (Open Food Facts photos
Dabur Honey 50g whose barcodes belong to other brands)
Dabur Honey 500g a Dabur Odomos mosquito repellent
Three causes, each pinned below:
1. Stage 6 searched once PER SIZE. The query was identical - the size is not
in it - but the providers are live and not deterministic, so four asks
came back four ways. One search per product, shared by every size.
2. `find_images_openfacts` threw away the `brands` field of the records it
fetched, so a title-only fallback ("Honey 1kg") handed over every honey on
the site. The brand is checked at the source now.
3. `choose_primary` kept rejected candidates in `image_urls` "for review" -
and the product card falls back to `image_urls[0]` when the primary is
withheld, so the review pile was what the shopper saw. Only eligible
candidates are stored.
Everything runs offline: the search function and the Open*Facts barcode
cross-check are monkeypatched; storage and embeddings are stubbed.
"""
from __future__ import annotations
import io
import pytest
from app.core import store_catalog_pipeline as pipeline
from app.services import image_corroboration as ic
from app.services import image_search
openpyxl = pytest.importorskip("openpyxl")
AMAZON = "https://m.media-amazon.com/images/I/71O4OnjaHVL.jpg"
ODOMOS = ("https://www.indianproductsstore.com/uploads/products/"
"dabur-odomos-naturals-mosquito-repellent-gel.jpg")
FOREIGN_HONEYS = [
"https://images.openfoodfacts.org/images/products/505/931/902/3762/front_fr.32.400.jpg",
"https://images.openfoodfacts.org/images/products/308/854/000/4440/front_fr.129.400.jpg",
]
DABUR_OFF = "https://images.openfoodfacts.org/images/products/890/120/702/6553/front_en.4.400.jpg"
NAMED = "https://cdn.example.in/dabur-honey-squeezy-bottle.jpg"
def _sheet(headers, rows) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(headers)
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def _off_barcodes_offline(monkeypatch):
"""Open*Facts says: the 890120... barcode is Dabur, the others are not."""
monkeypatch.setattr(
ic, "openfacts_product_matches_brand",
lambda url, brand, timeout=10: "/890/120/" in url or "openfoodfacts" not in url,
)
@pytest.fixture
def store(monkeypatch):
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand",
lambda b, **kw: [dict(r, brand=b) for r in table.values()])
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table
@pytest.fixture
def search(monkeypatch):
"""A scripted `find_all_image_urls`; records every call it receives."""
calls = []
def install(results):
def fake(title, brand=None, country_hint=None, validate=True, max_results=24, produce=False):
calls.append((title, brand))
return list(results)
monkeypatch.setattr(image_search, "find_all_image_urls", fake)
return calls
return install
def _run(headers, rows):
return pipeline.run_pipeline("store.xlsx", _sheet(headers, rows),
use_llm=False, fetch_images=True)
# ---------------------------------------------------------------------------
# 1. One search per product
# ---------------------------------------------------------------------------
def test_every_pack_size_of_a_product_gets_the_same_image_from_one_search(store, search):
calls = search([AMAZON, DABUR_OFF])
_run(["Product Name", "Brand", "Pack Size"], [["Dabur Honey", "Dabur", "1kg, 250g, 50g, 225g"]])
assert len(store) == 4
assert {r["image_url"] for r in store.values()} == {AMAZON}
assert len(calls) == 1, "four pack sizes, one search"
assert calls[0][0] == "Dabur Honey", "the size never enters the query"
def test_sizes_written_into_the_name_still_share_the_search(store, search):
calls = search([AMAZON])
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"], ["Dabur Honey 250g", "Dabur"]])
assert len(calls) == 1
assert calls[0][0] == "Dabur Honey"
assert {r["image_url"] for r in store.values()} == {AMAZON}
def test_different_products_of_one_brand_are_searched_separately(store, search):
calls = search([NAMED])
_run(["Product Name", "Brand"], [["Dabur Honey", "Dabur"], ["Dabur Chyawanprash", "Dabur"]])
assert sorted(c[0] for c in calls) == ["Dabur Chyawanprash", "Dabur Honey"]
# ---------------------------------------------------------------------------
# 2. Open*Facts records are filtered by brand at the source
# ---------------------------------------------------------------------------
def test_openfacts_photos_of_other_brands_honey_are_not_candidates(monkeypatch):
records = [
{"brands": "Dabur", "brands_tags": ["dabur"], "image_front_url": DABUR_OFF},
{"brands": "Rowse", "brands_tags": ["rowse"], "image_front_url": FOREIGN_HONEYS[0]},
{"brands": "Lune de Miel", "image_front_url": FOREIGN_HONEYS[1]},
{"brands": "", "image_front_url": "https://images.openfoodfacts.org/x/unlabelled.jpg"},
]
monkeypatch.setattr(image_search, "_query_openfacts", lambda query, n: records)
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
urls = image_search.find_images_openfacts("Honey", "Dabur")
assert DABUR_OFF in urls
assert not any(u in urls for u in FOREIGN_HONEYS)
assert "https://images.openfoodfacts.org/x/unlabelled.jpg" in urls, \
"a record with no brand at all is not evidence of another brand"
def test_the_title_only_fallback_is_also_brand_filtered(monkeypatch):
seen = []
def query(q, n):
seen.append(q)
if q.startswith("dabur"):
return []
return [{"brands": "Rowse", "image_front_url": FOREIGN_HONEYS[0]}]
monkeypatch.setattr(image_search, "_query_openfacts", query)
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
assert image_search.find_images_openfacts("Honey 1kg", "dabur") == []
assert seen == ["dabur Honey 1kg", "Honey 1kg"]
def test_without_a_brand_nothing_is_filtered(monkeypatch):
monkeypatch.setattr(image_search, "_query_openfacts",
lambda q, n: [{"brands": "Rowse", "image_front_url": FOREIGN_HONEYS[0]}])
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
assert image_search.find_images_openfacts("Honey", None) == FOREIGN_HONEYS[:1]
# ---------------------------------------------------------------------------
# 3. Rejected candidates are not stored where the UI will show them
# ---------------------------------------------------------------------------
def test_choose_primary_separates_eligible_from_rejected():
choice = ic.choose_primary([ODOMOS, FOREIGN_HONEYS[0], AMAZON, NAMED], "Dabur Honey 500g", "Dabur")
assert choice.primary == NAMED # tier 1: names "honey"
assert choice.eligible == [NAMED, AMAZON] # tier 1 then tier 2
assert choice.rejected == [ODOMOS, FOREIGN_HONEYS[0]]
assert choice.ordered == choice.eligible + choice.rejected
def test_a_same_brand_wrong_product_photo_is_never_stored(store, search):
"""The 500g row: a Dabur Odomos repellent as the primary image of honey."""
search([ODOMOS] + FOREIGN_HONEYS)
result = _run(["Product Name", "Brand"], [["Dabur Honey 500g", "Dabur"]])
row = next(iter(store.values()))
assert row["image_url"] is None
assert row["image_urls"] == [], "nothing the gate rejected may reach image_urls"
assert any("rejected" in w for w in result.warnings)
assert any("no primary image" in w for w in result.warnings)
def test_foreign_honeys_ride_along_only_when_a_real_primary_exists(store, search):
search([AMAZON, DABUR_OFF] + FOREIGN_HONEYS)
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
row = next(iter(store.values()))
assert row["image_url"] == AMAZON
assert row["image_urls"] == [AMAZON, DABUR_OFF]
# ---------------------------------------------------------------------------
# 4. A sibling pack size already in the table lends its photo - through the gate
# ---------------------------------------------------------------------------
def test_a_new_pack_size_adopts_the_siblings_image_without_searching(store, search):
calls = search([AMAZON])
_run(["Product Name", "Brand"], [["Dabur Honey 225g", "Dabur"]])
assert len(calls) == 1 and store["dabur_dabur_honey_225g"]["image_url"] == AMAZON
result = _run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
assert len(calls) == 1, "the sibling's photo made the search unnecessary"
assert store["dabur_dabur_honey_1kg"]["image_url"] == AMAZON
assert store["dabur_dabur_honey_1kg"]["image_urls"] == [AMAZON]
assert any("adopted from another pack size" in w for w in result.warnings)
def test_a_siblings_wrong_image_is_not_adopted(store, search):
"""A stored primary that would not pass the gate for this title stays put."""
store["dabur_dabur_honey_500g"] = {
"image_id": "dabur_dabur_honey_500g", "product_name": "Dabur Honey 500g",
"image_url": ODOMOS, "image_urls": [ODOMOS],
}
calls = search([NAMED])
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
assert len(calls) == 1, "an ineligible sibling image forces a fresh search"
assert store["dabur_dabur_honey_1kg"]["image_url"] == NAMED
def test_a_different_product_of_the_brand_is_not_a_sibling(store, search):
store["dabur_dabur_chyawanprash_500g"] = {
"image_id": "dabur_dabur_chyawanprash_500g", "product_name": "Dabur Chyawanprash 500g",
"image_url": NAMED, "image_urls": [NAMED],
}
calls = search([AMAZON])
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
assert len(calls) == 1
assert store["dabur_dabur_honey_1kg"]["image_url"] == AMAZON
def test_stage_six_without_a_cache_still_works():
"""Direct callers (and the produce tests) pass no cache."""
row = {"brand": "Dabur", "product_name": "Dabur Honey 1kg", "image_urls": [AMAZON]}
assert pipeline.stage_6_images(row) is row
# ---------------------------------------------------------------------------
# 5. The one-off repair for rows written before the fix
# ---------------------------------------------------------------------------
def test_regate_rows_reproduces_the_dabur_repair():
from scripts.regate_brand_images import regate_rows
rows = [
{"image_id": "dabur_dabur_honey_225g", "product_name": "Dabur Honey 225g",
"image_url": AMAZON, "image_urls": [AMAZON, DABUR_OFF] + FOREIGN_HONEYS},
{"image_id": "dabur_dabur_honey_1kg", "product_name": "Dabur Honey 1kg",
"image_url": None, "image_urls": list(FOREIGN_HONEYS)},
{"image_id": "dabur_dabur_honey_500g", "product_name": "Dabur Honey 500g",
"image_url": ODOMOS, "image_urls": [ODOMOS]},
{"image_id": "dabur_dabur_chyawanprash_500g", "product_name": "Dabur Chyawanprash 500g",
"image_url": None, "image_urls": [ODOMOS]},
]
by_id = {v["image_id"]: v for v in regate_rows(rows, "dabur")}
# 225g keeps its photo and loses the foreign honeys from its gallery.
assert by_id["dabur_dabur_honey_225g"]["new_url"] == AMAZON
assert by_id["dabur_dabur_honey_225g"]["new_urls"] == [AMAZON, DABUR_OFF]
# 1kg had nothing eligible and adopts the 225g photo.
assert by_id["dabur_dabur_honey_1kg"]["new_url"] == AMAZON
assert by_id["dabur_dabur_honey_1kg"]["why"].startswith("adopted")
# 500g drops the mosquito repellent and adopts the sibling photo too.
assert by_id["dabur_dabur_honey_500g"]["new_url"] == AMAZON
assert ODOMOS in by_id["dabur_dabur_honey_500g"]["rejected"]
# A different product with only a wrong image is blanked, not lent a honey.
assert by_id["dabur_dabur_chyawanprash_500g"]["new_url"] is None
assert by_id["dabur_dabur_chyawanprash_500g"]["new_urls"] == []

View File

@@ -0,0 +1,308 @@
"""The network half of `off_bulk`: what is asked of Open Food Facts, how the
answer is paged, and - the part that produced a user-visible lie - what gets
written to the corpus cache when the answer never arrives.
Background, measured 2026-09-11. Brand Discovery told the user that Open Food
Facts had nothing for "Naga". The v2 API returns two real Naga Limited rows
(Sooji 500 g / 8906011830068, Maida 500 g / 8906011831713). The old query hit
the Search-a-licious endpoint with a free-text `brands:Naga`, whose index
still filed 8906011830068 brandless under Kuwait, and any 503 it met on the
way was cached to disk as `hits: []`. Both failure modes are pinned here, with
no network: `requests.get` is replaced for every test.
"""
from __future__ import annotations
import json
import pytest
import requests
from app.services.enrichment.barcode.sources import off_bulk
from app.services.enrichment.barcode.sources.off_bulk import (
CACHE_SCHEMA,
CorpusFetch,
OffUnavailable,
brand_tag_slug,
fetch_brand_corpus,
fetch_brand_corpus_result,
)
NAGA_PRODUCTS = [
{"code": "8906011830068", "product_name": "Sooji", "product_name_en": "Sooji",
"brands": "Naga", "quantity": "500 g", "countries_tags": ["en:india"]},
{"code": "8906011831713", "product_name": "Maida", "product_name_en": "Maida",
"brands": "Naga", "quantity": "500 g", "countries_tags": ["en:india"]},
]
class _Resp:
def __init__(self, status=200, payload=None, text="", content_type="application/json"):
self.status_code = status
self._payload = payload
self.text = text
self.headers = {"content-type": content_type}
def json(self):
if self._payload is None:
raise ValueError("not json")
return self._payload
def _v2(products, count=None, page=1, page_size=None):
"""A v2 search body. `page_count` is deliberately the size of THIS page,
which is what the real API sends and what a naive paginator misreads;
`page_size` is the size the server APPLIED, not the one requested."""
return {"count": len(products) if count is None else count, "page": page,
"page_count": len(products),
"page_size": off_bulk.PAGE_SIZE if page_size is None else page_size,
"products": products, "skip": 0}
@pytest.fixture(autouse=True)
def _no_sleep(monkeypatch):
monkeypatch.setattr(off_bulk.time, "sleep", lambda s: None)
@pytest.fixture
def calls(monkeypatch):
"""Replace requests.get with a scripted responder. Returns the list of
(url, params) actually sent, so tests can assert on the query."""
sent = []
def install(responder):
def fake_get(url, params=None, headers=None, timeout=None):
sent.append((url, dict(params or {})))
r = responder(len(sent), dict(params or {}))
if isinstance(r, Exception):
raise r
return r
monkeypatch.setattr(off_bulk.requests, "get", fake_get)
return sent
return install
# ---------------------------------------------------------------------------
# The query
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("brand, slug", [
("Naga", "naga"),
("Hindustan Unilever", "hindustan-unilever"),
("P&G", "p-g"),
(" Double Horse ", "double-horse"),
("Coca-Cola", "coca-cola"),
])
def test_brand_tag_slug_matches_off_tag_form(brand, slug):
assert brand_tag_slug(brand) == slug
def test_query_is_an_exact_brand_tag_filter_on_the_v2_api(calls, tmp_path):
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
result = fetch_brand_corpus_result("Naga", country="india", cache_dir=tmp_path)
assert result.error is None and not result.from_cache
assert [h["code"] for h in result.hits] == ["8906011830068", "8906011831713"]
url, params = sent[0]
assert url == "https://world.openfoodfacts.org/api/v2/search"
assert params["brands_tags"] == "naga"
assert params["countries_tags"] == "en:india"
assert "q" not in params, "free-text brands: matching is what returned Mr Naga"
def test_no_country_filter_when_country_is_blank(calls, tmp_path):
sent = calls(lambda n, p: _Resp(payload=_v2([])))
fetch_brand_corpus_result("Naga", country="", cache_dir=tmp_path)
assert "countries_tags" not in sent[0][1]
# ---------------------------------------------------------------------------
# Pagination
# ---------------------------------------------------------------------------
def test_pagination_uses_count_not_page_count(calls, tmp_path):
"""150 products at PAGE_SIZE 100 is two pages. The first page's
`page_count` is 100 - reading that as "pages" would fetch 100 pages."""
page1 = [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)]
page2 = [{"code": f"891{i:010d}", "product_name": f"Q{i}"} for i in range(50)]
def responder(n, params):
return _Resp(payload=_v2(page1 if params["page"] == 1 else page2,
count=150, page=params["page"]))
sent = calls(responder)
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
assert [p["page"] for _, p in sent] == [1, 2]
assert len(result.hits) == 150
def test_pagination_follows_the_page_size_the_server_applied(calls, tmp_path, monkeypatch):
"""The real Amul case: 216 products, 250 requested, 100 served per page.
Dividing by what we ASKED for stops after one page and loses 116 rows."""
monkeypatch.setattr(off_bulk, "PAGE_SIZE", 250)
pages = {
1: [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)],
2: [{"code": f"891{i:010d}", "product_name": f"Q{i}"} for i in range(100)],
3: [{"code": f"892{i:010d}", "product_name": f"R{i}"} for i in range(16)],
}
sent = calls(lambda n, p: _Resp(payload=_v2(pages[p["page"]], count=216,
page=p["page"], page_size=100)))
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
assert [p["page"] for _, p in sent] == [1, 2, 3]
assert len(result.hits) == 216
def test_pagination_is_capped_at_max_pages(calls, tmp_path):
body = [{"code": "8900000000000", "product_name": "X"}] * off_bulk.PAGE_SIZE
sent = calls(lambda n, p: _Resp(payload=_v2(body, count=10 ** 6, page=p["page"])))
fetch_brand_corpus_result("Nestle", cache_dir=tmp_path)
assert len(sent) == off_bulk.MAX_PAGES
def test_nameless_products_are_dropped_from_the_corpus(calls, tmp_path):
calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS + [
{"code": "8906011839999", "product_name": ""},
{"code": "8906011839998"},
])))
assert len(fetch_brand_corpus_result("Naga", cache_dir=tmp_path).hits) == 2
# ---------------------------------------------------------------------------
# Failure is not emptiness, and is never cached
# ---------------------------------------------------------------------------
def test_503_is_retried_then_reported_and_not_cached(calls, tmp_path):
sent = calls(lambda n, p: _Resp(status=503, text="unavailable", content_type="text/html"))
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert len(sent) == 3, "with_retry gives a 5xx three attempts"
assert result.hits == []
assert result.error and "503" in result.error
assert list(tmp_path.glob("*.json")) == [], "a failed fetch must not become hits: []"
def test_503_then_success_recovers(calls, tmp_path):
def responder(n, params):
return _Resp(status=503) if n == 1 else _Resp(payload=_v2(NAGA_PRODUCTS))
calls(responder)
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert result.error is None and len(result.hits) == 2
def test_html_200_error_page_is_a_failure_not_an_empty_brand(calls, tmp_path):
calls(lambda n, p: _Resp(status=200, text="<html>temporarily unavailable</html>",
content_type="text/html"))
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert result.error and result.hits == []
assert list(tmp_path.glob("*.json")) == []
def test_transport_error_after_retries_is_reported(calls, tmp_path):
calls(lambda n, p: requests.exceptions.ConnectionError("dns"))
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert result.error and result.hits == []
assert list(tmp_path.glob("*.json")) == []
def test_partial_fetch_keeps_what_arrived_but_flags_it(calls, tmp_path):
page1 = [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)]
def responder(n, params):
if params["page"] == 1:
return _Resp(payload=_v2(page1, count=150))
return _Resp(status=502)
calls(responder)
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
assert len(result.hits) == 100 and result.error
assert list(tmp_path.glob("*.json")) == []
def test_list_form_still_returns_empty_on_failure(calls, tmp_path):
"""Backfill scripts and ingestion call `fetch_brand_corpus` and rely on
[] rather than an exception so one dead brand does not abort a run."""
calls(lambda n, p: _Resp(status=503))
assert fetch_brand_corpus("Naga", cache_dir=tmp_path) == []
def test_off_unavailable_is_a_connection_error_for_the_retry_decorator():
assert issubclass(OffUnavailable, requests.exceptions.ConnectionError)
# ---------------------------------------------------------------------------
# The cache
# ---------------------------------------------------------------------------
def test_successful_fetch_is_cached_with_schema_and_served_from_cache(calls, tmp_path):
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
first = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
second = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert len(sent) == 1
assert not first.from_cache and second.from_cache
assert second.hits == first.hits
on_disk = json.loads((tmp_path / "naga.json").read_text(encoding="utf-8"))
assert on_disk["schema"] == CACHE_SCHEMA
assert on_disk["brand_tag"] == "naga"
assert on_disk["endpoint"] == off_bulk.SEARCH_URL
def test_genuinely_empty_brand_is_cached_and_is_not_an_error(calls, tmp_path):
sent = calls(lambda n, p: _Resp(payload=_v2([])))
result = fetch_brand_corpus_result("Udhaiyam", cache_dir=tmp_path)
again = fetch_brand_corpus_result("Udhaiyam", cache_dir=tmp_path)
assert result == CorpusFetch(hits=[], error=None, from_cache=False)
assert again.from_cache and again.hits == []
assert len(sent) == 1
def test_empty_pre_schema_cache_is_refetched(calls, tmp_path):
"""The exact prod artefact: an old-endpoint file saying Naga has nothing."""
(tmp_path / "naga.json").write_text(json.dumps({
"brand": "Naga", "country": "india", "fetched_at": 0,
"fetched_at_human": "2026-09-10 12:00:00", "hits": [],
}), encoding="utf-8")
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
assert len(sent) == 1 and not result.from_cache
assert len(result.hits) == 2
assert json.loads((tmp_path / "naga.json").read_text(encoding="utf-8"))["schema"] == CACHE_SCHEMA
def test_non_empty_pre_schema_cache_is_still_trusted(calls, tmp_path):
(tmp_path / "amul.json").write_text(json.dumps({
"brand": "Amul", "country": "india", "fetched_at": 0,
"hits": [{"code": "8901262010115", "product_name": "Amul Butter"}],
}), encoding="utf-8")
sent = calls(lambda n, p: _Resp(status=503))
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
assert sent == [] and result.from_cache
assert result.hits[0]["product_name"] == "Amul Butter"
def test_refresh_bypasses_the_cache(calls, tmp_path):
(tmp_path / "naga.json").write_text(json.dumps({
"schema": CACHE_SCHEMA, "brand": "Naga", "hits": [],
}), encoding="utf-8")
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
assert len(fetch_brand_corpus_result("Naga", refresh=True, cache_dir=tmp_path).hits) == 2
assert len(sent) == 1
def test_unreadable_cache_is_refetched(calls, tmp_path):
(tmp_path / "naga.json").write_text("{not json", encoding="utf-8")
calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
assert len(fetch_brand_corpus_result("Naga", cache_dir=tmp_path).hits) == 2

View File

@@ -159,7 +159,11 @@ def test_the_sheets_values_are_kept_and_nothing_else_is_invented(store):
# What the sheet said.
assert row["product_name"] == "Apple 500g"
assert row["size_variants"] == ["500g"]
assert row["final_selling_price"] == 155
# "Selling Price" is the shop's price and lands in selling_price. The
# final (tax-inclusive) column is filled from it by upsert_brand_products,
# which this fixture stubs, so here it stays what the sheet said: nothing.
assert row["selling_price"] == 155
assert row["final_selling_price"] is None
assert row["price_range"] == "₹143-167" # +/-8% of the sheet's own price
# What it did not say, and what we therefore do not claim.

View File

@@ -3,7 +3,7 @@
WHY THIS EXISTS
---------------
Open Food Facts answers "does this product exist" for food and nothing else.
`off_bulk` queries only `search.openfoodfacts.org`, so a toothpaste or a
`off_bulk` queries only Open Food Facts' food database, so a toothpaste or a
detergent has no product-discovery source at all and its whole catalogue is
language-model output. A live retail lookup is the only signal in the codebase
that means "available right now" rather than "was in a database dump", and it

View File

@@ -0,0 +1,312 @@
"""What an uploaded sheet's own facts become in the brand table.
Three rules, each of which was violated before this file existed:
1. THE RETAIL PRICE IS THE SELLING PRICE. Every price header used to land in
`final_selling_price` and `selling_price` was NULL for every uploaded row;
worse, the HSN/GST stage then overwrote whatever price the sheet gave with
the ceiling of the derived band (155 became 167).
2. A PACK SIZE IS RECORDED ONLY WHEN THE SHEET OR THE NAME STATES IT.
"India Gate Basmati Rice 1kg" is one product in one size. "India Gate
Basmati Rice" used to become three products - 1kg, 5kg, 25kg - none of
which the sheet listed.
3. A BLANK DESCRIPTION IS WRITTEN, NOT PADDED. The LLM writes it when Ollama
is up; when it is not, the row gets a short factual line built only from
facts the row holds - never "<name> from <brand>." and never the marketing
essay in catalog_engine.generate_detailed_description.
Everything runs offline: storage and embeddings are monkeypatched on the
pipeline module, and USE_OLLAMA is false for the whole suite (conftest.py).
"""
from __future__ import annotations
import io
import pytest
from app.core import store_catalog_pipeline as pipeline
from app.services.enrichment.hsn_gst.models import enrich_pricing_fields
from app.services.generic_products import OWN_PRODUCTS_BRAND
openpyxl = pytest.importorskip("openpyxl")
def _sheet(headers, rows) -> bytes:
wb = openpyxl.Workbook()
ws = wb.active
ws.append(headers)
for row in rows:
ws.append(row)
buf = io.BytesIO()
wb.save(buf)
return buf.getvalue()
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture
def store(monkeypatch):
"""A fake brand table: image_id -> stored row, across every brand."""
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table
def _run(headers, rows, **kw):
kw.setdefault("use_llm", False)
kw.setdefault("fetch_images", False)
return pipeline.run_pipeline("store.xlsx", _sheet(headers, rows), **kw)
# Real-looking names: the validation gate rejects placeholder titles such as
# "Product 3", which would empty the store and hide what a test is about.
_AMUL_NAMES = ["Amul Butter 100g", "Amul Cheese 200g", "Amul Ghee 500ml", "Amul Taaza 1L",
"Amul Kool 200ml", "Amul Lassi 200ml", "Amul Masti Dahi 400g"]
def _only(store):
assert len(store) == 1, list(store)
return next(iter(store.values()))
# ---------------------------------------------------------------------------
# 1. Price
# ---------------------------------------------------------------------------
def test_retail_price_lands_in_selling_price_and_drives_the_band(store):
_run(["ProductName", "ProductSKU", "ProductBrand", "RetailPrice"],
[["India Gate Basmati Rice 1kg", "IG-BR-1", "India Gate", 120]])
row = _only(store)
assert row["selling_price"] == 120
assert row["price_range"] == "₹110-130" # +/-8% of the sheet's price
assert row["product_sku"] == "IG-BR-1" and row["sku_source"] == "sheet"
def test_a_final_price_column_is_kept_separate_and_wins_the_band(store):
_run(["Item Name", "Brand", "Selling Price", "Final Price"],
[["Amul Butter 500g", "Amul", 250, 262]])
row = _only(store)
assert row["selling_price"] == 250
assert row["final_selling_price"] == 262
assert row["price_range"] == "₹241-283" # band from the final price
def test_reingesting_a_sheet_with_a_selling_price_is_a_no_op(store):
headers, rows = ["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 250]]
first = _run(headers, rows)
assert first.inserted == 1
second = _run(headers, rows)
assert (second.inserted, second.backfilled, second.skipped_existing) == (0, 0, 1)
def test_a_held_final_price_is_not_clobbered_by_a_different_retail_price(store):
_run(["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 250]])
key = next(iter(store))
store[key]["final_selling_price"] = 262 # what the table already holds
_run(["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 199]])
assert store[key]["final_selling_price"] == 262
assert store[key]["selling_price"] == 250, "fill-only-blanks: a held price is kept"
class TestHsnStageKeepsHeldPrices:
"""The overwrite that turned a sheet's 155 into 167."""
def test_a_held_selling_price_is_returned_as_none_so_apply_keeps_it(self):
fields = enrich_pricing_fields({
"selling_price": 155, "final_selling_price": None,
"price_range": "₹143-167", "gst_percent": 18,
})
assert fields["selling_price"] is None
assert fields["tax_amount"] == 27.9 # 18% of the HELD 155, not of 167
assert fields["final_selling_price"] == 182.9
def test_a_held_final_price_is_also_kept(self):
fields = enrich_pricing_fields({
"selling_price": 155, "final_selling_price": 160,
"price_range": "₹143-167", "gst_percent": 18,
})
assert fields["selling_price"] is None
assert fields["final_selling_price"] is None
def test_without_a_sheet_price_the_band_ceiling_is_still_the_base(self):
fields = enrich_pricing_fields({"price_range": "₹143-167", "gst_percent": 5})
assert fields["selling_price"] == 167
assert fields["final_selling_price"] == 175.35
def test_a_held_final_price_alone_seeds_the_selling_price(self):
fields = enrich_pricing_fields({"final_selling_price": 200, "gst_percent": 5,
"price_range": "₹300-400"})
assert fields["selling_price"] == 200
assert fields["final_selling_price"] is None
# ---------------------------------------------------------------------------
# 2. Size
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("name, size", [
("India Gate Basmati Rice 1kg", "1kg"),
("Fortune Sunflower Oil 1L", "1L"),
("Amul Taaza Milk 1 Litre", "1 Litre"),
("Aashirvaad Atta 5 Kg", "5 Kg"),
("Tata Salt 1.5kg", "1.5kg"),
("Parle-G Biscuits 800 gm", "800 gm"),
("Saffola Gold 500ml", "500ml"),
])
def test_the_size_in_the_name_is_the_size_and_nothing_is_appended(store, name, size):
_run(["Product Name", "Brand"], [[name, name.split()[0]]])
row = _only(store)
assert row["size_variants"] == [size]
assert row["product_name"] == name, "the name already carries the size"
def test_a_branded_product_with_no_size_is_one_clean_row(store):
result = _run(["Product Name", "Brand"], [["India Gate Basmati Rice", "India Gate"]])
row = _only(store)
assert row["size_variants"] == []
assert row["product_name"] == "India Gate Basmati Rice"
assert "Standard" not in row["product_name"]
assert next(iter(store)) == "india_gate_india_gate_basmati_rice"
assert any("no pack size" in w for w in result.warnings)
def test_a_commodity_with_no_size_still_uses_the_standard_sentinel(store):
"""Own Products dedupes against a seeded base list built on "Standard"."""
_run(["Product Name"], [["Apple"]])
row = _only(store)
assert row["size_variants"] == ["Standard"]
assert next(iter(store)) == pipeline.build_image_id(OWN_PRODUCTS_BRAND, "Apple", "Standard")
def test_the_llm_may_not_supply_a_pack_size(store, monkeypatch):
from app.services import ollama_service
monkeypatch.setattr(ollama_service, "fetch_product_details",
lambda brand, title, category=None, size=None: {
"description": "Refined wheat flour milled for baking.",
"size_variants": ["500g", "1kg", "5kg"],
})
_run(["Product Name", "Brand"], [["Naga Maida", "Naga"]], use_llm=True)
row = _only(store)
assert row["size_variants"] == []
assert row["description"] == "Refined wheat flour milled for baking."
def test_a_sheet_size_column_still_wins_over_the_name(store):
_run(["Product Name", "Brand", "Pack Size"], [["Amul Butter 100g", "Amul", "500g"]])
assert _only(store)["size_variants"] == ["500g"]
# ---------------------------------------------------------------------------
# 3. Description
# ---------------------------------------------------------------------------
def test_the_llm_description_is_used_when_ollama_answers(store, monkeypatch):
from app.services import ollama_service
seen = {}
def fake(brand, title, category=None, size=None):
seen.update(brand=brand, title=title, category=category, size=size)
return {"description": "Long-grain aged basmati rice for biryani and pulao."}
monkeypatch.setattr(ollama_service, "fetch_product_details", fake)
_run(["Product Name", "Brand"], [["India Gate Basmati Rice 1kg", "India Gate"]], use_llm=True)
assert _only(store)["description"] == "Long-grain aged basmati rice for biryani and pulao."
assert seen["size"] == "1kg", "the pack size in the name is handed to the prompt"
def test_an_overlong_llm_description_is_clamped(store, monkeypatch):
from app.services import ollama_service
monkeypatch.setattr(ollama_service, "fetch_product_details",
lambda *a, **k: {"description": "x" * 2000})
_run(["Product Name", "Brand"], [["Amul Butter 500g", "Amul"]], use_llm=True)
assert len(_only(store)["description"]) == pipeline._MAX_LLM_DESCRIPTION_CHARS
def test_the_fallback_description_is_factual_and_short(store):
_run(["Product Name", "Brand", "Category"],
[["Fortune Sunflower Oil 1L", "Fortune", "Edible Oils"]])
desc = _only(store)["description"]
assert desc == "Fortune Sunflower Oil 1L, Edible Oils, by Fortune."
assert "from Fortune." not in desc and "Introducing" not in desc
def test_the_fallback_description_names_a_sheet_size_the_name_lacks(store):
_run(["Product Name", "Brand", "Pack Size"], [["Amul Butter", "Amul", "500g"]])
assert _only(store)["description"] == "Amul Butter 500g, Dairy, by Amul."
def test_a_sheet_description_is_never_overwritten(store, monkeypatch):
from app.services import ollama_service
monkeypatch.setattr(ollama_service, "fetch_product_details",
lambda *a, **k: pytest.fail("the LLM must not be asked"))
_run(["Product Name", "Brand", "Description"],
[["Amul Butter 500g", "Amul", "Salted table butter."]], use_llm=True)
assert _only(store)["description"] == "Salted table butter."
def test_the_llm_is_switched_off_after_three_consecutive_misses(store, monkeypatch):
from app.services import ollama_service
calls = []
monkeypatch.setattr(ollama_service, "fetch_product_details",
lambda brand, title, **k: calls.append(title) or None)
rows = [[name, "Amul"] for name in _AMUL_NAMES[:6]]
result = _run(["Product Name", "Brand"], rows, use_llm=True)
assert len(calls) == 3
assert any("consecutive" in w for w in result.warnings)
assert len(store) == 6, "the rows themselves are unaffected"
def test_one_answer_resets_the_breaker(store, monkeypatch):
from app.services import ollama_service
answers = iter([None, None, {"description": "ok"}, None, None, None, None])
calls = []
def fake(brand, title, **k):
calls.append(title)
return next(answers)
monkeypatch.setattr(ollama_service, "fetch_product_details", fake)
_run(["Product Name", "Brand"], [[name, "Amul"] for name in _AMUL_NAMES], use_llm=True)
# miss, miss, hit (reset), miss, miss, miss (trip) -> the 7th row is not asked.
assert len(calls) == 6
def test_stage_two_leaves_a_blank_description_blank_on_a_miss():
"""The fallback lives in _to_storage_row; stage 2 must not pre-empt it."""
row = {"brand": "Amul", "product_name": "Amul Butter 500g", "description": ""}
breaker = pipeline.LlmBreaker()
pipeline.stage_2_row_intake(row, use_llm=True, breaker=breaker)
assert row["description"] == ""
assert breaker.consecutive_misses == 1

View File

@@ -187,6 +187,45 @@ def test_template_headers_all_map_to_their_field():
assert mapping.ignored == []
@pytest.mark.parametrize("header, canonical", [
# The camel-cased headers a store's export tool produces, which
# _normalize_header collapses to one word - locked as exact aliases so
# they do not depend on the keyword rules' ordering.
("ProductName", "product_name"),
("ProductBrand", "brand"),
("Product Brand", "brand"),
("ProductSKU", "product_sku"),
("Product SKU", "product_sku"),
# What the shop charges -> selling_price.
("RetailPrice", "selling_price"),
("Retail Price", "selling_price"),
("Selling Price", "selling_price"),
("Sale Price", "selling_price"),
("SP", "selling_price"),
("Price", "selling_price"),
("Unit Price (Rs)", "selling_price"),
# The tax-inclusive ceiling -> final_selling_price. "Final Selling Price"
# contains "selling price" and must still win: rule order regression.
("Final Selling Price", "final_selling_price"),
("Final Price (₹)", "final_selling_price"),
("MRP", "final_selling_price"),
("Maximum MRP", "final_selling_price"),
])
def test_price_and_identity_headers_map_where_the_sheet_means_them(header, canonical):
"""Before the selling_price canonical existed every price header landed in
final_selling_price and the table's selling_price column stayed NULL."""
mapping = user_products.map_spreadsheet_columns([header])
assert mapping.columns == {canonical: header}
def test_a_sheet_may_carry_both_a_retail_and_a_final_price():
mapping = user_products.map_spreadsheet_columns(
["Item Name", "Brand", "Retail Price", "MRP"])
assert mapping.columns["selling_price"] == "Retail Price"
assert mapping.columns["final_selling_price"] == "MRP"
assert mapping.ignored == []
def test_a_product_sku_column_is_not_mistaken_for_the_product_name():
"""'Product SKU' contains 'product'. Matched loosely, it used to become a
second product_name column, and duplicate columns are what turned a row