Brand valid image generation
This commit is contained in:
@@ -17,14 +17,20 @@ now the only admin way in. What made it the survivor is the unit of work: five
|
||||
files are one batch with one id, so the question a colleague actually asks -
|
||||
"did the drop land?" - has one answer rather than five.
|
||||
|
||||
WHY THE NETWORK STAGES DEFAULT OFF HERE
|
||||
---------------------------------------
|
||||
Turning both on for a single file is a considered trade. At twenty files it is
|
||||
thousands of outbound requests and, for the image stage, a Playwright subprocess
|
||||
that can burn three minutes on its own - on a single-vCPU container that is also
|
||||
serving the API. So a batch opts IN to those stages; it does not opt out.
|
||||
`USE_OLLAMA` is false in production anyway, which makes `use_llm` a no-op there
|
||||
and the honest default obvious.
|
||||
WHY IMAGES DEFAULT OFF HERE AND THE LLM DEFAULTS ON
|
||||
---------------------------------------------------
|
||||
Image search for a single file is a considered trade. At twenty files it is
|
||||
thousands of outbound requests and a Playwright subprocess that can burn three
|
||||
minutes on its own - on a single-vCPU container that is also serving the API.
|
||||
So a batch opts IN to that stage; it does not opt out.
|
||||
|
||||
`use_llm` is different and defaults ON: it gates only the description written
|
||||
in stage 2 for rows the sheet left blank, which is the one field a shopper
|
||||
reads and a store almost never supplies. It is cheap to leave on because
|
||||
`ollama_service._ensure_client` caches its reachability probe per batch and
|
||||
`store_catalog_pipeline.LlmBreaker` stops calling after three consecutive
|
||||
misses, so with `USE_OLLAMA` false (production today) the cost is one warning
|
||||
per file and the rows fall back to a factual template.
|
||||
|
||||
Note that these defaults bind THIS router only. The open upload endpoint runs
|
||||
itself and takes its two flags from `UPLOAD_AUTORUN_FETCH_IMAGES` and
|
||||
@@ -145,7 +151,7 @@ async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
|
||||
dependencies=[Depends(require_admin)])
|
||||
async def ingest_catalog_batch(
|
||||
files: List[UploadFile] = File(...),
|
||||
use_llm: bool = False,
|
||||
use_llm: bool = True,
|
||||
fetch_images: bool = False,
|
||||
) -> BatchOut:
|
||||
"""Stage the files, queue the batch, and return an id to poll.
|
||||
@@ -325,7 +331,8 @@ class InboxStartRequest(InboxSelection):
|
||||
# Chosen HERE, not by the sender - see the note on the POST handler in
|
||||
# uploads.py. These commit the host to outbound work, so the decision
|
||||
# belongs to the person who can see what the machine is already doing.
|
||||
use_llm: bool = False
|
||||
# The LLM is on by default for the reason in the module docstring.
|
||||
use_llm: bool = True
|
||||
fetch_images: bool = False
|
||||
# Who runs it. "inprocess" is this container's worker thread and is the
|
||||
# default, so an existing client that never sends the field is unaffected.
|
||||
|
||||
@@ -124,11 +124,14 @@ def _normalize_header(raw: Any) -> str:
|
||||
|
||||
|
||||
_EXACT_HEADERS: Dict[str, str] = {
|
||||
# brand
|
||||
# brand. "productbrand" is what a camel-cased ProductBrand header becomes
|
||||
# after _normalize_header, and is listed so the mapping does not depend on
|
||||
# the keyword rules' ordering.
|
||||
"brand": "brand", "brand name": "brand", "brands": "brand",
|
||||
"company": "brand", "manufacturer": "brand", "company name": "brand",
|
||||
"product brand": "brand", "productbrand": "brand",
|
||||
# product name
|
||||
"product": "product_name", "product name": "product_name",
|
||||
"product": "product_name", "product name": "product_name", "productname": "product_name",
|
||||
"product name variant": "product_name", "product variant": "product_name",
|
||||
"variant": "product_name", "name": "product_name", "item": "product_name",
|
||||
"item name": "product_name", "product title": "product_name",
|
||||
@@ -138,18 +141,27 @@ _EXACT_HEADERS: Dict[str, str] = {
|
||||
# category
|
||||
"category": "category", "categories": "category", "cat": "category",
|
||||
"product category": "category", "segment": "category",
|
||||
# price
|
||||
# price. Two columns, two meanings: `selling_price` is what the shop
|
||||
# charges - a retail / sale / selling price, or a bare "Price" - and
|
||||
# `final_selling_price` is the tax-inclusive ceiling (MRP, "final price").
|
||||
# A sheet that carries only one of them still fills both, because
|
||||
# vector_store.upsert_brand_products copies selling -> final when final is
|
||||
# blank. Before this split every price header landed in final_selling_price
|
||||
# and selling_price was NULL for every uploaded row.
|
||||
"price range": "price_range", "range": "price_range", "mrp range": "price_range",
|
||||
"price": "final_selling_price", "final price": "final_selling_price",
|
||||
"final price rs": "final_selling_price", "final selling price": "final_selling_price",
|
||||
"selling price": "final_selling_price", "mrp": "final_selling_price",
|
||||
"rate": "final_selling_price", "amount": "final_selling_price",
|
||||
"cost": "final_selling_price", "unit price": "final_selling_price",
|
||||
"selling price": "selling_price", "sellingprice": "selling_price",
|
||||
"retail price": "selling_price", "retailprice": "selling_price",
|
||||
"sale price": "selling_price", "sp": "selling_price",
|
||||
"price": "selling_price", "unit price": "selling_price", "rate": "selling_price",
|
||||
"final price": "final_selling_price", "final price rs": "final_selling_price",
|
||||
"final selling price": "final_selling_price", "mrp": "final_selling_price",
|
||||
"amount": "final_selling_price", "cost": "final_selling_price",
|
||||
# identifiers
|
||||
"barcode": "barcode", "barcode gtin ean": "barcode", "bar code": "barcode",
|
||||
"gtin": "barcode", "ean": "barcode", "upc": "barcode", "ean13": "barcode",
|
||||
"hsn": "hsn_code", "hsn code": "hsn_code", "hsn sac": "hsn_code",
|
||||
"sku": "product_sku", "product sku": "product_sku", "sku code": "product_sku",
|
||||
"sku": "product_sku", "product sku": "product_sku", "productsku": "product_sku",
|
||||
"sku code": "product_sku",
|
||||
"fssai": "fssai_license", "fssai license": "fssai_license",
|
||||
"fssai license number": "fssai_license", "fssai number": "fssai_license",
|
||||
# text / media
|
||||
@@ -173,7 +185,13 @@ _KEYWORD_RULES: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
|
||||
("barcode", ("barcode", "bar code", "gtin", "ean", "upc")),
|
||||
("product_sku", ("sku",)),
|
||||
("price_range", ("price range", "range")),
|
||||
("final_selling_price", ("final price", "selling price", "price", "mrp", "rate", "cost")),
|
||||
# Order is load-bearing: "final selling price" contains "selling price",
|
||||
# and every price header contains "price", so the final-price forms must
|
||||
# be claimed before the generic ones. "sp" is deliberately exact-only - as
|
||||
# a substring it is inside "spice", "display" and "sponge".
|
||||
("final_selling_price", ("final price", "final selling price", "mrp")),
|
||||
("selling_price", ("selling price", "retail price", "sale price", "price", "rate")),
|
||||
("final_selling_price", ("cost",)),
|
||||
("image_url", ("image", "photo", "picture", "url", "link")),
|
||||
("description", ("description", "desc", "detail")),
|
||||
("category", ("category", "segment")),
|
||||
@@ -346,6 +364,7 @@ def row_to_request(row: Dict[str, Any], mapping: _ColumnMapping) -> AddProductRe
|
||||
product_sku=_text(row, mapping, "product_sku"),
|
||||
hsn_code=_text(row, mapping, "hsn_code"),
|
||||
final_selling_price=_number(row, mapping, "final_selling_price"),
|
||||
selling_price=_number(row, mapping, "selling_price"),
|
||||
barcode=_text(row, mapping, "barcode"),
|
||||
image_url=_text(row, mapping, "image_url"),
|
||||
)
|
||||
@@ -528,8 +547,9 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
|
||||
|
||||
price_range = req.price_range
|
||||
if not price_range:
|
||||
if req.final_selling_price:
|
||||
price_range = f"₹{req.final_selling_price}"
|
||||
sheet_price = req.final_selling_price or req.selling_price
|
||||
if sheet_price:
|
||||
price_range = f"₹{sheet_price}"
|
||||
elif sample_existing.get("price_range"):
|
||||
price_range = sample_existing.get("price_range")
|
||||
else:
|
||||
|
||||
@@ -91,7 +91,8 @@ def generate_product_highlights(product: Dict[str, Any], brand: str) -> List[str
|
||||
highlights.append(f"Available in {size} - {price}")
|
||||
elif size:
|
||||
highlights.append(f"Available in {size}")
|
||||
elif isinstance(v, str) and v.strip():
|
||||
elif isinstance(v, str) and v.strip() and v.strip().lower() != "standard":
|
||||
# "Standard" is the pipeline's no-size sentinel, not a pack size.
|
||||
highlights.append(f"Available in {v}")
|
||||
|
||||
# Quality indicators from description
|
||||
|
||||
@@ -47,7 +47,7 @@ import asyncio
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
from typing import Any, Callable, Dict, List, Optional, Sequence, Tuple
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
ENABLE_BARCODE_LOOKUP,
|
||||
@@ -132,8 +132,16 @@ _ENRICHABLE = (
|
||||
"selling_price", "image_url", "image_urls", "size_variants",
|
||||
)
|
||||
|
||||
# The pack size a store sheet usually carries inside the product name itself -
|
||||
# "India Gate Basmati Rice 1kg", "Fortune Sunflower Oil 1L", "Amul Milk 1
|
||||
# Litre". The match is kept as written in the name (case and spacing), so the
|
||||
# size a shopper sees on the card is the size the store typed. Longer unit
|
||||
# spellings precede their prefixes because alternation is first-match-wins.
|
||||
_SIZE_IN_TITLE = re.compile(
|
||||
r"\b(\d+(?:[.,]\d+)?)\s*(kg|kgs|g|gm|gms|gram|grams|ml|l|ltr|litre|liters?|pack|pcs|n)\b",
|
||||
r"\b(\d+(?:[.,]\d+)?)\s*"
|
||||
r"(kgs|kg|gms|gm|grams|gram|mg|g|"
|
||||
r"mls|ml|litres|litre|liters|liter|ltrs|ltr|lt|l|"
|
||||
r"pieces|piece|pcs|pc|nos|pack|n)\b",
|
||||
re.I,
|
||||
)
|
||||
|
||||
@@ -276,24 +284,71 @@ def stage_1_brand_and_fssai(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 2 - Row intake (replaces AI discovery)
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True) -> Dict[str, Any]:
|
||||
@dataclass
|
||||
class LlmBreaker:
|
||||
"""Stop asking the LLM once it has stopped answering.
|
||||
|
||||
`ollama_service._generate` swallows every exception and returns "", so a
|
||||
slow or dead Ollama never raises - it just costs up to
|
||||
OLLAMA_TIMEOUT_SECONDS x retries per row and yields nothing. On a
|
||||
2000-row sheet that is days. Three misses in a row and the rest of the
|
||||
batch goes straight to the factual fallback description; one success
|
||||
resets the count, so a single hiccup does not switch the LLM off.
|
||||
"""
|
||||
limit: int = 3
|
||||
consecutive_misses: int = 0
|
||||
tripped: bool = False
|
||||
|
||||
def record(self, ok: bool) -> None:
|
||||
if ok:
|
||||
self.consecutive_misses = 0
|
||||
return
|
||||
self.consecutive_misses += 1
|
||||
if self.consecutive_misses >= self.limit:
|
||||
self.tripped = True
|
||||
|
||||
|
||||
# The generated description is stored in a TEXT column, but a 1.5b model does
|
||||
# not reliably honour the length it is asked for and the catalog card shows a
|
||||
# short paragraph, not an essay.
|
||||
_MAX_LLM_DESCRIPTION_CHARS = 300
|
||||
|
||||
|
||||
def stage_2_row_intake(row: Dict[str, Any], *, use_llm: bool = True,
|
||||
breaker: Optional[LlmBreaker] = None) -> Dict[str, Any]:
|
||||
"""The spreadsheet row *is* the product. Only fill a missing description.
|
||||
|
||||
The LLM call is best-effort and optional: with Ollama unreachable the row
|
||||
keeps an empty description and later stages still work.
|
||||
keeps an empty description here and `_to_storage_row` supplies a factual
|
||||
template at storage time, so later stages still work.
|
||||
|
||||
DESCRIPTION ONLY. This stage used to copy the LLM's `size_variants` into a
|
||||
row that had none, which is how "Naga Maida" acquired a 500g/1kg/5kg it
|
||||
was never sold in. A pack size is a fact about the physical product and
|
||||
comes from the sheet or the product name (stage 4), never from a guess.
|
||||
"""
|
||||
if not _blank(row.get("description")) or not use_llm:
|
||||
return row
|
||||
if breaker is not None and breaker.tripped:
|
||||
return row
|
||||
ok = False
|
||||
try:
|
||||
from app.services.ollama_service import fetch_product_details
|
||||
|
||||
details = fetch_product_details(row.get("brand", ""), row.get("product_name", ""))
|
||||
sheet_sizes = [str(s).strip() for s in (row.get("size_variants") or []) if str(s).strip()]
|
||||
title_size = _SIZE_IN_TITLE.search(row.get("product_name") or "")
|
||||
size = sheet_sizes[0] if sheet_sizes else (title_size.group(0).strip() if title_size else None)
|
||||
details = fetch_product_details(
|
||||
row.get("brand", ""), row.get("product_name", ""),
|
||||
category=row.get("category") or None, size=size,
|
||||
)
|
||||
if details and details.get("description"):
|
||||
row["description"] = str(details["description"]).strip()
|
||||
if _blank(row.get("size_variants")) and details and details.get("size_variants"):
|
||||
row["size_variants"] = list(details["size_variants"])
|
||||
row["description"] = str(details["description"]).strip()[:_MAX_LLM_DESCRIPTION_CHARS]
|
||||
ok = True
|
||||
except Exception as exc: # noqa: BLE001 - enrichment is best-effort
|
||||
logger.debug("LLM enrichment skipped for %r: %s", row.get("product_name"), exc)
|
||||
if breaker is not None:
|
||||
breaker.record(ok)
|
||||
return row
|
||||
|
||||
|
||||
@@ -442,30 +497,32 @@ def _sizes_for(row: Dict[str, Any]) -> List[str]:
|
||||
if match:
|
||||
return [match.group(0).strip()]
|
||||
|
||||
# A COMMODITY WITH NO WEIGHT COLUMN GETS ONE UNSIZED ROW, NOT THREE MADE-UP
|
||||
# ONES. default_size_variants() invents a plausible set - 100g/250g/500g -
|
||||
# and for packaged goods that is a reasonable guess at what a brand sells.
|
||||
# For loose produce it is not: a shop sells apples by the kilo at whatever
|
||||
# the customer asks for, so "Apple 250g" is a product that does not exist.
|
||||
# NO SIZE ANYWHERE MEANS ONE UNSIZED ROW, NOT THREE MADE-UP ONES - for
|
||||
# every brand. This used to fall through to default_size_variants() for
|
||||
# branded rows, which invents a plausible set (1kg/5kg/25kg for rice) and
|
||||
# so turned "India Gate Basmati Rice" into three products that the sheet
|
||||
# never listed and the shop may never have stocked. It was also the
|
||||
# mechanism behind the catalogue drift the integrator reported: the
|
||||
# invented set is keyed on the resolved category, so the same product
|
||||
# ingested twice with the category resolved differently produced two
|
||||
# disjoint size sets, two sets of image_ids, and a re-scrape that looked
|
||||
# like the old rows were deleted. A pack size is recorded only when the
|
||||
# sheet or the product name states it.
|
||||
#
|
||||
# It is also the mechanism behind the catalogue drift the integrator
|
||||
# reported: the invented set is keyed on the resolved category, so the same
|
||||
# product ingested twice with the category resolved differently produces
|
||||
# two disjoint size sets, two sets of image_ids, and a re-scrape that looks
|
||||
# like the old rows were deleted. Keeping produce out of that from the
|
||||
# start is cheaper than repairing it later.
|
||||
# "Standard" and not "": an empty size fails validate_size ("size/pack is
|
||||
# missing") and stage 4 would drop the row, which is the rejection this
|
||||
# whole area exists to prevent. "Standard" is also what _to_storage_row
|
||||
# already substitutes for a blank size and what the seeded base list uses,
|
||||
# so an uploaded "Apple" deduplicates onto the seeded "Apple" instead of
|
||||
# creating a second row.
|
||||
if row.get("brand") == OWN_PRODUCTS_BRAND:
|
||||
return ["Standard"]
|
||||
|
||||
return list(price_estimator.default_size_variants(
|
||||
row.get("category") or "", row.get("product_name") or ""
|
||||
))
|
||||
# missing") and stage 10 would penalise the row for a fact nobody had. It is
|
||||
# the pipeline's internal no-size sentinel, understood by the validator,
|
||||
# the price estimator and the SKU resolver. What reaches the table depends
|
||||
# on the brand - see _to_storage_row: Own Products keeps "Standard" so an
|
||||
# uploaded "Apple" deduplicates onto the seeded "Apple" (same image_id);
|
||||
# a branded row stores an empty size_variants and an unsuffixed name.
|
||||
if row.get("brand") != OWN_PRODUCTS_BRAND:
|
||||
# Worth a line in the job report for a packaged good; for loose
|
||||
# produce an absent size is the normal case and would only be noise.
|
||||
row.setdefault("_notes", []).append(
|
||||
"no pack size in the sheet or the product name; stored without one"
|
||||
)
|
||||
return ["Standard"]
|
||||
|
||||
|
||||
def stage_4_explode_sizes(row: Dict[str, Any]) -> Tuple[List[Dict[str, Any]], List[str]]:
|
||||
@@ -519,8 +576,14 @@ def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
if not _blank(row.get("price_range")):
|
||||
return row
|
||||
|
||||
if not _blank(row.get("final_selling_price")):
|
||||
price = float(row["final_selling_price"])
|
||||
# Either price the sheet gave is a real price for this pack; the final
|
||||
# (tax-inclusive) one is preferred when both are present because the band
|
||||
# is what the card shows a shopper.
|
||||
sheet_price = row.get("final_selling_price")
|
||||
if _blank(sheet_price):
|
||||
sheet_price = row.get("selling_price")
|
||||
if not _blank(sheet_price):
|
||||
price = float(sheet_price)
|
||||
lo = int(round(price * (1 - _SHEET_PRICE_BAND)))
|
||||
hi = int(round(price * (1 + _SHEET_PRICE_BAND)))
|
||||
elif row.get("brand") == OWN_PRODUCTS_BRAND:
|
||||
@@ -537,7 +600,84 @@ def stage_5_pricing(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 6 - Images
|
||||
# ---------------------------------------------------------------------------
|
||||
def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, Any]:
|
||||
class ImageSearchCache:
|
||||
"""One image search per PRODUCT, shared by every pack size of it.
|
||||
|
||||
Stage 4 turns one source row into one row per size, and this stage used to
|
||||
search once per resulting row. The query was the same each time - the size
|
||||
is not in it - but the providers behind `find_all_image_urls` are live and
|
||||
not deterministic, so "Dabur Honey" asked four times came back four ways:
|
||||
the 225g row drew an Amazon photo of Dabur Honey, the 1kg row drew nothing
|
||||
usable. Same product, four different images, three of them wrong or blank.
|
||||
|
||||
Keyed on `image_corroboration.search_key`, which strips a trailing pack
|
||||
size from the name, so sheet rows that carry the size inside the name
|
||||
("India Gate Basmati Rice 1kg" / "... 5kg") share as well. The cached value
|
||||
is the RANKED candidate list; each size still runs its own gate, which
|
||||
ignores size tokens and so reaches the same verdict for every sibling.
|
||||
|
||||
`existing` maps a search key to an already-stored eligible primary for
|
||||
the same product under the same brand, so a re-ingest that finds nothing
|
||||
new can inherit the photo a sibling size already has - after the gate.
|
||||
"""
|
||||
|
||||
def __init__(self, existing_rows: Optional[Sequence[Dict[str, Any]]] = None) -> None:
|
||||
self.candidates: Dict[tuple, List[str]] = {}
|
||||
self.existing: Dict[tuple, str] = {}
|
||||
for prior in existing_rows or ():
|
||||
url = prior.get("image_url")
|
||||
if not url or not str(url).startswith("http"):
|
||||
continue
|
||||
key = image_corroboration.search_key(prior.get("product_name") or "",
|
||||
prior.get("brand") or "")
|
||||
self.existing.setdefault(key, str(url))
|
||||
|
||||
|
||||
def _adopt_sibling_image(row: Dict[str, Any], brand: str, cache: ImageSearchCache) -> bool:
|
||||
"""Reuse the eligible primary of another pack size of the same product.
|
||||
|
||||
Only ever through the gate: the sibling's URL is corroborated against THIS
|
||||
row's title before it is adopted, so a wrong image stored under one size
|
||||
cannot propagate to the rest. A sibling is the same product in a different
|
||||
pack, which is what `search_key` means, and the same photo on every pack is
|
||||
what a retailer shows too.
|
||||
"""
|
||||
key = image_corroboration.search_key(row.get("product_name") or "", brand)
|
||||
url = cache.existing.get(key)
|
||||
if not url:
|
||||
return False
|
||||
choice = image_corroboration.choose_primary([url], row.get("product_name") or "", brand)
|
||||
if choice.primary is None:
|
||||
return False
|
||||
row["image_url"] = choice.primary
|
||||
row["image_urls"] = [choice.primary]
|
||||
row.setdefault("_notes", []).append("image adopted from another pack size of this product")
|
||||
return True
|
||||
|
||||
|
||||
def _image_cache_for(rows: Sequence[Dict[str, Any]]) -> ImageSearchCache:
|
||||
"""The run's shared image cache, primed with what each brand already holds.
|
||||
|
||||
Reads each destination brand once (stage 11 reads it again for the merge;
|
||||
two reads of a small table beat threading the rows through five stages).
|
||||
A read failure primes nothing - the search still runs.
|
||||
"""
|
||||
existing: List[Dict[str, Any]] = []
|
||||
for brand in sorted({r.get("brand") for r in rows if r.get("brand")}):
|
||||
if brand == OWN_PRODUCTS_BRAND:
|
||||
continue
|
||||
try:
|
||||
for prior in get_products_by_brand(brand) or []:
|
||||
prior = dict(prior)
|
||||
prior.setdefault("brand", brand)
|
||||
existing.append(prior)
|
||||
except Exception as exc: # noqa: BLE001 - priming is an optimisation
|
||||
logger.warning("Could not read existing images for %s: %s", brand, exc)
|
||||
return ImageSearchCache(existing)
|
||||
|
||||
|
||||
def stage_6_images(row: Dict[str, Any], *, enabled: bool = True,
|
||||
cache: Optional[ImageSearchCache] = None) -> Dict[str, Any]:
|
||||
if not _blank(row.get("image_urls")) or not enabled:
|
||||
return row
|
||||
try:
|
||||
@@ -580,33 +720,61 @@ def stage_6_images(row: Dict[str, Any], *, enabled: bool = True) -> Dict[str, An
|
||||
# drops the "-plant -tree -fish -botanical" Wikimedia exclusion block
|
||||
# that fights every produce query. Without it this stage recreates the
|
||||
# exact defect the Own Products image repair exists to undo.
|
||||
candidates = find_all_image_urls(
|
||||
product_name, brand=brand or None, max_results=24, produce=is_produce
|
||||
)
|
||||
if candidates:
|
||||
best = catalog_engine._select_best_images(
|
||||
candidates, product_name, brand, max_images=10
|
||||
# A pack size of this product that is already in the table may carry
|
||||
# the photo. Take it - through the gate - before spending a search:
|
||||
# the same product in a different pack is the same photo, every size
|
||||
# then agrees, and a re-ingest costs no network. See _adopt_sibling_image.
|
||||
if cache is not None and not is_produce and _adopt_sibling_image(row, brand, cache):
|
||||
return row
|
||||
|
||||
# One search per product, not per pack size - see ImageSearchCache.
|
||||
# The query is the size-less name: a photo's filename or alt text
|
||||
# almost never carries the pack size, and a size token in the query
|
||||
# only steers the engines towards pages LISTING that size, which for
|
||||
# "Honey 1kg" was every other brand's 1kg honey.
|
||||
key = image_corroboration.search_key(product_name, brand)
|
||||
base_name = image_corroboration.SIZE_TAIL.sub("", product_name).strip() or product_name
|
||||
if cache is not None and key in cache.candidates:
|
||||
best = cache.candidates[key]
|
||||
else:
|
||||
candidates = find_all_image_urls(
|
||||
base_name, brand=brand or None, max_results=24, produce=is_produce
|
||||
)
|
||||
if best:
|
||||
# `_select_best_images` ORDERS candidates; it does not judge
|
||||
# whether any of them is this product. Its scoring awards points
|
||||
# for the brand appearing in the URL, so for a brand that is
|
||||
# also a personal name it actively rewarded the wrong photo -
|
||||
# `Anil_Kapoor_2019.jpg` outscored everything and became
|
||||
# `image_url`. choose_primary decides what may be promoted.
|
||||
#
|
||||
# The list is still stored in full. A product whose only
|
||||
# candidates are uncorroborated keeps them for review; it just
|
||||
# does not get one of them presented as fact.
|
||||
choice = image_corroboration.choose_primary(
|
||||
best, product_name, brand or ""
|
||||
best = catalog_engine._select_best_images(
|
||||
candidates, base_name, brand, max_images=10
|
||||
) if candidates else []
|
||||
if cache is not None:
|
||||
cache.candidates[key] = list(best)
|
||||
|
||||
if best:
|
||||
# `_select_best_images` ORDERS candidates; it does not judge
|
||||
# whether any of them is this product. Its scoring awards points
|
||||
# for the brand appearing in the URL, so for a brand that is
|
||||
# also a personal name it actively rewarded the wrong photo -
|
||||
# `Anil_Kapoor_2019.jpg` outscored everything and became
|
||||
# `image_url`. choose_primary decides what may be promoted.
|
||||
#
|
||||
# ONLY ELIGIBLE CANDIDATES ARE STORED. The rejects used to be kept
|
||||
# in `image_urls` "for review" - but the product card falls back to
|
||||
# `image_urls[0]` when `image_url` is empty and the modal shows the
|
||||
# list as a gallery, so the review pile was what the shopper saw:
|
||||
# four foreign honeys for "Dabur Honey 1kg", a mosquito repellent
|
||||
# for "Dabur Honey 500g". A reject is logged, not displayed.
|
||||
choice = image_corroboration.choose_primary(
|
||||
best, product_name, brand or ""
|
||||
)
|
||||
row["image_urls"] = list(choice.eligible)
|
||||
row["image_url"] = choice.primary
|
||||
if choice.rejected:
|
||||
row.setdefault("_notes", []).append(
|
||||
f"{len(choice.rejected)} image candidate(s) rejected: "
|
||||
"they do not name this product"
|
||||
)
|
||||
row["image_urls"] = list(choice.ordered)
|
||||
row["image_url"] = choice.primary
|
||||
if choice.primary is None:
|
||||
row.setdefault("_notes", []).append(
|
||||
f"no primary image: {choice.reason}"
|
||||
)
|
||||
if choice.primary is None:
|
||||
row.setdefault("_notes", []).append(
|
||||
f"no primary image: {choice.reason}"
|
||||
)
|
||||
|
||||
except Exception as exc: # noqa: BLE001 - an image is not worth the row
|
||||
# `warning`, not `debug`: at the default log level a debug line is
|
||||
# invisible, so a row that silently lost its images looked identical to
|
||||
@@ -731,18 +899,50 @@ def build_image_id(brand: str, product_name: str, size: str) -> str:
|
||||
return f"{_sanitize_name(brand)}_{slug}"
|
||||
|
||||
|
||||
def _fallback_description(brand: str, name: str, size: str, category: str) -> str:
|
||||
"""What the card says when neither the sheet nor the LLM said anything.
|
||||
|
||||
Factual by construction - only facts the row already holds: the product,
|
||||
its pack size, its category, its maker. Deliberately NOT
|
||||
`catalog_engine.generate_detailed_description`, which pads any product
|
||||
into eight hundred characters of "premium quality" and "exceptional
|
||||
taste" that nobody verified. An honest short line beats a confident essay.
|
||||
"""
|
||||
parts = [name.strip()]
|
||||
if size and size.lower() not in name.lower():
|
||||
parts.append(f"{size} pack")
|
||||
if category and category.lower() not in ("general", "uncategorized"):
|
||||
parts.append(category)
|
||||
if brand:
|
||||
# resolve_parent_brand hands back the alias KEY ("amul"), which is a
|
||||
# lookup token rather than a name. Title-case only that form; a brand
|
||||
# written with its own casing ("India Gate", "P&G") is left alone.
|
||||
parts.append(f"by {brand.title() if brand.islower() else brand}")
|
||||
return ", ".join(p for p in parts if p) + "."
|
||||
|
||||
|
||||
def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Project an enriched row onto the brand-table columns."""
|
||||
name = row.get("product_name") or row.get("title") or ""
|
||||
size = row.get("size") or "Standard"
|
||||
display = name if size.lower() in name.lower() else f"{name} {size}".strip()
|
||||
category = row.get("category") or "General"
|
||||
own = row.get("brand") == OWN_PRODUCTS_BRAND
|
||||
# "Standard" is the pipeline's internal no-size sentinel (see _sizes_for).
|
||||
# Own Products keeps it in the table: the seeded base list uses it and an
|
||||
# uploaded "Apple" has to build the same image_id as the seeded "Apple".
|
||||
# A branded row has no such counterpart, so the sentinel is dropped here
|
||||
# and the row is stored the way the sheet described it - "India Gate
|
||||
# Basmati Rice", size_variants [], image_id without a size suffix - rather
|
||||
# than as "India Gate Basmati Rice Standard".
|
||||
raw_size = row.get("size") or "Standard"
|
||||
size = raw_size if (own or raw_size != "Standard") else ""
|
||||
display = name if (not size or size.lower() in name.lower()) else f"{name} {size}".strip()
|
||||
category = row.get("category") or "General"
|
||||
# "Apple 500g from Own Products." is a sentence nobody wrote and nobody
|
||||
# wants, and "Own Products" is a bucket rather than a maker, so the
|
||||
# template reads as a false provenance claim. A commodity keeps whatever
|
||||
# description the sheet gave, including none.
|
||||
description = row.get("description") or (None if own else f"{display} from {row.get('brand')}.")
|
||||
description = row.get("description") or (
|
||||
None if own else _fallback_description(row.get("brand") or "", display, size, category)
|
||||
)
|
||||
# The embedding text. Interpolating a null description and the bucket name
|
||||
# would embed the literal "Own Products ... None", so a commodity is
|
||||
# described to the vector index by what actually identifies it.
|
||||
@@ -759,7 +959,7 @@ def _to_storage_row(row: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"image_url": row.get("image_url"),
|
||||
"image_urls": list(row.get("image_urls") or []),
|
||||
"price_range": row.get("price_range"),
|
||||
"size_variants": [size],
|
||||
"size_variants": [size] if size else [],
|
||||
"providers": list(row.get("providers") or []),
|
||||
"fssai_license": row.get("fssai_license"),
|
||||
"product_sku": row.get("product_sku"),
|
||||
@@ -1001,9 +1201,16 @@ def run_pipeline(
|
||||
progress(1, STAGE_NAMES[0], 0, len(prepared))
|
||||
prepared = [stage_1_brand_and_fssai(r) for r in prepared]
|
||||
|
||||
breaker = LlmBreaker() if use_llm else None
|
||||
for index, row in enumerate(prepared, start=1):
|
||||
stage_2_row_intake(row, use_llm=use_llm)
|
||||
stage_2_row_intake(row, use_llm=use_llm, breaker=breaker)
|
||||
progress(2, STAGE_NAMES[1], index, len(prepared))
|
||||
if breaker is not None and breaker.tripped:
|
||||
result.warnings.append(
|
||||
f"LLM descriptions stopped after {breaker.limit} consecutive failures "
|
||||
"(Ollama unreachable or too slow); the remaining rows use the factual "
|
||||
"template description."
|
||||
)
|
||||
|
||||
progress(3, STAGE_NAMES[2], 0, len(prepared))
|
||||
prepared = [stage_3_title_category(r) for r in prepared]
|
||||
@@ -1031,8 +1238,9 @@ def run_pipeline(
|
||||
for index, row in enumerate(exploded, start=1):
|
||||
stage_5_pricing(row)
|
||||
progress(5, STAGE_NAMES[4], index, len(exploded))
|
||||
image_cache = _image_cache_for(exploded) if fetch_images else None
|
||||
for index, row in enumerate(exploded, start=1):
|
||||
stage_6_images(row, enabled=fetch_images)
|
||||
stage_6_images(row, enabled=fetch_images, cache=image_cache)
|
||||
progress(6, STAGE_NAMES[5], index, len(exploded))
|
||||
for index, row in enumerate(exploded, start=1):
|
||||
stage_7_sku(row)
|
||||
|
||||
@@ -107,6 +107,7 @@ from app.infrastructure.settings import (
|
||||
)
|
||||
from app.services import active_brands, brand_store, ollama_service, retail_presence
|
||||
from app.services.brand_registry import (
|
||||
get_brand_store_domain,
|
||||
get_fssai_license,
|
||||
get_known_sub_brands,
|
||||
resolve_parent_brand,
|
||||
@@ -449,19 +450,29 @@ def _existing_title_index(brand: str) -> Dict[str, str]:
|
||||
# ---------------------------------------------------------------------------
|
||||
# Source A - Open Food Facts (ground truth)
|
||||
# ---------------------------------------------------------------------------
|
||||
class OpenFactsUnavailable(RuntimeError):
|
||||
"""Open Food Facts could not be consulted - as opposed to consulted and
|
||||
found empty. `discover_brand_products` turns this into a warning that says
|
||||
so, because "OFF has nothing for this brand" is a claim about the brand
|
||||
and "OFF was down" is not."""
|
||||
|
||||
|
||||
def _from_open_facts(brand: str, *, refresh: bool = False) -> List[Dict[str, Any]]:
|
||||
"""Real products for `brand`, with a real GTIN and often a real pack size.
|
||||
|
||||
`off_bulk.fetch_brand_corpus` is already disk-cached, retry-wrapped and
|
||||
paced at 2s between pages, and `scripts/backfill_barcodes_from_off.py`
|
||||
already imports it from outside the enrichment package, so reaching for it
|
||||
here is an established pattern rather than a new one.
|
||||
`off_bulk.fetch_brand_corpus_result` is disk-cached, retry-wrapped and
|
||||
paced at 6s between 100-row pages. Raises `OpenFactsUnavailable` when the fetch
|
||||
did not complete and produced nothing; anything else is returned as-is,
|
||||
an empty list meaning OFF genuinely has no products under this brand tag.
|
||||
"""
|
||||
try:
|
||||
hits = off_bulk.fetch_brand_corpus(brand, refresh=refresh)
|
||||
result = off_bulk.fetch_brand_corpus_result(brand, refresh=refresh)
|
||||
except Exception as exc: # noqa: BLE001 - a dead OFF must not kill discovery
|
||||
logger.warning("Open Food Facts lookup failed for %r: %s", brand, exc)
|
||||
return []
|
||||
raise OpenFactsUnavailable(str(exc) or exc.__class__.__name__) from exc
|
||||
if result.error and not result.hits:
|
||||
raise OpenFactsUnavailable(result.error)
|
||||
hits = result.hits
|
||||
|
||||
out: List[Dict[str, Any]] = []
|
||||
for hit in hits or ():
|
||||
@@ -759,10 +770,10 @@ def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str,
|
||||
which is what makes `image_id` reproducible - the name is half of it.
|
||||
|
||||
The LLM's list is a guess and comes last. Only when all three are empty is
|
||||
the cell left blank, which hands the decision to stage 4's
|
||||
`default_size_variants` - and that invents 100g/250g/500g, which
|
||||
`_sizes_for`'s own comment identifies as the mechanism behind observed
|
||||
catalogue drift. Leaving it blank is the last resort, not the default.
|
||||
the cell left blank; stage 4 then stores ONE unsized row (it no longer
|
||||
invents 100g/250g/500g - see `_sizes_for`, which names that invention as
|
||||
the mechanism behind observed catalogue drift). Blank is honest, but a
|
||||
real size is what makes the row a distinct pack, so it is worth finding.
|
||||
"""
|
||||
in_title = _TITLE_SIZE_RE.search(title or "")
|
||||
if in_title:
|
||||
@@ -850,25 +861,21 @@ def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int,
|
||||
# semantic search.
|
||||
#
|
||||
# A blank description is honest and already handled: `_to_storage_row`
|
||||
# substitutes "<name> <size> from <brand>." exactly as it does for a store
|
||||
# sheet that left the column empty.
|
||||
# substitutes a short factual line (name, size, category, brand) exactly
|
||||
# as it does for a store sheet that left the column empty.
|
||||
description = (candidate.get("description") or "").strip()
|
||||
|
||||
# A BARCODE IDENTIFIES ONE PACK, SO IT ONLY SURVIVES A KNOWN PACK SIZE.
|
||||
# With no real size, stage 4 falls back to `default_size_variants` and
|
||||
# invents 100g/250g/500g - and because every column is copied into each
|
||||
# exploded variant, one real GTIN would be stamped onto three packs, two of
|
||||
# which do not exist. That is a worse error than a blank barcode: it is
|
||||
# wrong data that looks authoritative, and `nutrition_by_barcode` would
|
||||
# happily resolve all three to the same product.
|
||||
# A BARCODE IDENTIFIES ONE PACK. That used to mean dropping it whenever no
|
||||
# pack size was known, because stage 4 then invented 100g/250g/500g and
|
||||
# every column is copied into each exploded variant - one real GTIN
|
||||
# stamped onto three packs, two of which did not exist. Stage 4 now stores
|
||||
# a single unsized row when it has no size, and one GTIN on one row is
|
||||
# exactly what a GTIN means, so the barcode survives; the row is flagged
|
||||
# so a reviewer knows the pack size is still unknown.
|
||||
barcode = candidate.get("barcode")
|
||||
notes: List[str] = []
|
||||
if barcode and not sizes:
|
||||
notes.append(
|
||||
"barcode dropped: no real pack size is known, and stage 4 will "
|
||||
"invent several - a GTIN names one pack, not three"
|
||||
)
|
||||
barcode = None
|
||||
notes.append("pack size unknown: the barcode names one pack, but its size was not found")
|
||||
|
||||
shape = {"title": title, "category": category_hint, "size_variants": sizes,
|
||||
"description": description}
|
||||
@@ -929,12 +936,25 @@ def discover_brand_products(
|
||||
registry_terms = _registry_terms(brand)
|
||||
fssai = get_fssai_license(brand)
|
||||
|
||||
off_candidates = _from_open_facts(brand, refresh=refresh_corpus) if use_openfacts else []
|
||||
if use_openfacts and not off_candidates:
|
||||
warnings.append(
|
||||
"Open Food Facts returned nothing for this brand. Every row below "
|
||||
"rests on the language model alone - review them individually."
|
||||
)
|
||||
# Three distinct outcomes, three distinct warnings. "OFF has nothing under
|
||||
# this brand tag" is a fact about the brand; "OFF could not be reached" is
|
||||
# a fact about the network and must not be reported as the former - that
|
||||
# is how Naga (2 real rows on OFF) was being shown as unknown to it.
|
||||
off_candidates: List[Dict[str, Any]] = []
|
||||
if use_openfacts:
|
||||
try:
|
||||
off_candidates = _from_open_facts(brand, refresh=refresh_corpus)
|
||||
except OpenFactsUnavailable as exc:
|
||||
warnings.append(
|
||||
f"Open Food Facts could not be reached ({exc}). Nothing was "
|
||||
"cached, so the next preview will try again; the rows below "
|
||||
"come from the brand's shop and the language model only."
|
||||
)
|
||||
else:
|
||||
if not off_candidates:
|
||||
warnings.append(
|
||||
"Open Food Facts has no products tagged with this brand."
|
||||
)
|
||||
off_keys = {}
|
||||
for candidate in off_candidates:
|
||||
key = _normalise_title(brand, candidate["title"])
|
||||
@@ -951,12 +971,26 @@ def discover_brand_products(
|
||||
if store_candidates:
|
||||
logger.info("%s: %d products from the brand's own shop", brand, len(store_candidates))
|
||||
elif use_store and not off_candidates:
|
||||
# Say what was actually checked. An unregistered storefront is not a
|
||||
# storefront that "has nothing" - nobody has looked.
|
||||
store_domain = get_brand_store_domain(brand)
|
||||
if store_domain:
|
||||
warnings.append(
|
||||
f"{store_domain} is registered as this brand's shop but its "
|
||||
"catalogue has not been fetched - run "
|
||||
f"scripts/backfill_brand_stores.py --brand \"{brand}\" to load it."
|
||||
)
|
||||
else:
|
||||
warnings.append(
|
||||
"No verified storefront is registered for this brand. If it "
|
||||
"has its own online shop, add it to "
|
||||
"brand_registry.BRAND_STORE_DOMAINS and run "
|
||||
"scripts/backfill_brand_stores.py to populate a real catalogue."
|
||||
)
|
||||
if not off_candidates and not store_candidates:
|
||||
warnings.append(
|
||||
"Neither Open Food Facts nor a brand storefront has anything for "
|
||||
"this brand. Every row below rests on the language model alone - "
|
||||
"review them individually. If this brand has its own online shop, "
|
||||
"adding it to brand_registry.BRAND_STORE_DOMAINS and running "
|
||||
"scripts/backfill_brand_stores.py will populate a real catalogue."
|
||||
"Every row below rests on the language model alone - review them "
|
||||
"individually."
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -11,9 +11,12 @@ never heard of there is nothing to verify, and the catalogue is whatever
|
||||
`qwen2.5:1.5b` invents.
|
||||
|
||||
Measured 2026-09-10, Open Food Facts hits: Udhaiyam 0, Tenali Double Horse 0,
|
||||
Gopuram 0, Double Horse 14, and Naga 2 - where both "Naga" rows are a UK /
|
||||
Bangladeshi pickle brand, not the Tamil Nadu one. Regional South Indian brands
|
||||
are simply not in that database.
|
||||
Gopuram 0, Double Horse 14. (An earlier note here said Naga's two rows were a
|
||||
UK / Bangladeshi pickle brand - that was the old free-text `brands:Naga`
|
||||
Search-a-licious query matching "Mr Naga" and "Bombay Naga Jhal". Under the
|
||||
exact `brands_tags=naga` filter the v2 API returns two real Naga Limited rows,
|
||||
Sooji 500 g and Maida 500 g, both on the 890 GS1 prefix - see off_bulk.py.)
|
||||
Regional South Indian brands are still thin in that database.
|
||||
|
||||
They are, however, on their own shop, and modern storefronts publish their
|
||||
whole catalogue as structured JSON:
|
||||
|
||||
@@ -10,34 +10,50 @@ resulting cascade "misses far more often than it hits". That module is wired
|
||||
into the live enrichment pipeline and is deliberately NOT touched by this file.
|
||||
|
||||
This module inverts the problem: fetch a brand's **entire** OFF catalogue in one
|
||||
or two requests from the Search-a-licious endpoint, cache it on disk, then match
|
||||
every one of our products against that corpus offline. A whole-catalogue
|
||||
backfill costs ~5 HTTP requests instead of ~600, and re-tuning the similarity
|
||||
threshold costs zero network because the corpus is cached.
|
||||
or two requests from the v2 search API, cache it on disk, then match every one
|
||||
of our products against that corpus offline. A whole-catalogue backfill costs
|
||||
~5 HTTP requests instead of ~600, and re-tuning the similarity threshold costs
|
||||
zero network because the corpus is cached.
|
||||
|
||||
Nothing here is imported by the running app - the only consumer is
|
||||
`scripts/backfill_barcodes_from_off.py`. Everything except `fetch_brand_corpus`
|
||||
is a pure function so it can be unit-tested without network or database.
|
||||
`fetch_brand_corpus` is the one network function here and is shared by the
|
||||
backfill scripts, `product_grounding`, `post_ingest_barcodes` and
|
||||
`brand_discovery`. Everything else is a pure function so it can be unit-tested
|
||||
without network or database.
|
||||
|
||||
ENDPOINT NOTES (verified empirically, 2026-09)
|
||||
GET https://search.openfoodfacts.org/search
|
||||
?q=brands:amul AND countries_tags:"en:india"
|
||||
ENDPOINT NOTES (verified empirically, 2026-09-11)
|
||||
GET https://world.openfoodfacts.org/api/v2/search
|
||||
?brands_tags=amul
|
||||
&countries_tags=en:india
|
||||
&fields=code,product_name,quantity,...
|
||||
&page_size=250
|
||||
* The double quotes around "en:india" are REQUIRED. Without them the query
|
||||
parses as a bare term and silently returns count=0 rather than erroring.
|
||||
* page_size up to 1000 is accepted; 250 keeps responses small.
|
||||
&page_size=100&page=1
|
||||
* `brands_tags` is an EXACT match on the brand's tag slug ("naga", not
|
||||
"Naga"), which is what a brand catalogue needs. The earlier Search-a-licious
|
||||
query `q=brands:Naga` was a free-text match: it returned "Mr Naga" and
|
||||
"Bombay Naga Jhal" - other companies - and MISSED the real Naga rows,
|
||||
because Search-a-licious reads a separate Elasticsearch index that was
|
||||
stale for them (barcode 8906011830068 was still filed brandless under
|
||||
Kuwait). The v2 API reads the live product database. Measured on the same
|
||||
day: Amul 216 products here vs 140 there; Naga 2 vs 0.
|
||||
* `page_count` in this API is the number of products ON THIS PAGE, not the
|
||||
number of pages, and `page_size` in the RESPONSE is what the server
|
||||
actually applied (a larger request is capped to 100 without complaint).
|
||||
Paginate from `count` / that served `page_size`.
|
||||
* OFF rate-limits all search endpoints to 10 requests/minute per IP.
|
||||
PAUSE_SECONDS keeps a multi-page brand under that.
|
||||
* `world.openfoodfacts.org` intermittently serves an HTML "Page temporarily
|
||||
unavailable" page with a 200 status, so every response is content-type
|
||||
checked before parsing.
|
||||
unavailable" page with a 200 status, and plain 503s under load, so every
|
||||
response is status- and content-type-checked before parsing, and a failed
|
||||
fetch is never written to the cache as an empty corpus.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
|
||||
|
||||
@@ -63,21 +79,50 @@ from app.services.quantity_utils import quantities_match
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SEARCH_URL = "https://search.openfoodfacts.org/search"
|
||||
SEARCH_URL = "https://world.openfoodfacts.org/api/v2/search"
|
||||
|
||||
# OFF's usage policy requires a contactable custom User-Agent; requests sent
|
||||
# with the default python-requests agent are treated as anonymous crawling.
|
||||
USER_AGENT = "BrandCatalogRAG/1.0 (suriya@tenext.in)"
|
||||
|
||||
FIELDS = "code,product_name,product_name_en,brands,quantity,countries_tags"
|
||||
PAGE_SIZE = 250
|
||||
MAX_PAGES = 10 # 2500 products per brand is far beyond any real brand
|
||||
PAUSE_SECONDS = 2.0 # search endpoints are the strictly-limited ones
|
||||
PAGE_SIZE = 100 # the v2 API caps this at 100 and silently serves that
|
||||
MAX_PAGES = 25 # 2500 products per brand is far beyond any real brand
|
||||
PAUSE_SECONDS = 6.0 # 10 search requests/minute is OFF's published limit
|
||||
|
||||
# Bumped when the cache file's meaning changes. Files written before this key
|
||||
# existed came from the Search-a-licious endpoint; an EMPTY one of those is
|
||||
# more likely to be that endpoint's stale index (or a swallowed 503) than a
|
||||
# real absence, so it is re-fetched. A non-empty one is real data and is kept.
|
||||
CACHE_SCHEMA = 2
|
||||
|
||||
# backend/app/services/enrichment/barcode/sources/off_bulk.py -> backend/
|
||||
_BACKEND_DIR = Path(__file__).resolve().parents[5]
|
||||
CACHE_DIR = _BACKEND_DIR / "data" / "cache" / "off_brand_corpus"
|
||||
|
||||
|
||||
class OffUnavailable(requests.exceptions.ConnectionError):
|
||||
"""OFF answered but could not serve (429 / 5xx).
|
||||
|
||||
Subclasses ConnectionError so `retry.with_retry` treats it exactly like a
|
||||
dropped socket - a few seconds later the same request usually succeeds -
|
||||
without widening the retry set for every other barcode source.
|
||||
"""
|
||||
|
||||
|
||||
@dataclass
|
||||
class CorpusFetch:
|
||||
"""What `fetch_brand_corpus_result` learned.
|
||||
|
||||
`error` is None when OFF answered every page, even if it answered with
|
||||
nothing - that is a real "this brand is not on Open Food Facts". When
|
||||
`error` is set the fetch did not complete and NOTHING was cached, so a
|
||||
caller can say "could not be reached" instead of "has nothing".
|
||||
"""
|
||||
hits: List[Dict[str, Any]] = field(default_factory=list)
|
||||
error: Optional[str] = None
|
||||
from_cache: bool = False
|
||||
|
||||
# Pack sizes embedded in a product name ("Marie Gold 250g", "Butter 1L").
|
||||
# Mirrors the unit list in scripts/backfill_nutrition_from_barcodes.py, widened
|
||||
# with the count-based units this catalog also uses.
|
||||
@@ -122,84 +167,149 @@ def _cache_path(slug: str, cache_dir: Optional[Path] = None) -> Path:
|
||||
return (cache_dir or CACHE_DIR) / f"{slug}.json"
|
||||
|
||||
|
||||
def brand_tag_slug(brand: str) -> str:
|
||||
"""The brand as Open Food Facts tags it: lowercase, runs of anything that
|
||||
is not a letter or digit collapsed to one hyphen. "Hindustan Unilever" ->
|
||||
"hindustan-unilever", "P&G" -> "p-g", "Naga" -> "naga". OFF matches
|
||||
`brands_tags` case-insensitively, but sending the slug form is what its
|
||||
own site does and avoids depending on that."""
|
||||
return re.sub(r"[^a-z0-9]+", "-", (brand or "").lower()).strip("-")
|
||||
|
||||
|
||||
@with_retry(max_attempts=3, min_wait=2.0, max_wait=10.0)
|
||||
def _get_page(brand: str, country: Optional[str], page: int) -> Dict[str, Any]:
|
||||
"""One page of OFF search results. Raises on transport errors (retried by
|
||||
the decorator); returns an empty result dict for any response that is not
|
||||
parseable JSON, which is how the HTML "temporarily unavailable" page and
|
||||
any future error page are absorbed without killing the run."""
|
||||
query = f"brands:{brand}"
|
||||
def _get_page(brand: str, country: Optional[str], page: int) -> Optional[Dict[str, Any]]:
|
||||
"""One page of OFF v2 search results.
|
||||
|
||||
Raises on transport errors and on 429 / 5xx (both retried by the
|
||||
decorator); returns None for any other response that is not parseable
|
||||
JSON - an HTML "temporarily unavailable" page, a 4xx, a truncated body.
|
||||
None means "this page FAILED", which the caller must keep distinct from a
|
||||
page that parsed fine and simply held no products.
|
||||
"""
|
||||
params: Dict[str, Any] = {
|
||||
"brands_tags": brand_tag_slug(brand),
|
||||
"fields": FIELDS,
|
||||
"page_size": PAGE_SIZE,
|
||||
"page": page,
|
||||
}
|
||||
if country:
|
||||
# The quotes are load-bearing - see the module docstring.
|
||||
query += f' AND countries_tags:"en:{country}"'
|
||||
params["countries_tags"] = f"en:{country}"
|
||||
|
||||
resp = requests.get(
|
||||
SEARCH_URL,
|
||||
params={"q": query, "fields": FIELDS, "page_size": PAGE_SIZE, "page": page},
|
||||
params=params,
|
||||
headers={"User-Agent": USER_AGENT, "Accept": "application/json"},
|
||||
timeout=BARCODE_LOOKUP_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
if resp.status_code == 429 or resp.status_code >= 500:
|
||||
raise OffUnavailable(f"HTTP {resp.status_code} from Open Food Facts")
|
||||
if resp.status_code != 200:
|
||||
logger.warning("OFF search returned HTTP %s for brand %r page %s",
|
||||
resp.status_code, brand, page)
|
||||
return {}
|
||||
return None
|
||||
if "json" not in (resp.headers.get("content-type") or "").lower():
|
||||
logger.warning("OFF search returned non-JSON (%s) for brand %r - "
|
||||
"the service is probably serving an error page",
|
||||
resp.headers.get("content-type"), brand)
|
||||
return {}
|
||||
return None
|
||||
try:
|
||||
payload = resp.json()
|
||||
except ValueError as e:
|
||||
logger.warning("OFF search returned unparseable JSON for brand %r: %s", brand, e)
|
||||
return {}
|
||||
return payload if isinstance(payload, dict) else {}
|
||||
return None
|
||||
return payload if isinstance(payload, dict) else None
|
||||
|
||||
|
||||
def fetch_brand_corpus(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
|
||||
"""Every OFF product for `brand`, from disk cache unless `refresh`.
|
||||
def _read_cache(path: Path, brand: str) -> Optional[List[Dict[str, Any]]]:
|
||||
"""Cached hits, or None when the cache must not be trusted: unreadable, or
|
||||
an empty file from before CACHE_SCHEMA existed (see that constant)."""
|
||||
try:
|
||||
cached = json.loads(path.read_text(encoding="utf-8"))
|
||||
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
|
||||
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
|
||||
return None
|
||||
hits = cached.get("hits") or []
|
||||
if not hits and cached.get("schema") != CACHE_SCHEMA:
|
||||
logger.info(" Ignoring empty pre-v%s OFF cache for %r - re-fetching",
|
||||
CACHE_SCHEMA, brand)
|
||||
return None
|
||||
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
|
||||
brand, len(hits), cached.get("fetched_at_human", "?"))
|
||||
return hits
|
||||
|
||||
|
||||
def fetch_brand_corpus_result(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> CorpusFetch:
|
||||
"""Every OFF product for `brand`, from disk cache unless `refresh`, plus
|
||||
whether the fetch actually completed.
|
||||
|
||||
Hits with no usable product name are dropped here rather than at match time
|
||||
(6 of 146 Amul hits, 1 of 233 Britannia hits) - a nameless hit can never
|
||||
clear a name-similarity threshold, so carrying it forward only inflates the
|
||||
corpus. Returns [] rather than raising when OFF is unreachable, so one bad
|
||||
brand does not abort a multi-brand backfill.
|
||||
corpus.
|
||||
|
||||
A fetch that fails part-way returns what it got with `error` set and
|
||||
writes NO cache file. Caching a failure as `hits: []` is how "Open Food
|
||||
Facts has nothing for this brand" was being asserted for brands OFF had
|
||||
simply been too busy to answer about.
|
||||
"""
|
||||
country = BARCODE_COUNTRY_TAG if country is None else (country or None)
|
||||
slug = re.sub(r"[^a-z0-9]+", "_", brand.lower()).strip("_")
|
||||
path = _cache_path(slug, cache_dir)
|
||||
|
||||
if not refresh and path.exists():
|
||||
try:
|
||||
cached = json.loads(path.read_text(encoding="utf-8"))
|
||||
hits = cached.get("hits") or []
|
||||
logger.info(" OFF corpus for %r: %d product(s) (cached %s)",
|
||||
brand, len(hits), cached.get("fetched_at_human", "?"))
|
||||
return hits
|
||||
except Exception as e: # noqa: BLE001 - a corrupt cache must not be fatal
|
||||
logger.warning(" Ignoring unreadable OFF cache %s: %s", path.name, e)
|
||||
cached = _read_cache(path, brand)
|
||||
if cached is not None:
|
||||
return CorpusFetch(hits=cached, from_cache=True)
|
||||
|
||||
hits: List[Dict[str, Any]] = []
|
||||
error: Optional[str] = None
|
||||
page = 1
|
||||
while page <= MAX_PAGES:
|
||||
total_pages = 1
|
||||
while page <= min(total_pages, MAX_PAGES):
|
||||
if page > 1:
|
||||
time.sleep(PAUSE_SECONDS)
|
||||
payload = _get_page(brand, country, page)
|
||||
batch = payload.get("hits") or []
|
||||
try:
|
||||
payload = _get_page(brand, country, page)
|
||||
except requests.exceptions.RequestException as e:
|
||||
# Retries exhausted (OffUnavailable is a ConnectionError too).
|
||||
error = str(e) or e.__class__.__name__
|
||||
break
|
||||
if payload is None:
|
||||
error = "Open Food Facts returned an unusable response"
|
||||
break
|
||||
batch = payload.get("products") or []
|
||||
hits.extend(h for h in batch if isinstance(h, dict) and (h.get("product_name") or "").strip())
|
||||
page_count = payload.get("page_count") or 0
|
||||
if page >= page_count or not batch:
|
||||
if page == 1:
|
||||
# `page_count` is products-on-this-page in the v2 API; the real
|
||||
# page total is `count` over the page size the SERVER applied -
|
||||
# it silently caps the requested size (250 asked, 100 served for
|
||||
# Amul), so dividing by PAGE_SIZE under-pages.
|
||||
count = payload.get("count") or 0
|
||||
served = payload.get("page_size") or len(batch) or PAGE_SIZE
|
||||
try:
|
||||
total_pages = max(1, math.ceil(int(count) / int(served)))
|
||||
except (TypeError, ValueError, ZeroDivisionError):
|
||||
total_pages = 1
|
||||
if not batch:
|
||||
break
|
||||
page += 1
|
||||
|
||||
if error:
|
||||
logger.warning(" OFF corpus for %r: fetch failed after %d usable product(s) - %s "
|
||||
"(nothing cached)", brand, len(hits), error)
|
||||
return CorpusFetch(hits=hits, error=error)
|
||||
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = path.with_suffix(".json.tmp")
|
||||
tmp.write_text(json.dumps({
|
||||
"schema": CACHE_SCHEMA,
|
||||
"endpoint": SEARCH_URL,
|
||||
"brand": brand,
|
||||
"brand_tag": brand_tag_slug(brand),
|
||||
"country": country,
|
||||
"fetched_at": time.time(),
|
||||
"fetched_at_human": time.strftime("%Y-%m-%d %H:%M:%S"),
|
||||
@@ -208,7 +318,20 @@ def fetch_brand_corpus(brand: str,
|
||||
os.replace(tmp, path)
|
||||
|
||||
logger.info(" OFF corpus for %r: %d usable product(s) fetched", brand, len(hits))
|
||||
return hits
|
||||
return CorpusFetch(hits=hits)
|
||||
|
||||
|
||||
def fetch_brand_corpus(brand: str,
|
||||
country: Optional[str] = None,
|
||||
refresh: bool = False,
|
||||
cache_dir: Optional[Path] = None) -> List[Dict[str, Any]]:
|
||||
"""`fetch_brand_corpus_result(...).hits` - the list-only form every
|
||||
backfill and ingestion caller uses. Returns [] rather than raising when
|
||||
OFF is unreachable, so one bad brand does not abort a multi-brand run;
|
||||
callers that need to tell "unreachable" from "empty" use the result form.
|
||||
"""
|
||||
return fetch_brand_corpus_result(brand, country=country, refresh=refresh,
|
||||
cache_dir=cache_dir).hits
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -243,6 +243,16 @@ def _round2(value: Optional[float]) -> Optional[float]:
|
||||
return round(value, 2)
|
||||
|
||||
|
||||
def _held_number(value: object) -> Optional[float]:
|
||||
"""A price the row already carries, or None for blank / unparseable."""
|
||||
if value is None or (isinstance(value, str) and not value.strip()):
|
||||
return None
|
||||
try:
|
||||
return float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def enrich_pricing_fields(product: dict) -> Dict[str, object]:
|
||||
"""Compute the pricing fields added by this stage for one catalog row.
|
||||
|
||||
@@ -253,8 +263,24 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
|
||||
`cost_price`, `profit_before_tax` and `profit_after_tax` are not
|
||||
derivable from the data this pipeline generates, so they stay None
|
||||
(exactly as the existing enriched exports store them).
|
||||
|
||||
A PRICE THE ROW ALREADY HOLDS WINS. A store sheet that says "Retail Price
|
||||
155" has stated a fact; the band derived from it is Rs143-167, and this
|
||||
stage used to read the band's ceiling back as "selling_price = 167" and
|
||||
hand that to `EnrichmentStage.apply`, which overwrites a held value with
|
||||
a different one (it only refuses to BLANK one). Held prices are therefore
|
||||
returned as None here so `apply` keeps them, the base for the tax is the
|
||||
held selling price, then the held final price, and only then the band.
|
||||
"""
|
||||
selling_price = _extract_selling_price(product.get("price_range"))
|
||||
held_selling = _held_number(product.get("selling_price"))
|
||||
held_final = _held_number(product.get("final_selling_price"))
|
||||
if held_selling is not None:
|
||||
selling_price = held_selling
|
||||
elif held_final is not None:
|
||||
selling_price = held_final
|
||||
else:
|
||||
selling_price = _extract_selling_price(product.get("price_range"))
|
||||
|
||||
gst = product.get("gst_percent")
|
||||
tax_amount = None
|
||||
final_selling_price = None
|
||||
@@ -263,10 +289,10 @@ def enrich_pricing_fields(product: dict) -> Dict[str, object]:
|
||||
final_selling_price = _round2(selling_price + tax_amount)
|
||||
|
||||
return {
|
||||
"selling_price": _round2(selling_price),
|
||||
"selling_price": None if held_selling is not None else _round2(selling_price),
|
||||
"cost_price": None,
|
||||
"tax_amount": tax_amount,
|
||||
"final_selling_price": final_selling_price,
|
||||
"final_selling_price": None if held_final is not None else final_selling_price,
|
||||
"profit_before_tax": None,
|
||||
"profit_after_tax": None,
|
||||
}
|
||||
|
||||
@@ -76,7 +76,7 @@ logger = logging.getLogger(__name__)
|
||||
# Matches scripts/backfill_barcodes_from_off.py, which measured it.
|
||||
DEFAULT_MIN_SIMILARITY = 0.88
|
||||
|
||||
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
|
||||
BARCODE_SOURCE = "openfoodfacts_bulk (world.openfoodfacts.org/api/v2)"
|
||||
|
||||
|
||||
def _rows_needing_a_barcode(cur, table: str) -> List[Dict[str, Any]]:
|
||||
|
||||
@@ -63,7 +63,7 @@ import sqlite3
|
||||
import threading
|
||||
import time
|
||||
from contextlib import closing
|
||||
from dataclasses import dataclass
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Iterable, List, Optional, Sequence
|
||||
from urllib.parse import urlparse
|
||||
@@ -175,6 +175,23 @@ def distinctive_tokens(product_name: str, brand: str) -> List[str]:
|
||||
return out
|
||||
|
||||
|
||||
def _token_in(token: str, lowered_url: str) -> bool:
|
||||
"""Does the URL carry this word, allowing for the plural on either side?
|
||||
|
||||
The catalogue names "Aachi Appalams" and "Aachi Pickles"; the photo is
|
||||
`Aachi-Appalam-100-g-1.webp`. A whole-token substring test rejected the
|
||||
correct image for the plural and then promoted an opaque Amazon URL over
|
||||
it. The singular is tried as well - only for tokens long enough that
|
||||
stripping the "s" leaves a real word ("gems" -> "gem" is fine; "kgs" never
|
||||
gets here, size tokens are removed upstream).
|
||||
"""
|
||||
if token in lowered_url:
|
||||
return True
|
||||
if len(token) > 4 and token.endswith("s") and token[:-1] in lowered_url:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def search_key(product_name: str, brand: str) -> tuple:
|
||||
"""Collapse a trailing pack size so sizes of one product share a lookup.
|
||||
|
||||
@@ -213,7 +230,7 @@ def corroborate(url: str, product_name: str, brand: str) -> Corroboration:
|
||||
|
||||
distinctive = distinctive_tokens(product_name, brand)
|
||||
if distinctive:
|
||||
hit = next((t for t in distinctive if t in lowered), None)
|
||||
hit = next((t for t in distinctive if _token_in(t, lowered)), None)
|
||||
if hit:
|
||||
return Corroboration(True, f"url names {hit!r}")
|
||||
return Corroboration(
|
||||
@@ -272,6 +289,26 @@ _lock = threading.Lock()
|
||||
_initialized = False
|
||||
_CACHE_TTL_SECONDS = 30 * 24 * 3600
|
||||
|
||||
# Open*Facts allows 100 product reads a minute per IP. A brand ingestion or a
|
||||
# re-gate asks about every distinct barcode it meets, and the Dabur run issued
|
||||
# about a hundred in two minutes - the tail was throttled, and a throttled
|
||||
# lookup fails open, which is how a Kellogg's honey got past the gate for a
|
||||
# moment. Pacing the calls keeps the gate answering instead of guessing.
|
||||
_OFF_MIN_INTERVAL_SECONDS = 0.65
|
||||
_OFF_LOOKUP_ATTEMPTS = 2
|
||||
_OFF_RETRY_SLEEP_SECONDS = 3.0
|
||||
_off_last_call = 0.0
|
||||
_off_pace_lock = threading.Lock()
|
||||
|
||||
|
||||
def _pace_off_lookup() -> None:
|
||||
global _off_last_call
|
||||
with _off_pace_lock:
|
||||
wait = _OFF_MIN_INTERVAL_SECONDS - (time.monotonic() - _off_last_call)
|
||||
if wait > 0:
|
||||
time.sleep(wait)
|
||||
_off_last_call = time.monotonic()
|
||||
|
||||
|
||||
def _connect() -> sqlite3.Connection:
|
||||
_DB_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
@@ -366,29 +403,58 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
|
||||
if not tokens:
|
||||
return True
|
||||
|
||||
key = f"{api_host}:{barcode}:{(brand or '').lower()}"
|
||||
# "v2:" - entries written before the fail-open verdicts stopped being
|
||||
# cached are ignored rather than trusted; they age out with the TTL.
|
||||
key = f"v2:{api_host}:{barcode}:{(brand or '').lower()}"
|
||||
cached = _cache_get(key)
|
||||
if cached is not None:
|
||||
return cached
|
||||
|
||||
verdict = True
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"https://{api_host}/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=timeout,
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
)
|
||||
# ONLY A REAL ANSWER IS CACHED. The fail-open True for a lookup that did
|
||||
# not happen - a 429 or 503 from Open*Facts, a non-JSON body - used to be
|
||||
# written to the cache too, for thirty days. A Dabur ingestion issued a
|
||||
# hundred lookups in two minutes, the tail of them were throttled, and
|
||||
# Kellogg's "Miel Pops", a Toblerone and a Nature Valley bar were filed as
|
||||
# Dabur products; "Dabur Honey 1kg" then showed the Kellogg's honey. A
|
||||
# throttled lookup still fails open for THIS call, but the next call asks
|
||||
# again.
|
||||
payload = None
|
||||
for attempt in range(_OFF_LOOKUP_ATTEMPTS):
|
||||
_pace_off_lookup()
|
||||
try:
|
||||
resp = requests.get(
|
||||
f"https://{api_host}/api/v2/product/{barcode}.json",
|
||||
params={"fields": "brands,product_name"},
|
||||
timeout=timeout,
|
||||
headers={"User-Agent": _BROWSER_UA},
|
||||
)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
return True
|
||||
if resp.ok:
|
||||
product = (resp.json() or {}).get("product") or {}
|
||||
haystack = (
|
||||
f"{product.get('brands') or ''} {product.get('product_name') or ''}"
|
||||
).lower()
|
||||
if haystack.strip():
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
except Exception: # noqa: BLE001 - a lookup failure must not reject a good image
|
||||
try:
|
||||
payload = resp.json() or {}
|
||||
except ValueError:
|
||||
return True
|
||||
break
|
||||
if resp.status_code == 429 or resp.status_code >= 500:
|
||||
# Throttled or unwell. One paced retry is cheap and turns most of
|
||||
# these into a real answer; a second failure fails open, uncached.
|
||||
time.sleep(_OFF_RETRY_SLEEP_SECONDS * (attempt + 1))
|
||||
continue
|
||||
return True
|
||||
if payload is None:
|
||||
return True
|
||||
|
||||
product = payload.get("product") or {}
|
||||
if not product and payload.get("status") == 0:
|
||||
# Open*Facts positively says: no such barcode. Nothing to compare
|
||||
# against, and asking again will not change that - cache the open verdict.
|
||||
_cache_set(key, True)
|
||||
return True
|
||||
haystack = f"{product.get('brands') or ''} {product.get('product_name') or ''}".lower()
|
||||
if not haystack.strip():
|
||||
return True
|
||||
verdict = any(t in haystack for t in tokens)
|
||||
_cache_set(key, verdict)
|
||||
return verdict
|
||||
|
||||
@@ -398,11 +464,25 @@ def openfacts_product_matches_brand(url: str, brand: str, *, timeout: int = 10)
|
||||
# ---------------------------------------------------------------------------
|
||||
@dataclass
|
||||
class PrimaryChoice:
|
||||
"""The outcome of choosing a product's `image_url` from its candidates."""
|
||||
"""The outcome of choosing a product's `image_url` from its candidates.
|
||||
|
||||
`ordered` is every candidate, best first. `eligible` is the subset that
|
||||
may be SHOWN - tiers 1 and 2 - and is what the store pipeline persists as
|
||||
`image_urls`. The two differ for a reason that was learned the hard way:
|
||||
tier 3 was kept in `image_urls` "for review, never promoted", but the
|
||||
product card falls back to `image_urls[0]` whenever `image_url` is empty
|
||||
and the product modal shows the whole list as a gallery. So for "Dabur
|
||||
Honey 1kg" the gate correctly withheld the primary - every candidate was
|
||||
another company's honey - and the UI displayed those very honeys anyway.
|
||||
A candidate the gate would not promote must not be stored where the UI
|
||||
will promote it.
|
||||
"""
|
||||
|
||||
primary: Optional[str]
|
||||
ordered: List[str]
|
||||
reason: str
|
||||
eligible: List[str] = field(default_factory=list)
|
||||
rejected: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def choose_primary(
|
||||
@@ -424,7 +504,10 @@ def choose_primary(
|
||||
catalog_engine's comment protects, where dropping uncorroborated
|
||||
candidates would leave real products with no image at all.
|
||||
3. The URL could have named the product and did not - a human-authored
|
||||
Commons filename, say. Kept in `image_urls`, never promoted.
|
||||
Commons filename, say - or an Open*Facts photo whose barcode belongs
|
||||
to another brand. Returned in `ordered` and `rejected`, never in
|
||||
`eligible`, and the store pipeline does not persist it (see
|
||||
PrimaryChoice for why "kept for review" was not safe).
|
||||
|
||||
Nothing eligible means `primary is None`. A blank image renders as the
|
||||
brand monogram, which is honest; another company's product is not, and it
|
||||
@@ -466,12 +549,16 @@ def choose_primary(
|
||||
tier3.sort(key=looks_like_person_photo)
|
||||
|
||||
ordered = tier1 + tier2 + tier3
|
||||
eligible = tier1 + tier2
|
||||
if tier1:
|
||||
return PrimaryChoice(tier1[0], ordered, "url names the product")
|
||||
return PrimaryChoice(tier1[0], ordered, "url names the product",
|
||||
eligible=eligible, rejected=tier3)
|
||||
if tier2:
|
||||
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host")
|
||||
return PrimaryChoice(tier2[0], ordered, "opaque path on a known image host",
|
||||
eligible=eligible, rejected=tier3)
|
||||
return PrimaryChoice(
|
||||
None,
|
||||
ordered,
|
||||
"no candidate names this product; primary withheld rather than guessed",
|
||||
eligible=eligible, rejected=tier3,
|
||||
)
|
||||
|
||||
@@ -114,11 +114,44 @@ def _query_openfacts(query: str, max_results: int) -> list:
|
||||
return []
|
||||
|
||||
|
||||
def _openfacts_product_is_brand(product: dict, brand: Optional[str]) -> bool:
|
||||
"""Does this Open*Facts record belong to `brand`?
|
||||
|
||||
Open*Facts' free-text search matches on the NAME, not the brand: asked for
|
||||
"Dabur Honey 1kg" it answered with a UK, a French, a Swiss and a Spanish
|
||||
honey, every one a real front-of-pack photo of somebody else's product.
|
||||
The record says who made it - `brands` / `brands_tags` - and that field
|
||||
used to be thrown away here, leaving a per-URL API round trip downstream
|
||||
(`image_corroboration.openfacts_product_matches_brand`) as the only thing
|
||||
standing between those photos and the catalogue. Check it at the source.
|
||||
|
||||
Fails OPEN on a record with no brand at all: an unlabelled record is not
|
||||
evidence of another brand, and the downstream gate still runs.
|
||||
"""
|
||||
if not brand:
|
||||
return True
|
||||
from app.services.image_corroboration import brand_tokens
|
||||
|
||||
tokens = brand_tokens(brand)
|
||||
if not tokens:
|
||||
return True
|
||||
tags = product.get("brands_tags") or []
|
||||
haystack = " ".join([str(product.get("brands") or "")] + [str(t) for t in tags]).lower()
|
||||
if not haystack.strip():
|
||||
return True
|
||||
return any(t in haystack for t in tokens)
|
||||
|
||||
|
||||
def find_images_openfacts(title: str, brand: Optional[str] = None, max_results: int = 20) -> list:
|
||||
"""Query the Open *Facts family of open product databases for real
|
||||
product photos. No API key required. Falls back from a brand+title
|
||||
query to a title-only query if the combined query is too specific to
|
||||
match anything (small/regional brand name variants are a common case)."""
|
||||
match anything (small/regional brand name variants are a common case).
|
||||
|
||||
Only records whose own `brands` field names our brand are used - see
|
||||
`_openfacts_product_is_brand`. The title-only fallback makes this filter
|
||||
load-bearing: "Honey 1kg" matches every honey on the site.
|
||||
"""
|
||||
if not USE_OPEN_FACTS:
|
||||
return []
|
||||
|
||||
@@ -126,12 +159,12 @@ def find_images_openfacts(title: str, brand: Optional[str] = None, max_results:
|
||||
if not query:
|
||||
return []
|
||||
|
||||
products = _query_openfacts(query, max_results)
|
||||
products = [p for p in _query_openfacts(query, max_results) if _openfacts_product_is_brand(p, brand)]
|
||||
if not products and brand and title:
|
||||
# Combined "brand + title" query found nothing - retry with just
|
||||
# the title, since Open*Facts' free-text search is exact-ish and
|
||||
# brand naming conventions vary (e.g. "Dettol" vs "Reckitt Dettol").
|
||||
products = _query_openfacts(title, max_results)
|
||||
products = [p for p in _query_openfacts(title, max_results) if _openfacts_product_is_brand(p, brand)]
|
||||
|
||||
urls: List[str] = []
|
||||
for product in products:
|
||||
|
||||
@@ -332,16 +332,29 @@ def fetch_brand_catalog_exhaustive(brand: str, max_products: int = 300) -> Dict[
|
||||
return {"brand": brand, "products": products_out}
|
||||
|
||||
|
||||
def fetch_product_details(brand: str, product_title: str) -> Dict[str, Any] | None:
|
||||
def fetch_product_details(brand: str, product_title: str,
|
||||
category: str | None = None,
|
||||
size: str | None = None) -> Dict[str, Any] | None:
|
||||
"""Get details for a single product: description, image_url, pricing fields.
|
||||
Returns a dict with keys: description, image_url, size_variants, price_ranges, price_range, provider_examples.
|
||||
|
||||
`category` and `size` are optional context. A 1.5b model asked about
|
||||
"Naga Maida" alone will happily describe a curry; told it is a 500g pack
|
||||
in Flours & Grains it describes refined wheat flour. Both are facts the
|
||||
caller already holds, so they cost nothing to pass.
|
||||
"""
|
||||
if not _ensure_client():
|
||||
return None
|
||||
context = f"Brand: {brand}\nProduct: {product_title}\n"
|
||||
if category:
|
||||
context += f"Category: {category}\n"
|
||||
if size:
|
||||
context += f"Pack size: {size}\n"
|
||||
user_prompt = (
|
||||
f"Brand: {brand}\nProduct: {product_title}\n"
|
||||
"Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
|
||||
"description must be <=160 chars, concise and factual."
|
||||
context
|
||||
+ "Return strictly JSON with keys: description, image_url?, size_variants?, price_ranges?, price_range?, provider_examples?.\n"
|
||||
"description: 1-2 sentences, <=220 chars, factual - what the product is, "
|
||||
"its form and its typical use. No marketing adjectives, no claims you cannot know."
|
||||
)
|
||||
text = _generate(SYSTEM_PROMPT, user_prompt)
|
||||
if not text:
|
||||
|
||||
@@ -5,7 +5,7 @@ WHY THIS FILE EXISTS
|
||||
--------------------
|
||||
Open Food Facts answers "does this product exist" for food, and answers it
|
||||
well. It does not answer it for anything else: `off_bulk` queries only
|
||||
`search.openfoodfacts.org`, so a toothpaste or a detergent has no
|
||||
Open Food Facts' food database, so a toothpaste or a detergent has no
|
||||
product-discovery source at all and its entire catalogue is language-model
|
||||
output. Measured 2026-09-10, India-tagged coverage on the sibling databases is
|
||||
2-13 products per non-food brand against Britannia's 218 on OFF - real, but
|
||||
|
||||
1588
data/brand_image_repair_backup_20260911_152831.json
Normal file
1588
data/brand_image_repair_backup_20260911_152831.json
Normal file
File diff suppressed because it is too large
Load Diff
BIN
data/cache/image_corroboration.db
vendored
BIN
data/cache/image_corroboration.db
vendored
Binary file not shown.
3042
data/cache/off_brand_corpus/amul.json
vendored
3042
data/cache/off_brand_corpus/amul.json
vendored
File diff suppressed because it is too large
Load Diff
31
data/cache/off_brand_corpus/naga.json
vendored
Normal file
31
data/cache/off_brand_corpus/naga.json
vendored
Normal file
@@ -0,0 +1,31 @@
|
||||
{
|
||||
"schema": 2,
|
||||
"endpoint": "https://world.openfoodfacts.org/api/v2/search",
|
||||
"brand": "Naga",
|
||||
"brand_tag": "naga",
|
||||
"country": "india",
|
||||
"fetched_at": 1789104574.7944398,
|
||||
"fetched_at_human": "2026-09-11 10:59:34",
|
||||
"hits": [
|
||||
{
|
||||
"brands": "Naga",
|
||||
"code": "8906011830068",
|
||||
"countries_tags": [
|
||||
"en:india"
|
||||
],
|
||||
"product_name": "Sooji",
|
||||
"product_name_en": "Sooji",
|
||||
"quantity": "500 g"
|
||||
},
|
||||
{
|
||||
"brands": "Naga",
|
||||
"code": "8906011831713",
|
||||
"countries_tags": [
|
||||
"en:india"
|
||||
],
|
||||
"product_name": "Maida",
|
||||
"product_name_en": "Maida",
|
||||
"quantity": "500 g"
|
||||
}
|
||||
]
|
||||
}
|
||||
10
data/cache/off_brand_corpus/testbrand.json
vendored
Normal file
10
data/cache/off_brand_corpus/testbrand.json
vendored
Normal file
@@ -0,0 +1,10 @@
|
||||
{
|
||||
"schema": 2,
|
||||
"endpoint": "https://world.openfoodfacts.org/api/v2/search",
|
||||
"brand": "TestBrand",
|
||||
"brand_tag": "testbrand",
|
||||
"country": "india",
|
||||
"fetched_at": 1789110182.5709507,
|
||||
"fetched_at_human": "2026-09-11 12:33:02",
|
||||
"hits": []
|
||||
}
|
||||
@@ -98,7 +98,7 @@ from app.services.vector_store import _connect
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("backfill_barcodes_off")
|
||||
|
||||
BARCODE_SOURCE = "openfoodfacts_bulk (search.openfoodfacts.org)"
|
||||
BARCODE_SOURCE = "openfoodfacts_bulk (world.openfoodfacts.org/api/v2)"
|
||||
|
||||
# The nine keys BarcodeResult.as_product_fields() emits. Named here so the JSON
|
||||
# writer can prove it touches nothing else.
|
||||
|
||||
@@ -30,7 +30,10 @@ USAGE
|
||||
python scripts/backfill_brand_stores.py --all # dry run
|
||||
python scripts/backfill_brand_stores.py --all --apply
|
||||
python scripts/backfill_brand_stores.py --brand Gopuram --apply --refresh
|
||||
python scripts/backfill_brand_stores.py --brand Naga --domain nagafoods.in --apply
|
||||
python scripts/backfill_brand_stores.py --brand Aachi --domain aachifoods.com --apply
|
||||
|
||||
(Naga has no confirmed storefront to try: nagalimited.com is a parked domain
|
||||
and nagafoods.in / nagafoods.com do not resolve, checked 2026-09-11.)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
212
scripts/regate_brand_images.py
Normal file
212
scripts/regate_brand_images.py
Normal file
@@ -0,0 +1,212 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Re-run the image gate over a brand's STORED images, and drop what fails it.
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
`scripts/repair_brand_images.py` fixes rows whose image URLs are DEAD. This
|
||||
one fixes rows whose image URLs are ALIVE AND WRONG, which the repair script
|
||||
refuses to touch by design ("never rewrites a row that already has one working
|
||||
URL").
|
||||
|
||||
The Dabur run of 2026-09-11 left exactly that: "Dabur Honey 1kg / 250g / 50g"
|
||||
with an empty `image_url` and four foreign honeys in `image_urls` - which the
|
||||
product card shows as soon as the primary is blank - and "Dabur Honey 500g"
|
||||
with a Dabur Odomos mosquito repellent as its primary. The ingestion path no
|
||||
longer stores rejected candidates (`store_catalog_pipeline.stage_6_images`),
|
||||
but rows written before that keep them, and a re-ingest does not revisit a
|
||||
row that already has images. Hence this script.
|
||||
|
||||
WHAT IT DOES, PER ROW
|
||||
---------------------
|
||||
1. Runs `image_corroboration.choose_primary` over every URL the row holds
|
||||
(primary first). `image_url` becomes the eligible primary, `image_urls`
|
||||
the eligible list; rejected URLs are dropped.
|
||||
2. If nothing is eligible, looks for another pack size of the same product
|
||||
(same `search_key`) whose primary passed, and adopts it - through the gate
|
||||
for THIS title, exactly as stage 6 does.
|
||||
3. If still nothing, `image_url` becomes NULL and `image_urls` empty. The card
|
||||
renders the brand monogram, which is honest; a foreign honey is not.
|
||||
|
||||
Own Products is skipped: its images are curated (`produce_reference`) and the
|
||||
gate was never meant for a bucket without a brand.
|
||||
|
||||
USAGE
|
||||
-----
|
||||
python -m scripts.regate_brand_images --brands Dabur # dry run
|
||||
python -m scripts.regate_brand_images --brands Dabur --apply
|
||||
python -m scripts.regate_brand_images --all # dry run, every brand
|
||||
|
||||
`--apply` writes a full backup of each targeted table first, in the same
|
||||
format `repair_brand_images.py --restore` reads, so the change is reversible:
|
||||
|
||||
python -m scripts.repair_brand_images --restore data/brand_image_repair_backup_X.json --apply
|
||||
|
||||
The Open*Facts barcode cross-check is a network call per distinct barcode
|
||||
(cached 30 days in data/cache/image_corroboration.db). Everything else is
|
||||
offline.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||
|
||||
from app.infrastructure.settings import DB_HOST, DB_NAME # noqa: E402
|
||||
from app.services import image_corroboration as ic # noqa: E402
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND # noqa: E402
|
||||
from app.services.vector_store import ( # noqa: E402
|
||||
_connect,
|
||||
_sanitize_name,
|
||||
invalidate_brand_overview_cache,
|
||||
)
|
||||
from scripts.repair_brand_images import ( # noqa: E402
|
||||
_backup,
|
||||
_brand_tables,
|
||||
_read_rows,
|
||||
)
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
logger = logging.getLogger("regate_brand_images")
|
||||
|
||||
|
||||
def _row_urls(row: Dict[str, Any]) -> List[str]:
|
||||
"""Primary first, then the rest, deduplicated, http only."""
|
||||
out: List[str] = []
|
||||
for url in [row.get("image_url")] + list(row.get("image_urls") or []):
|
||||
if url and str(url).startswith("http") and url not in out:
|
||||
out.append(str(url))
|
||||
return out
|
||||
|
||||
|
||||
def regate_rows(rows: List[Dict[str, Any]], brand: str) -> List[Dict[str, Any]]:
|
||||
"""The new (image_url, image_urls) for every row, plus why.
|
||||
|
||||
Pure apart from the Open*Facts barcode cross-check inside choose_primary,
|
||||
so it can be unit-tested with that patched.
|
||||
"""
|
||||
verdicts: List[Dict[str, Any]] = []
|
||||
for row in rows:
|
||||
urls = _row_urls(row)
|
||||
name = row.get("product_name") or ""
|
||||
if not urls:
|
||||
verdicts.append({**row, "new_url": None, "new_urls": [], "why": "no images stored"})
|
||||
continue
|
||||
choice = ic.choose_primary(urls, name, brand)
|
||||
verdicts.append({
|
||||
**row,
|
||||
"new_url": choice.primary,
|
||||
"new_urls": list(choice.eligible),
|
||||
"rejected": list(choice.rejected),
|
||||
"why": choice.reason,
|
||||
})
|
||||
|
||||
# Sibling adoption for rows left with nothing, from rows that kept a primary.
|
||||
eligible_by_key: Dict[tuple, str] = {}
|
||||
for v in verdicts:
|
||||
if v["new_url"]:
|
||||
eligible_by_key.setdefault(ic.search_key(v.get("product_name") or "", brand), v["new_url"])
|
||||
for v in verdicts:
|
||||
if v["new_url"]:
|
||||
continue
|
||||
candidate = eligible_by_key.get(ic.search_key(v.get("product_name") or "", brand))
|
||||
if not candidate:
|
||||
continue
|
||||
adopted = ic.choose_primary([candidate], v.get("product_name") or "", brand)
|
||||
if adopted.primary:
|
||||
v["new_url"] = adopted.primary
|
||||
v["new_urls"] = [adopted.primary]
|
||||
v["why"] = "adopted from another pack size of this product"
|
||||
return verdicts
|
||||
|
||||
|
||||
def _changed(v: Dict[str, Any]) -> bool:
|
||||
return (v.get("image_url") or None) != v["new_url"] or list(v.get("image_urls") or []) != v["new_urls"]
|
||||
|
||||
|
||||
def regate(cur, conn, brands: Optional[List[str]], apply: bool) -> Dict[str, int]:
|
||||
totals = {"rows": 0, "changed": 0, "blanked": 0, "adopted": 0}
|
||||
snapshot: Dict[str, List[Dict[str, Any]]] = {}
|
||||
plans: Dict[str, List[Dict[str, Any]]] = {}
|
||||
|
||||
for suffix, table in _brand_tables(cur, brands):
|
||||
if suffix == _sanitize_name(OWN_PRODUCTS_BRAND):
|
||||
logger.info("%s: skipped (curated images)", table)
|
||||
continue
|
||||
rows = _read_rows(cur, table)
|
||||
if not rows:
|
||||
continue
|
||||
snapshot[table] = rows
|
||||
verdicts = [v for v in regate_rows(rows, suffix) if _changed(v)]
|
||||
plans[table] = verdicts
|
||||
totals["rows"] += len(rows)
|
||||
logger.info("%s: %d row(s), %d to change", table, len(rows), len(verdicts))
|
||||
for v in verdicts:
|
||||
arrow = "->"
|
||||
logger.info(" %-42s %s %s", (v.get("product_name") or "")[:42], arrow,
|
||||
(v["new_url"] or "(none)")[:80])
|
||||
logger.info(" was %s (+%d more) %s", (v.get("image_url") or "(none)")[:70],
|
||||
max(0, len(v.get("image_urls") or []) - 1), v["why"])
|
||||
for dropped in v.get("rejected") or []:
|
||||
logger.info(" drop %s", dropped[:90])
|
||||
totals["changed"] += 1
|
||||
if v["new_url"] is None:
|
||||
totals["blanked"] += 1
|
||||
if v["why"].startswith("adopted"):
|
||||
totals["adopted"] += 1
|
||||
|
||||
if apply and any(plans.values()):
|
||||
logger.info("Backup written: %s", _backup(snapshot))
|
||||
for table, verdicts in plans.items():
|
||||
for v in verdicts:
|
||||
cur.execute(
|
||||
f"UPDATE {table} SET image_url = %s, image_urls = %s, "
|
||||
f"updated_at = CURRENT_TIMESTAMP WHERE image_id = %s",
|
||||
(v["new_url"], v["new_urls"], v["image_id"]),
|
||||
)
|
||||
conn.commit()
|
||||
try:
|
||||
invalidate_brand_overview_cache()
|
||||
except Exception: # noqa: BLE001 - the cache is a convenience
|
||||
pass
|
||||
return totals
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
mode = parser.add_mutually_exclusive_group(required=True)
|
||||
mode.add_argument("--brands", help="comma-separated brands to re-gate")
|
||||
mode.add_argument("--all", action="store_true", help="every brand table")
|
||||
parser.add_argument("--apply", action="store_true", help="write the changes (default: dry run)")
|
||||
args = parser.parse_args()
|
||||
|
||||
logger.info("Target database: %s / %s", DB_HOST, DB_NAME)
|
||||
logger.info("Mode: %s", "APPLY - this writes" if args.apply else "DRY RUN - nothing is written")
|
||||
|
||||
conn = _connect()
|
||||
if conn is None:
|
||||
logger.error("No database connection.")
|
||||
return 1
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
brands = None if args.all else [b for b in args.brands.split(",") if b.strip()]
|
||||
totals = regate(cur, conn, brands, args.apply)
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
logger.info("")
|
||||
logger.info("%s: %d row(s) read, %d %s, %d blanked, %d adopted from a sibling size",
|
||||
"Applied" if args.apply else "Dry run", totals["rows"], totals["changed"],
|
||||
"changed" if args.apply else "would change", totals["blanked"], totals["adopted"])
|
||||
if args.apply:
|
||||
logger.info("The API caches the brand overview; hit GET /api/brands/overview?refresh=true "
|
||||
"or wait out BRAND_OVERVIEW_TTL_SECONDS.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -55,6 +55,11 @@ os.environ.setdefault("USE_PGVECTOR", "true")
|
||||
os.environ.setdefault("DB_PASSWORD", "test-password-not-real")
|
||||
os.environ.setdefault("USE_S3", "false")
|
||||
os.environ.setdefault("USE_GOOGLE_CSE", "false")
|
||||
# Unconditional: the LLM description stage is on by default for every batch
|
||||
# and a developer machine often has `ollama serve` running, so without this a
|
||||
# pipeline test would make real, slow, non-deterministic model calls. Tests
|
||||
# that exercise the probe itself monkeypatch `ollama_service.USE_OLLAMA`.
|
||||
os.environ["USE_OLLAMA"] = "false"
|
||||
|
||||
# Unconditional, NOT setdefault. The suite pins brand-name behaviour all over
|
||||
# the place ("any Nestle chocolates?", the suggest ranking fixtures), and a
|
||||
|
||||
@@ -5,8 +5,8 @@ boundary *on the pipeline module object*, build real spreadsheets in memory,
|
||||
and point the batch directory at tmp_path so nothing is written into the repo.
|
||||
|
||||
Nothing here loads sentence-transformers or torch, and nothing reaches the
|
||||
network: every batch runs with use_llm=False and fetch_images=False, which are
|
||||
also the defaults the endpoint ships.
|
||||
network: every batch runs with fetch_images=False (the endpoint's default) and
|
||||
the LLM stage, on by default, finds Ollama disabled and falls back at once.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -486,7 +486,7 @@ def test_ingest_returns_202_and_a_batch_id(client, admin_headers, store, batch_r
|
||||
assert response.status_code == 202
|
||||
body = response.json()
|
||||
assert body["files_total"] == 2
|
||||
assert body["use_llm"] is False, "the LLM stage must be opt-in for a batch"
|
||||
assert body["use_llm"] is True, "LLM descriptions are on by default; a batch opts OUT"
|
||||
assert body["fetch_images"] is False, "image search must be opt-in for a batch"
|
||||
assert len(body["files"]) == 2
|
||||
assert body["files"][0]["total_stages"] == pipeline.TOTAL_STAGES
|
||||
|
||||
@@ -21,6 +21,9 @@ from app.api.routers.user_products import map_spreadsheet_columns, row_to_reques
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services import brand_discovery as bd
|
||||
|
||||
# Captured before the autouse `_no_network` fixture replaces it on the module.
|
||||
_REAL_FROM_OPEN_FACTS = bd._from_open_facts
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fixtures
|
||||
@@ -220,12 +223,11 @@ def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch):
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: one GTIN, one pack
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
|
||||
"""A GTIN identifies one pack, and stage 4 invents three when it has none.
|
||||
|
||||
Every column is copied into each exploded variant, so a surviving barcode
|
||||
would be stamped onto two packs that do not exist - wrong data that looks
|
||||
authoritative.
|
||||
def test_a_barcode_survives_when_no_pack_size_is_known(monkeypatch):
|
||||
"""A GTIN identifies one pack, and stage 4 now stores exactly one row for
|
||||
a product with no size (it used to invent three and stamp the same GTIN
|
||||
on all of them, which is why the barcode was dropped before). One GTIN on
|
||||
one unsized row is what a GTIN means; the row is only flagged for review.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", code="8901063017702"),
|
||||
@@ -234,8 +236,8 @@ def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.size_variants == []
|
||||
assert product.barcode is None
|
||||
assert any("barcode dropped" in note for note in product.notes)
|
||||
assert product.barcode == "8901063017702"
|
||||
assert any("pack size unknown" in note for note in product.notes)
|
||||
|
||||
|
||||
def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch):
|
||||
@@ -549,3 +551,113 @@ def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch):
|
||||
assert row.get(column), f"{column} was left empty"
|
||||
assert row["barcode"] == "8901063012516"
|
||||
assert row["fssai_license"] == "10012022000103"
|
||||
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# What the warnings claim - "unreachable" vs "empty" vs "no storefront"
|
||||
# ---------------------------------------------------------------------------
|
||||
# Naga (2026-09-11): Open Food Facts holds two real rows for the brand, yet the
|
||||
# tab said "Neither Open Food Facts nor a brand storefront has anything for
|
||||
# this brand". The lookup had failed / been mis-indexed, no storefront was
|
||||
# registered, and one warning text covered all of it. Each claim now has to
|
||||
# be true on its own.
|
||||
|
||||
def _store_rows(monkeypatch, rows):
|
||||
monkeypatch.setattr(bd, "_from_brand_store", lambda brand: list(rows))
|
||||
|
||||
|
||||
def test_an_unreachable_off_is_reported_as_unreachable_not_empty(monkeypatch):
|
||||
def down(brand, refresh=False):
|
||||
raise bd.OpenFactsUnavailable("HTTP 503 from Open Food Facts")
|
||||
|
||||
monkeypatch.setattr(bd, "_from_open_facts", down)
|
||||
_store_rows(monkeypatch, [])
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Naga Sooji", "category": None, "description": None,
|
||||
"sizes": ["500g"], "providers": [], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Naga", require_evidence=False)
|
||||
|
||||
reached = [w for w in result.warnings if "could not be reached" in w]
|
||||
assert len(reached) == 1 and "503" in reached[0]
|
||||
assert not any("has no products tagged" in w for w in result.warnings)
|
||||
assert [p.title for p in result.products] == ["Naga Sooji"], "the LLM rows must still come through when OFF is down"
|
||||
|
||||
|
||||
def test_a_genuinely_empty_off_says_so_without_blaming_the_network(monkeypatch):
|
||||
_store_rows(monkeypatch, [])
|
||||
result = bd.discover_brand_products("Udhaiyam")
|
||||
assert any("has no products tagged" in w for w in result.warnings)
|
||||
assert not any("could not be reached" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_unregistered_storefront_is_not_described_as_empty(monkeypatch):
|
||||
_store_rows(monkeypatch, [])
|
||||
monkeypatch.setattr(bd, "get_brand_store_domain", lambda brand: None)
|
||||
|
||||
result = bd.discover_brand_products("Naga")
|
||||
|
||||
assert any("No verified storefront is registered" in w for w in result.warnings)
|
||||
assert not any("Neither Open Food Facts" in w for w in result.warnings)
|
||||
assert not any("has not been fetched" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_registered_but_unfetched_storefront_names_the_backfill(monkeypatch):
|
||||
_store_rows(monkeypatch, [])
|
||||
monkeypatch.setattr(bd, "get_brand_store_domain", lambda brand: "gopuramproducts.com")
|
||||
|
||||
result = bd.discover_brand_products("Gopuram")
|
||||
|
||||
hit = [w for w in result.warnings if "has not been fetched" in w]
|
||||
assert len(hit) == 1
|
||||
assert "gopuramproducts.com" in hit[0] and "backfill_brand_stores.py" in hit[0]
|
||||
assert not any("No verified storefront" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_the_llm_only_caveat_appears_exactly_once(monkeypatch):
|
||||
_store_rows(monkeypatch, [])
|
||||
result = bd.discover_brand_products("Naga")
|
||||
assert sum("rests on the language model alone" in w for w in result.warnings) == 1
|
||||
|
||||
|
||||
def test_no_storefront_warning_when_the_shop_or_off_has_rows(monkeypatch):
|
||||
_store_rows(monkeypatch, [{"title": "Naga Maida 1kg", "source": "store", "size": "1kg"}])
|
||||
result = bd.discover_brand_products("Naga")
|
||||
assert not any("storefront" in w.lower() for w in result.warnings)
|
||||
assert not any("rests on the language model alone" in w for w in result.warnings)
|
||||
|
||||
_store_rows(monkeypatch, [])
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Sooji", code="8906011830068", quantity="500 g"),
|
||||
])
|
||||
result = bd.discover_brand_products("Naga")
|
||||
assert not any("storefront" in w.lower() for w in result.warnings)
|
||||
assert not any("rests on the language model alone" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_from_open_facts_raises_when_the_fetch_failed_with_nothing(monkeypatch):
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
|
||||
monkeypatch.setattr(off_bulk, "fetch_brand_corpus_result",
|
||||
lambda brand, refresh=False: off_bulk.CorpusFetch(hits=[], error="dns"))
|
||||
with pytest.raises(bd.OpenFactsUnavailable):
|
||||
_REAL_FROM_OPEN_FACTS("Naga")
|
||||
|
||||
|
||||
def test_from_open_facts_maps_v2_hits_to_candidates(monkeypatch):
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
|
||||
monkeypatch.setattr(off_bulk, "fetch_brand_corpus_result",
|
||||
lambda brand, refresh=False: off_bulk.CorpusFetch(hits=[
|
||||
{"code": "8906011830068", "product_name": "Sooji",
|
||||
"product_name_en": "Sooji", "quantity": "500 g"},
|
||||
{"code": "8906011831713", "product_name": "Maida", "quantity": "500 g"},
|
||||
]))
|
||||
|
||||
rows = _REAL_FROM_OPEN_FACTS("Naga")
|
||||
|
||||
assert [(r["title"], r["barcode"], r["source"]) for r in rows] == [
|
||||
("Sooji", "8906011830068", "off"), ("Maida", "8906011831713", "off"),
|
||||
]
|
||||
|
||||
@@ -174,6 +174,94 @@ def test_a_lookup_failure_never_rejects_an_image(monkeypatch):
|
||||
assert ic.openfacts_product_matches_brand(url, "Anil") is True
|
||||
|
||||
|
||||
class _Resp:
|
||||
def __init__(self, ok=True, payload=None):
|
||||
self.ok = ok
|
||||
self.status_code = 200 if ok else 503
|
||||
self._payload = payload or {}
|
||||
|
||||
def json(self):
|
||||
return self._payload
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def verdict_cache(tmp_path, monkeypatch):
|
||||
"""A throwaway sqlite cache, so these tests neither read nor poison the
|
||||
real one in data/cache."""
|
||||
monkeypatch.setattr(ic, "_DB_PATH", tmp_path / "verdicts.db")
|
||||
monkeypatch.setattr(ic, "_initialized", False) # the schema is per file
|
||||
monkeypatch.setattr(ic.time, "sleep", lambda s: None) # pacing/retry waits
|
||||
return tmp_path / "verdicts.db"
|
||||
|
||||
|
||||
KELLOGGS_HONEY = "https://images.openfoodfacts.org/images/products/505/931/902/3762/front_fr.32.400.jpg"
|
||||
|
||||
|
||||
def test_a_throttled_lookup_fails_open_but_is_not_remembered(monkeypatch, verdict_cache):
|
||||
"""The Dabur Honey defect. A hundred lookups in two minutes got the tail
|
||||
of them a 503; the fail-open True was cached for thirty days, and
|
||||
Kellogg's "Miel Pops" became a Dabur product for a month."""
|
||||
throttled = _Resp(ok=False)
|
||||
throttled.status_code = 503
|
||||
answers = iter([throttled, throttled, _Resp(payload={"status": 1, "product": {
|
||||
"brands": "KELLOG'S", "product_name": "Miel Pops"}})])
|
||||
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: next(answers))
|
||||
|
||||
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is True, "throttled on both attempts: fails open now"
|
||||
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False, "asks again next time"
|
||||
|
||||
|
||||
def test_one_throttled_attempt_is_retried_into_a_real_answer(monkeypatch, verdict_cache):
|
||||
throttled = _Resp(ok=False)
|
||||
throttled.status_code = 429
|
||||
answers = iter([throttled, _Resp(payload={"status": 1, "product": {
|
||||
"brands": "KELLOG'S", "product_name": "Miel Pops"}})])
|
||||
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: next(answers))
|
||||
|
||||
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False
|
||||
|
||||
|
||||
def test_a_real_verdict_is_cached(monkeypatch, verdict_cache):
|
||||
calls = []
|
||||
|
||||
def get(*a, **k):
|
||||
calls.append(1)
|
||||
return _Resp(payload={"status": 1, "product": {"brands": "Toblerone",
|
||||
"product_name": "Milk Chocolate"}})
|
||||
|
||||
monkeypatch.setattr(ic.requests, "get", get)
|
||||
url = "https://images.openfoodfacts.org/images/products/761/450/001/0013/front_en.362.400.jpg"
|
||||
|
||||
assert ic.openfacts_product_matches_brand(url, "Dabur") is False
|
||||
assert ic.openfacts_product_matches_brand(url, "Dabur") is False
|
||||
assert len(calls) == 1
|
||||
|
||||
|
||||
def test_an_unknown_barcode_is_cached_as_open(monkeypatch, verdict_cache):
|
||||
calls = []
|
||||
|
||||
def get(*a, **k):
|
||||
calls.append(1)
|
||||
return _Resp(payload={"status": 0, "status_verbose": "product not found"})
|
||||
|
||||
monkeypatch.setattr(ic.requests, "get", get)
|
||||
url = "https://images.openfoodfacts.org/images/products/000/000/000/0001/front.jpg"
|
||||
|
||||
assert ic.openfacts_product_matches_brand(url, "Dabur") is True
|
||||
assert ic.openfacts_product_matches_brand(url, "Dabur") is True
|
||||
assert len(calls) == 1
|
||||
|
||||
|
||||
def test_pre_fix_cache_entries_are_ignored(monkeypatch, verdict_cache):
|
||||
"""Entries written by the old code (no "v2:" prefix) may be fail-open
|
||||
Trues; they must not be trusted."""
|
||||
ic._cache_set("world.openfoodfacts.org:5059319023762:dabur", True)
|
||||
monkeypatch.setattr(ic.requests, "get", lambda *a, **k: _Resp(payload={
|
||||
"status": 1, "product": {"brands": "KELLOG'S", "product_name": "Miel Pops"}}))
|
||||
|
||||
assert ic.openfacts_product_matches_brand(KELLOGGS_HONEY, "Dabur") is False
|
||||
|
||||
|
||||
def test_a_non_openfacts_url_is_not_cross_checked():
|
||||
assert ic.openfacts_product_matches_brand(REAL_RAVA, "Anil") is True
|
||||
|
||||
@@ -193,3 +281,20 @@ def test_pack_sizes_of_one_product_share_a_search_key():
|
||||
def test_different_products_do_not_share_a_search_key():
|
||||
assert (ic.search_key("Anil Wheat Vermicelli 180g", "Anil")
|
||||
!= ic.search_key("Anil Samba Rava 500g", "Anil"))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Plural titles, singular filenames
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_plural_title_is_named_by_a_singular_filename():
|
||||
"""The all-brands re-gate would have dropped the correct
|
||||
`Aachi-Appalam-100-g-1.webp` from "Aachi Appalams 500g" and promoted an
|
||||
opaque Amazon URL over it."""
|
||||
url = "https://thedesifood.com/media/Aachi-Appalam-100-g-1.webp"
|
||||
assert ic.corroborate(url, "Aachi Appalams 500g", "Aachi").corroborated
|
||||
assert ic.corroborate("https://x.in/img/aachi-pickle-jar.jpg", "Aachi Pickles 200g", "Aachi").corroborated
|
||||
|
||||
|
||||
def test_the_singular_fallback_does_not_widen_short_tokens():
|
||||
# "gems" -> "gem" is too short a stem to trust; four letters is the floor.
|
||||
assert not ic.corroborate("https://x.in/gem-ring.jpg", "Cadbury Gems 20g", "Cadbury").corroborated
|
||||
|
||||
297
tests/test_image_precision.py
Normal file
297
tests/test_image_precision.py
Normal file
@@ -0,0 +1,297 @@
|
||||
"""The Dabur Honey regression: one product, several pack sizes, one photo.
|
||||
|
||||
Brand Discovery for Dabur (2026-09-11) stored, for the same product:
|
||||
|
||||
Dabur Honey 225g Amazon photo of Dabur Honey correct
|
||||
Dabur Honey 1kg image_url empty; image_urls = 4 honeys from the UK,
|
||||
Dabur Honey 250g France, Switzerland and Spain (Open Food Facts photos
|
||||
Dabur Honey 50g whose barcodes belong to other brands)
|
||||
Dabur Honey 500g a Dabur Odomos mosquito repellent
|
||||
|
||||
Three causes, each pinned below:
|
||||
|
||||
1. Stage 6 searched once PER SIZE. The query was identical - the size is not
|
||||
in it - but the providers are live and not deterministic, so four asks
|
||||
came back four ways. One search per product, shared by every size.
|
||||
2. `find_images_openfacts` threw away the `brands` field of the records it
|
||||
fetched, so a title-only fallback ("Honey 1kg") handed over every honey on
|
||||
the site. The brand is checked at the source now.
|
||||
3. `choose_primary` kept rejected candidates in `image_urls` "for review" -
|
||||
and the product card falls back to `image_urls[0]` when the primary is
|
||||
withheld, so the review pile was what the shopper saw. Only eligible
|
||||
candidates are stored.
|
||||
|
||||
Everything runs offline: the search function and the Open*Facts barcode
|
||||
cross-check are monkeypatched; storage and embeddings are stubbed.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services import image_corroboration as ic
|
||||
from app.services import image_search
|
||||
|
||||
openpyxl = pytest.importorskip("openpyxl")
|
||||
|
||||
AMAZON = "https://m.media-amazon.com/images/I/71O4OnjaHVL.jpg"
|
||||
ODOMOS = ("https://www.indianproductsstore.com/uploads/products/"
|
||||
"dabur-odomos-naturals-mosquito-repellent-gel.jpg")
|
||||
FOREIGN_HONEYS = [
|
||||
"https://images.openfoodfacts.org/images/products/505/931/902/3762/front_fr.32.400.jpg",
|
||||
"https://images.openfoodfacts.org/images/products/308/854/000/4440/front_fr.129.400.jpg",
|
||||
]
|
||||
DABUR_OFF = "https://images.openfoodfacts.org/images/products/890/120/702/6553/front_en.4.400.jpg"
|
||||
NAMED = "https://cdn.example.in/dabur-honey-squeezy-bottle.jpg"
|
||||
|
||||
|
||||
def _sheet(headers, rows) -> bytes:
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(headers)
|
||||
for row in rows:
|
||||
ws.append(row)
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _off_barcodes_offline(monkeypatch):
|
||||
"""Open*Facts says: the 890120... barcode is Dabur, the others are not."""
|
||||
monkeypatch.setattr(
|
||||
ic, "openfacts_product_matches_brand",
|
||||
lambda url, brand, timeout=10: "/890/120/" in url or "openfoodfacts" not in url,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
for row in rows:
|
||||
table[row["image_id"]] = dict(row)
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand",
|
||||
lambda b, **kw: [dict(r, brand=b) for r in table.values()])
|
||||
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def search(monkeypatch):
|
||||
"""A scripted `find_all_image_urls`; records every call it receives."""
|
||||
calls = []
|
||||
|
||||
def install(results):
|
||||
def fake(title, brand=None, country_hint=None, validate=True, max_results=24, produce=False):
|
||||
calls.append((title, brand))
|
||||
return list(results)
|
||||
monkeypatch.setattr(image_search, "find_all_image_urls", fake)
|
||||
return calls
|
||||
|
||||
return install
|
||||
|
||||
|
||||
def _run(headers, rows):
|
||||
return pipeline.run_pipeline("store.xlsx", _sheet(headers, rows),
|
||||
use_llm=False, fetch_images=True)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. One search per product
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_every_pack_size_of_a_product_gets_the_same_image_from_one_search(store, search):
|
||||
calls = search([AMAZON, DABUR_OFF])
|
||||
|
||||
_run(["Product Name", "Brand", "Pack Size"], [["Dabur Honey", "Dabur", "1kg, 250g, 50g, 225g"]])
|
||||
|
||||
assert len(store) == 4
|
||||
assert {r["image_url"] for r in store.values()} == {AMAZON}
|
||||
assert len(calls) == 1, "four pack sizes, one search"
|
||||
assert calls[0][0] == "Dabur Honey", "the size never enters the query"
|
||||
|
||||
|
||||
def test_sizes_written_into_the_name_still_share_the_search(store, search):
|
||||
calls = search([AMAZON])
|
||||
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"], ["Dabur Honey 250g", "Dabur"]])
|
||||
|
||||
assert len(calls) == 1
|
||||
assert calls[0][0] == "Dabur Honey"
|
||||
assert {r["image_url"] for r in store.values()} == {AMAZON}
|
||||
|
||||
|
||||
def test_different_products_of_one_brand_are_searched_separately(store, search):
|
||||
calls = search([NAMED])
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey", "Dabur"], ["Dabur Chyawanprash", "Dabur"]])
|
||||
assert sorted(c[0] for c in calls) == ["Dabur Chyawanprash", "Dabur Honey"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Open*Facts records are filtered by brand at the source
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_openfacts_photos_of_other_brands_honey_are_not_candidates(monkeypatch):
|
||||
records = [
|
||||
{"brands": "Dabur", "brands_tags": ["dabur"], "image_front_url": DABUR_OFF},
|
||||
{"brands": "Rowse", "brands_tags": ["rowse"], "image_front_url": FOREIGN_HONEYS[0]},
|
||||
{"brands": "Lune de Miel", "image_front_url": FOREIGN_HONEYS[1]},
|
||||
{"brands": "", "image_front_url": "https://images.openfoodfacts.org/x/unlabelled.jpg"},
|
||||
]
|
||||
monkeypatch.setattr(image_search, "_query_openfacts", lambda query, n: records)
|
||||
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
|
||||
|
||||
urls = image_search.find_images_openfacts("Honey", "Dabur")
|
||||
|
||||
assert DABUR_OFF in urls
|
||||
assert not any(u in urls for u in FOREIGN_HONEYS)
|
||||
assert "https://images.openfoodfacts.org/x/unlabelled.jpg" in urls, \
|
||||
"a record with no brand at all is not evidence of another brand"
|
||||
|
||||
|
||||
def test_the_title_only_fallback_is_also_brand_filtered(monkeypatch):
|
||||
seen = []
|
||||
|
||||
def query(q, n):
|
||||
seen.append(q)
|
||||
if q.startswith("dabur"):
|
||||
return []
|
||||
return [{"brands": "Rowse", "image_front_url": FOREIGN_HONEYS[0]}]
|
||||
|
||||
monkeypatch.setattr(image_search, "_query_openfacts", query)
|
||||
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
|
||||
|
||||
assert image_search.find_images_openfacts("Honey 1kg", "dabur") == []
|
||||
assert seen == ["dabur Honey 1kg", "Honey 1kg"]
|
||||
|
||||
|
||||
def test_without_a_brand_nothing_is_filtered(monkeypatch):
|
||||
monkeypatch.setattr(image_search, "_query_openfacts",
|
||||
lambda q, n: [{"brands": "Rowse", "image_front_url": FOREIGN_HONEYS[0]}])
|
||||
monkeypatch.setattr(image_search, "USE_OPEN_FACTS", True)
|
||||
assert image_search.find_images_openfacts("Honey", None) == FOREIGN_HONEYS[:1]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Rejected candidates are not stored where the UI will show them
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_choose_primary_separates_eligible_from_rejected():
|
||||
choice = ic.choose_primary([ODOMOS, FOREIGN_HONEYS[0], AMAZON, NAMED], "Dabur Honey 500g", "Dabur")
|
||||
|
||||
assert choice.primary == NAMED # tier 1: names "honey"
|
||||
assert choice.eligible == [NAMED, AMAZON] # tier 1 then tier 2
|
||||
assert choice.rejected == [ODOMOS, FOREIGN_HONEYS[0]]
|
||||
assert choice.ordered == choice.eligible + choice.rejected
|
||||
|
||||
|
||||
def test_a_same_brand_wrong_product_photo_is_never_stored(store, search):
|
||||
"""The 500g row: a Dabur Odomos repellent as the primary image of honey."""
|
||||
search([ODOMOS] + FOREIGN_HONEYS)
|
||||
|
||||
result = _run(["Product Name", "Brand"], [["Dabur Honey 500g", "Dabur"]])
|
||||
|
||||
row = next(iter(store.values()))
|
||||
assert row["image_url"] is None
|
||||
assert row["image_urls"] == [], "nothing the gate rejected may reach image_urls"
|
||||
assert any("rejected" in w for w in result.warnings)
|
||||
assert any("no primary image" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_foreign_honeys_ride_along_only_when_a_real_primary_exists(store, search):
|
||||
search([AMAZON, DABUR_OFF] + FOREIGN_HONEYS)
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
|
||||
|
||||
row = next(iter(store.values()))
|
||||
assert row["image_url"] == AMAZON
|
||||
assert row["image_urls"] == [AMAZON, DABUR_OFF]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. A sibling pack size already in the table lends its photo - through the gate
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_new_pack_size_adopts_the_siblings_image_without_searching(store, search):
|
||||
calls = search([AMAZON])
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 225g", "Dabur"]])
|
||||
assert len(calls) == 1 and store["dabur_dabur_honey_225g"]["image_url"] == AMAZON
|
||||
|
||||
result = _run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
|
||||
|
||||
assert len(calls) == 1, "the sibling's photo made the search unnecessary"
|
||||
assert store["dabur_dabur_honey_1kg"]["image_url"] == AMAZON
|
||||
assert store["dabur_dabur_honey_1kg"]["image_urls"] == [AMAZON]
|
||||
assert any("adopted from another pack size" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_a_siblings_wrong_image_is_not_adopted(store, search):
|
||||
"""A stored primary that would not pass the gate for this title stays put."""
|
||||
store["dabur_dabur_honey_500g"] = {
|
||||
"image_id": "dabur_dabur_honey_500g", "product_name": "Dabur Honey 500g",
|
||||
"image_url": ODOMOS, "image_urls": [ODOMOS],
|
||||
}
|
||||
calls = search([NAMED])
|
||||
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
|
||||
|
||||
assert len(calls) == 1, "an ineligible sibling image forces a fresh search"
|
||||
assert store["dabur_dabur_honey_1kg"]["image_url"] == NAMED
|
||||
|
||||
|
||||
def test_a_different_product_of_the_brand_is_not_a_sibling(store, search):
|
||||
store["dabur_dabur_chyawanprash_500g"] = {
|
||||
"image_id": "dabur_dabur_chyawanprash_500g", "product_name": "Dabur Chyawanprash 500g",
|
||||
"image_url": NAMED, "image_urls": [NAMED],
|
||||
}
|
||||
calls = search([AMAZON])
|
||||
_run(["Product Name", "Brand"], [["Dabur Honey 1kg", "Dabur"]])
|
||||
assert len(calls) == 1
|
||||
assert store["dabur_dabur_honey_1kg"]["image_url"] == AMAZON
|
||||
|
||||
|
||||
def test_stage_six_without_a_cache_still_works():
|
||||
"""Direct callers (and the produce tests) pass no cache."""
|
||||
row = {"brand": "Dabur", "product_name": "Dabur Honey 1kg", "image_urls": [AMAZON]}
|
||||
assert pipeline.stage_6_images(row) is row
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. The one-off repair for rows written before the fix
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_regate_rows_reproduces_the_dabur_repair():
|
||||
from scripts.regate_brand_images import regate_rows
|
||||
|
||||
rows = [
|
||||
{"image_id": "dabur_dabur_honey_225g", "product_name": "Dabur Honey 225g",
|
||||
"image_url": AMAZON, "image_urls": [AMAZON, DABUR_OFF] + FOREIGN_HONEYS},
|
||||
{"image_id": "dabur_dabur_honey_1kg", "product_name": "Dabur Honey 1kg",
|
||||
"image_url": None, "image_urls": list(FOREIGN_HONEYS)},
|
||||
{"image_id": "dabur_dabur_honey_500g", "product_name": "Dabur Honey 500g",
|
||||
"image_url": ODOMOS, "image_urls": [ODOMOS]},
|
||||
{"image_id": "dabur_dabur_chyawanprash_500g", "product_name": "Dabur Chyawanprash 500g",
|
||||
"image_url": None, "image_urls": [ODOMOS]},
|
||||
]
|
||||
|
||||
by_id = {v["image_id"]: v for v in regate_rows(rows, "dabur")}
|
||||
|
||||
# 225g keeps its photo and loses the foreign honeys from its gallery.
|
||||
assert by_id["dabur_dabur_honey_225g"]["new_url"] == AMAZON
|
||||
assert by_id["dabur_dabur_honey_225g"]["new_urls"] == [AMAZON, DABUR_OFF]
|
||||
# 1kg had nothing eligible and adopts the 225g photo.
|
||||
assert by_id["dabur_dabur_honey_1kg"]["new_url"] == AMAZON
|
||||
assert by_id["dabur_dabur_honey_1kg"]["why"].startswith("adopted")
|
||||
# 500g drops the mosquito repellent and adopts the sibling photo too.
|
||||
assert by_id["dabur_dabur_honey_500g"]["new_url"] == AMAZON
|
||||
assert ODOMOS in by_id["dabur_dabur_honey_500g"]["rejected"]
|
||||
# A different product with only a wrong image is blanked, not lent a honey.
|
||||
assert by_id["dabur_dabur_chyawanprash_500g"]["new_url"] is None
|
||||
assert by_id["dabur_dabur_chyawanprash_500g"]["new_urls"] == []
|
||||
308
tests/test_off_bulk_fetch.py
Normal file
308
tests/test_off_bulk_fetch.py
Normal file
@@ -0,0 +1,308 @@
|
||||
"""The network half of `off_bulk`: what is asked of Open Food Facts, how the
|
||||
answer is paged, and - the part that produced a user-visible lie - what gets
|
||||
written to the corpus cache when the answer never arrives.
|
||||
|
||||
Background, measured 2026-09-11. Brand Discovery told the user that Open Food
|
||||
Facts had nothing for "Naga". The v2 API returns two real Naga Limited rows
|
||||
(Sooji 500 g / 8906011830068, Maida 500 g / 8906011831713). The old query hit
|
||||
the Search-a-licious endpoint with a free-text `brands:Naga`, whose index
|
||||
still filed 8906011830068 brandless under Kuwait, and any 503 it met on the
|
||||
way was cached to disk as `hits: []`. Both failure modes are pinned here, with
|
||||
no network: `requests.get` is replaced for every test.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
|
||||
from app.services.enrichment.barcode.sources import off_bulk
|
||||
from app.services.enrichment.barcode.sources.off_bulk import (
|
||||
CACHE_SCHEMA,
|
||||
CorpusFetch,
|
||||
OffUnavailable,
|
||||
brand_tag_slug,
|
||||
fetch_brand_corpus,
|
||||
fetch_brand_corpus_result,
|
||||
)
|
||||
|
||||
NAGA_PRODUCTS = [
|
||||
{"code": "8906011830068", "product_name": "Sooji", "product_name_en": "Sooji",
|
||||
"brands": "Naga", "quantity": "500 g", "countries_tags": ["en:india"]},
|
||||
{"code": "8906011831713", "product_name": "Maida", "product_name_en": "Maida",
|
||||
"brands": "Naga", "quantity": "500 g", "countries_tags": ["en:india"]},
|
||||
]
|
||||
|
||||
|
||||
class _Resp:
|
||||
def __init__(self, status=200, payload=None, text="", content_type="application/json"):
|
||||
self.status_code = status
|
||||
self._payload = payload
|
||||
self.text = text
|
||||
self.headers = {"content-type": content_type}
|
||||
|
||||
def json(self):
|
||||
if self._payload is None:
|
||||
raise ValueError("not json")
|
||||
return self._payload
|
||||
|
||||
|
||||
def _v2(products, count=None, page=1, page_size=None):
|
||||
"""A v2 search body. `page_count` is deliberately the size of THIS page,
|
||||
which is what the real API sends and what a naive paginator misreads;
|
||||
`page_size` is the size the server APPLIED, not the one requested."""
|
||||
return {"count": len(products) if count is None else count, "page": page,
|
||||
"page_count": len(products),
|
||||
"page_size": off_bulk.PAGE_SIZE if page_size is None else page_size,
|
||||
"products": products, "skip": 0}
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_sleep(monkeypatch):
|
||||
monkeypatch.setattr(off_bulk.time, "sleep", lambda s: None)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def calls(monkeypatch):
|
||||
"""Replace requests.get with a scripted responder. Returns the list of
|
||||
(url, params) actually sent, so tests can assert on the query."""
|
||||
sent = []
|
||||
|
||||
def install(responder):
|
||||
def fake_get(url, params=None, headers=None, timeout=None):
|
||||
sent.append((url, dict(params or {})))
|
||||
r = responder(len(sent), dict(params or {}))
|
||||
if isinstance(r, Exception):
|
||||
raise r
|
||||
return r
|
||||
monkeypatch.setattr(off_bulk.requests, "get", fake_get)
|
||||
return sent
|
||||
|
||||
return install
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The query
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@pytest.mark.parametrize("brand, slug", [
|
||||
("Naga", "naga"),
|
||||
("Hindustan Unilever", "hindustan-unilever"),
|
||||
("P&G", "p-g"),
|
||||
(" Double Horse ", "double-horse"),
|
||||
("Coca-Cola", "coca-cola"),
|
||||
])
|
||||
def test_brand_tag_slug_matches_off_tag_form(brand, slug):
|
||||
assert brand_tag_slug(brand) == slug
|
||||
|
||||
|
||||
def test_query_is_an_exact_brand_tag_filter_on_the_v2_api(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
|
||||
result = fetch_brand_corpus_result("Naga", country="india", cache_dir=tmp_path)
|
||||
|
||||
assert result.error is None and not result.from_cache
|
||||
assert [h["code"] for h in result.hits] == ["8906011830068", "8906011831713"]
|
||||
url, params = sent[0]
|
||||
assert url == "https://world.openfoodfacts.org/api/v2/search"
|
||||
assert params["brands_tags"] == "naga"
|
||||
assert params["countries_tags"] == "en:india"
|
||||
assert "q" not in params, "free-text brands: matching is what returned Mr Naga"
|
||||
|
||||
|
||||
def test_no_country_filter_when_country_is_blank(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2([])))
|
||||
fetch_brand_corpus_result("Naga", country="", cache_dir=tmp_path)
|
||||
assert "countries_tags" not in sent[0][1]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Pagination
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_pagination_uses_count_not_page_count(calls, tmp_path):
|
||||
"""150 products at PAGE_SIZE 100 is two pages. The first page's
|
||||
`page_count` is 100 - reading that as "pages" would fetch 100 pages."""
|
||||
page1 = [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)]
|
||||
page2 = [{"code": f"891{i:010d}", "product_name": f"Q{i}"} for i in range(50)]
|
||||
|
||||
def responder(n, params):
|
||||
return _Resp(payload=_v2(page1 if params["page"] == 1 else page2,
|
||||
count=150, page=params["page"]))
|
||||
|
||||
sent = calls(responder)
|
||||
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
|
||||
|
||||
assert [p["page"] for _, p in sent] == [1, 2]
|
||||
assert len(result.hits) == 150
|
||||
|
||||
|
||||
def test_pagination_follows_the_page_size_the_server_applied(calls, tmp_path, monkeypatch):
|
||||
"""The real Amul case: 216 products, 250 requested, 100 served per page.
|
||||
Dividing by what we ASKED for stops after one page and loses 116 rows."""
|
||||
monkeypatch.setattr(off_bulk, "PAGE_SIZE", 250)
|
||||
pages = {
|
||||
1: [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)],
|
||||
2: [{"code": f"891{i:010d}", "product_name": f"Q{i}"} for i in range(100)],
|
||||
3: [{"code": f"892{i:010d}", "product_name": f"R{i}"} for i in range(16)],
|
||||
}
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(pages[p["page"]], count=216,
|
||||
page=p["page"], page_size=100)))
|
||||
|
||||
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
|
||||
|
||||
assert [p["page"] for _, p in sent] == [1, 2, 3]
|
||||
assert len(result.hits) == 216
|
||||
|
||||
|
||||
def test_pagination_is_capped_at_max_pages(calls, tmp_path):
|
||||
body = [{"code": "8900000000000", "product_name": "X"}] * off_bulk.PAGE_SIZE
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(body, count=10 ** 6, page=p["page"])))
|
||||
fetch_brand_corpus_result("Nestle", cache_dir=tmp_path)
|
||||
assert len(sent) == off_bulk.MAX_PAGES
|
||||
|
||||
|
||||
def test_nameless_products_are_dropped_from_the_corpus(calls, tmp_path):
|
||||
calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS + [
|
||||
{"code": "8906011839999", "product_name": ""},
|
||||
{"code": "8906011839998"},
|
||||
])))
|
||||
assert len(fetch_brand_corpus_result("Naga", cache_dir=tmp_path).hits) == 2
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Failure is not emptiness, and is never cached
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_503_is_retried_then_reported_and_not_cached(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(status=503, text="unavailable", content_type="text/html"))
|
||||
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
|
||||
assert len(sent) == 3, "with_retry gives a 5xx three attempts"
|
||||
assert result.hits == []
|
||||
assert result.error and "503" in result.error
|
||||
assert list(tmp_path.glob("*.json")) == [], "a failed fetch must not become hits: []"
|
||||
|
||||
|
||||
def test_503_then_success_recovers(calls, tmp_path):
|
||||
def responder(n, params):
|
||||
return _Resp(status=503) if n == 1 else _Resp(payload=_v2(NAGA_PRODUCTS))
|
||||
|
||||
calls(responder)
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
assert result.error is None and len(result.hits) == 2
|
||||
|
||||
|
||||
def test_html_200_error_page_is_a_failure_not_an_empty_brand(calls, tmp_path):
|
||||
calls(lambda n, p: _Resp(status=200, text="<html>temporarily unavailable</html>",
|
||||
content_type="text/html"))
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
assert result.error and result.hits == []
|
||||
assert list(tmp_path.glob("*.json")) == []
|
||||
|
||||
|
||||
def test_transport_error_after_retries_is_reported(calls, tmp_path):
|
||||
calls(lambda n, p: requests.exceptions.ConnectionError("dns"))
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
assert result.error and result.hits == []
|
||||
assert list(tmp_path.glob("*.json")) == []
|
||||
|
||||
|
||||
def test_partial_fetch_keeps_what_arrived_but_flags_it(calls, tmp_path):
|
||||
page1 = [{"code": f"890{i:010d}", "product_name": f"P{i}"} for i in range(100)]
|
||||
|
||||
def responder(n, params):
|
||||
if params["page"] == 1:
|
||||
return _Resp(payload=_v2(page1, count=150))
|
||||
return _Resp(status=502)
|
||||
|
||||
calls(responder)
|
||||
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
|
||||
assert len(result.hits) == 100 and result.error
|
||||
assert list(tmp_path.glob("*.json")) == []
|
||||
|
||||
|
||||
def test_list_form_still_returns_empty_on_failure(calls, tmp_path):
|
||||
"""Backfill scripts and ingestion call `fetch_brand_corpus` and rely on
|
||||
[] rather than an exception so one dead brand does not abort a run."""
|
||||
calls(lambda n, p: _Resp(status=503))
|
||||
assert fetch_brand_corpus("Naga", cache_dir=tmp_path) == []
|
||||
|
||||
|
||||
def test_off_unavailable_is_a_connection_error_for_the_retry_decorator():
|
||||
assert issubclass(OffUnavailable, requests.exceptions.ConnectionError)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The cache
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def test_successful_fetch_is_cached_with_schema_and_served_from_cache(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
|
||||
first = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
second = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
|
||||
assert len(sent) == 1
|
||||
assert not first.from_cache and second.from_cache
|
||||
assert second.hits == first.hits
|
||||
on_disk = json.loads((tmp_path / "naga.json").read_text(encoding="utf-8"))
|
||||
assert on_disk["schema"] == CACHE_SCHEMA
|
||||
assert on_disk["brand_tag"] == "naga"
|
||||
assert on_disk["endpoint"] == off_bulk.SEARCH_URL
|
||||
|
||||
|
||||
def test_genuinely_empty_brand_is_cached_and_is_not_an_error(calls, tmp_path):
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2([])))
|
||||
|
||||
result = fetch_brand_corpus_result("Udhaiyam", cache_dir=tmp_path)
|
||||
again = fetch_brand_corpus_result("Udhaiyam", cache_dir=tmp_path)
|
||||
|
||||
assert result == CorpusFetch(hits=[], error=None, from_cache=False)
|
||||
assert again.from_cache and again.hits == []
|
||||
assert len(sent) == 1
|
||||
|
||||
|
||||
def test_empty_pre_schema_cache_is_refetched(calls, tmp_path):
|
||||
"""The exact prod artefact: an old-endpoint file saying Naga has nothing."""
|
||||
(tmp_path / "naga.json").write_text(json.dumps({
|
||||
"brand": "Naga", "country": "india", "fetched_at": 0,
|
||||
"fetched_at_human": "2026-09-10 12:00:00", "hits": [],
|
||||
}), encoding="utf-8")
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
|
||||
result = fetch_brand_corpus_result("Naga", cache_dir=tmp_path)
|
||||
|
||||
assert len(sent) == 1 and not result.from_cache
|
||||
assert len(result.hits) == 2
|
||||
assert json.loads((tmp_path / "naga.json").read_text(encoding="utf-8"))["schema"] == CACHE_SCHEMA
|
||||
|
||||
|
||||
def test_non_empty_pre_schema_cache_is_still_trusted(calls, tmp_path):
|
||||
(tmp_path / "amul.json").write_text(json.dumps({
|
||||
"brand": "Amul", "country": "india", "fetched_at": 0,
|
||||
"hits": [{"code": "8901262010115", "product_name": "Amul Butter"}],
|
||||
}), encoding="utf-8")
|
||||
sent = calls(lambda n, p: _Resp(status=503))
|
||||
|
||||
result = fetch_brand_corpus_result("Amul", cache_dir=tmp_path)
|
||||
|
||||
assert sent == [] and result.from_cache
|
||||
assert result.hits[0]["product_name"] == "Amul Butter"
|
||||
|
||||
|
||||
def test_refresh_bypasses_the_cache(calls, tmp_path):
|
||||
(tmp_path / "naga.json").write_text(json.dumps({
|
||||
"schema": CACHE_SCHEMA, "brand": "Naga", "hits": [],
|
||||
}), encoding="utf-8")
|
||||
sent = calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
assert len(fetch_brand_corpus_result("Naga", refresh=True, cache_dir=tmp_path).hits) == 2
|
||||
assert len(sent) == 1
|
||||
|
||||
|
||||
def test_unreadable_cache_is_refetched(calls, tmp_path):
|
||||
(tmp_path / "naga.json").write_text("{not json", encoding="utf-8")
|
||||
calls(lambda n, p: _Resp(payload=_v2(NAGA_PRODUCTS)))
|
||||
assert len(fetch_brand_corpus_result("Naga", cache_dir=tmp_path).hits) == 2
|
||||
@@ -159,7 +159,11 @@ def test_the_sheets_values_are_kept_and_nothing_else_is_invented(store):
|
||||
# What the sheet said.
|
||||
assert row["product_name"] == "Apple 500g"
|
||||
assert row["size_variants"] == ["500g"]
|
||||
assert row["final_selling_price"] == 155
|
||||
# "Selling Price" is the shop's price and lands in selling_price. The
|
||||
# final (tax-inclusive) column is filled from it by upsert_brand_products,
|
||||
# which this fixture stubs, so here it stays what the sheet said: nothing.
|
||||
assert row["selling_price"] == 155
|
||||
assert row["final_selling_price"] is None
|
||||
assert row["price_range"] == "₹143-167" # +/-8% of the sheet's own price
|
||||
|
||||
# What it did not say, and what we therefore do not claim.
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
Open Food Facts answers "does this product exist" for food and nothing else.
|
||||
`off_bulk` queries only `search.openfoodfacts.org`, so a toothpaste or a
|
||||
`off_bulk` queries only Open Food Facts' food database, so a toothpaste or a
|
||||
detergent has no product-discovery source at all and its whole catalogue is
|
||||
language-model output. A live retail lookup is the only signal in the codebase
|
||||
that means "available right now" rather than "was in a database dump", and it
|
||||
|
||||
312
tests/test_upload_sheet_fields.py
Normal file
312
tests/test_upload_sheet_fields.py
Normal file
@@ -0,0 +1,312 @@
|
||||
"""What an uploaded sheet's own facts become in the brand table.
|
||||
|
||||
Three rules, each of which was violated before this file existed:
|
||||
|
||||
1. THE RETAIL PRICE IS THE SELLING PRICE. Every price header used to land in
|
||||
`final_selling_price` and `selling_price` was NULL for every uploaded row;
|
||||
worse, the HSN/GST stage then overwrote whatever price the sheet gave with
|
||||
the ceiling of the derived band (155 became 167).
|
||||
2. A PACK SIZE IS RECORDED ONLY WHEN THE SHEET OR THE NAME STATES IT.
|
||||
"India Gate Basmati Rice 1kg" is one product in one size. "India Gate
|
||||
Basmati Rice" used to become three products - 1kg, 5kg, 25kg - none of
|
||||
which the sheet listed.
|
||||
3. A BLANK DESCRIPTION IS WRITTEN, NOT PADDED. The LLM writes it when Ollama
|
||||
is up; when it is not, the row gets a short factual line built only from
|
||||
facts the row holds - never "<name> from <brand>." and never the marketing
|
||||
essay in catalog_engine.generate_detailed_description.
|
||||
|
||||
Everything runs offline: storage and embeddings are monkeypatched on the
|
||||
pipeline module, and USE_OLLAMA is false for the whole suite (conftest.py).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services.enrichment.hsn_gst.models import enrich_pricing_fields
|
||||
from app.services.generic_products import OWN_PRODUCTS_BRAND
|
||||
|
||||
openpyxl = pytest.importorskip("openpyxl")
|
||||
|
||||
|
||||
def _sheet(headers, rows) -> bytes:
|
||||
wb = openpyxl.Workbook()
|
||||
ws = wb.active
|
||||
ws.append(headers)
|
||||
for row in rows:
|
||||
ws.append(row)
|
||||
buf = io.BytesIO()
|
||||
wb.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""A fake brand table: image_id -> stored row, across every brand."""
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
assert cleanup is False
|
||||
for row in rows:
|
||||
table[row["image_id"]] = dict(row)
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
|
||||
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
|
||||
|
||||
def _run(headers, rows, **kw):
|
||||
kw.setdefault("use_llm", False)
|
||||
kw.setdefault("fetch_images", False)
|
||||
return pipeline.run_pipeline("store.xlsx", _sheet(headers, rows), **kw)
|
||||
|
||||
|
||||
# Real-looking names: the validation gate rejects placeholder titles such as
|
||||
# "Product 3", which would empty the store and hide what a test is about.
|
||||
_AMUL_NAMES = ["Amul Butter 100g", "Amul Cheese 200g", "Amul Ghee 500ml", "Amul Taaza 1L",
|
||||
"Amul Kool 200ml", "Amul Lassi 200ml", "Amul Masti Dahi 400g"]
|
||||
|
||||
|
||||
def _only(store):
|
||||
assert len(store) == 1, list(store)
|
||||
return next(iter(store.values()))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Price
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_retail_price_lands_in_selling_price_and_drives_the_band(store):
|
||||
_run(["ProductName", "ProductSKU", "ProductBrand", "RetailPrice"],
|
||||
[["India Gate Basmati Rice 1kg", "IG-BR-1", "India Gate", 120]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["selling_price"] == 120
|
||||
assert row["price_range"] == "₹110-130" # +/-8% of the sheet's price
|
||||
assert row["product_sku"] == "IG-BR-1" and row["sku_source"] == "sheet"
|
||||
|
||||
|
||||
def test_a_final_price_column_is_kept_separate_and_wins_the_band(store):
|
||||
_run(["Item Name", "Brand", "Selling Price", "Final Price"],
|
||||
[["Amul Butter 500g", "Amul", 250, 262]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["selling_price"] == 250
|
||||
assert row["final_selling_price"] == 262
|
||||
assert row["price_range"] == "₹241-283" # band from the final price
|
||||
|
||||
|
||||
def test_reingesting_a_sheet_with_a_selling_price_is_a_no_op(store):
|
||||
headers, rows = ["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 250]]
|
||||
first = _run(headers, rows)
|
||||
assert first.inserted == 1
|
||||
|
||||
second = _run(headers, rows)
|
||||
assert (second.inserted, second.backfilled, second.skipped_existing) == (0, 0, 1)
|
||||
|
||||
|
||||
def test_a_held_final_price_is_not_clobbered_by_a_different_retail_price(store):
|
||||
_run(["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 250]])
|
||||
key = next(iter(store))
|
||||
store[key]["final_selling_price"] = 262 # what the table already holds
|
||||
|
||||
_run(["Item Name", "Brand", "Retail Price"], [["Amul Butter 500g", "Amul", 199]])
|
||||
|
||||
assert store[key]["final_selling_price"] == 262
|
||||
assert store[key]["selling_price"] == 250, "fill-only-blanks: a held price is kept"
|
||||
|
||||
|
||||
class TestHsnStageKeepsHeldPrices:
|
||||
"""The overwrite that turned a sheet's 155 into 167."""
|
||||
|
||||
def test_a_held_selling_price_is_returned_as_none_so_apply_keeps_it(self):
|
||||
fields = enrich_pricing_fields({
|
||||
"selling_price": 155, "final_selling_price": None,
|
||||
"price_range": "₹143-167", "gst_percent": 18,
|
||||
})
|
||||
assert fields["selling_price"] is None
|
||||
assert fields["tax_amount"] == 27.9 # 18% of the HELD 155, not of 167
|
||||
assert fields["final_selling_price"] == 182.9
|
||||
|
||||
def test_a_held_final_price_is_also_kept(self):
|
||||
fields = enrich_pricing_fields({
|
||||
"selling_price": 155, "final_selling_price": 160,
|
||||
"price_range": "₹143-167", "gst_percent": 18,
|
||||
})
|
||||
assert fields["selling_price"] is None
|
||||
assert fields["final_selling_price"] is None
|
||||
|
||||
def test_without_a_sheet_price_the_band_ceiling_is_still_the_base(self):
|
||||
fields = enrich_pricing_fields({"price_range": "₹143-167", "gst_percent": 5})
|
||||
assert fields["selling_price"] == 167
|
||||
assert fields["final_selling_price"] == 175.35
|
||||
|
||||
def test_a_held_final_price_alone_seeds_the_selling_price(self):
|
||||
fields = enrich_pricing_fields({"final_selling_price": 200, "gst_percent": 5,
|
||||
"price_range": "₹300-400"})
|
||||
assert fields["selling_price"] == 200
|
||||
assert fields["final_selling_price"] is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Size
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("name, size", [
|
||||
("India Gate Basmati Rice 1kg", "1kg"),
|
||||
("Fortune Sunflower Oil 1L", "1L"),
|
||||
("Amul Taaza Milk 1 Litre", "1 Litre"),
|
||||
("Aashirvaad Atta 5 Kg", "5 Kg"),
|
||||
("Tata Salt 1.5kg", "1.5kg"),
|
||||
("Parle-G Biscuits 800 gm", "800 gm"),
|
||||
("Saffola Gold 500ml", "500ml"),
|
||||
])
|
||||
def test_the_size_in_the_name_is_the_size_and_nothing_is_appended(store, name, size):
|
||||
_run(["Product Name", "Brand"], [[name, name.split()[0]]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["size_variants"] == [size]
|
||||
assert row["product_name"] == name, "the name already carries the size"
|
||||
|
||||
|
||||
def test_a_branded_product_with_no_size_is_one_clean_row(store):
|
||||
result = _run(["Product Name", "Brand"], [["India Gate Basmati Rice", "India Gate"]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["size_variants"] == []
|
||||
assert row["product_name"] == "India Gate Basmati Rice"
|
||||
assert "Standard" not in row["product_name"]
|
||||
assert next(iter(store)) == "india_gate_india_gate_basmati_rice"
|
||||
assert any("no pack size" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_a_commodity_with_no_size_still_uses_the_standard_sentinel(store):
|
||||
"""Own Products dedupes against a seeded base list built on "Standard"."""
|
||||
_run(["Product Name"], [["Apple"]])
|
||||
|
||||
row = _only(store)
|
||||
assert row["size_variants"] == ["Standard"]
|
||||
assert next(iter(store)) == pipeline.build_image_id(OWN_PRODUCTS_BRAND, "Apple", "Standard")
|
||||
|
||||
|
||||
def test_the_llm_may_not_supply_a_pack_size(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details",
|
||||
lambda brand, title, category=None, size=None: {
|
||||
"description": "Refined wheat flour milled for baking.",
|
||||
"size_variants": ["500g", "1kg", "5kg"],
|
||||
})
|
||||
|
||||
_run(["Product Name", "Brand"], [["Naga Maida", "Naga"]], use_llm=True)
|
||||
|
||||
row = _only(store)
|
||||
assert row["size_variants"] == []
|
||||
assert row["description"] == "Refined wheat flour milled for baking."
|
||||
|
||||
|
||||
def test_a_sheet_size_column_still_wins_over_the_name(store):
|
||||
_run(["Product Name", "Brand", "Pack Size"], [["Amul Butter 100g", "Amul", "500g"]])
|
||||
assert _only(store)["size_variants"] == ["500g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Description
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_llm_description_is_used_when_ollama_answers(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
seen = {}
|
||||
|
||||
def fake(brand, title, category=None, size=None):
|
||||
seen.update(brand=brand, title=title, category=category, size=size)
|
||||
return {"description": "Long-grain aged basmati rice for biryani and pulao."}
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details", fake)
|
||||
|
||||
_run(["Product Name", "Brand"], [["India Gate Basmati Rice 1kg", "India Gate"]], use_llm=True)
|
||||
|
||||
assert _only(store)["description"] == "Long-grain aged basmati rice for biryani and pulao."
|
||||
assert seen["size"] == "1kg", "the pack size in the name is handed to the prompt"
|
||||
|
||||
|
||||
def test_an_overlong_llm_description_is_clamped(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details",
|
||||
lambda *a, **k: {"description": "x" * 2000})
|
||||
_run(["Product Name", "Brand"], [["Amul Butter 500g", "Amul"]], use_llm=True)
|
||||
assert len(_only(store)["description"]) == pipeline._MAX_LLM_DESCRIPTION_CHARS
|
||||
|
||||
|
||||
def test_the_fallback_description_is_factual_and_short(store):
|
||||
_run(["Product Name", "Brand", "Category"],
|
||||
[["Fortune Sunflower Oil 1L", "Fortune", "Edible Oils"]])
|
||||
|
||||
desc = _only(store)["description"]
|
||||
assert desc == "Fortune Sunflower Oil 1L, Edible Oils, by Fortune."
|
||||
assert "from Fortune." not in desc and "Introducing" not in desc
|
||||
|
||||
|
||||
def test_the_fallback_description_names_a_sheet_size_the_name_lacks(store):
|
||||
_run(["Product Name", "Brand", "Pack Size"], [["Amul Butter", "Amul", "500g"]])
|
||||
assert _only(store)["description"] == "Amul Butter 500g, Dairy, by Amul."
|
||||
|
||||
|
||||
def test_a_sheet_description_is_never_overwritten(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details",
|
||||
lambda *a, **k: pytest.fail("the LLM must not be asked"))
|
||||
_run(["Product Name", "Brand", "Description"],
|
||||
[["Amul Butter 500g", "Amul", "Salted table butter."]], use_llm=True)
|
||||
assert _only(store)["description"] == "Salted table butter."
|
||||
|
||||
|
||||
def test_the_llm_is_switched_off_after_three_consecutive_misses(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
calls = []
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details",
|
||||
lambda brand, title, **k: calls.append(title) or None)
|
||||
|
||||
rows = [[name, "Amul"] for name in _AMUL_NAMES[:6]]
|
||||
result = _run(["Product Name", "Brand"], rows, use_llm=True)
|
||||
|
||||
assert len(calls) == 3
|
||||
assert any("consecutive" in w for w in result.warnings)
|
||||
assert len(store) == 6, "the rows themselves are unaffected"
|
||||
|
||||
|
||||
def test_one_answer_resets_the_breaker(store, monkeypatch):
|
||||
from app.services import ollama_service
|
||||
|
||||
answers = iter([None, None, {"description": "ok"}, None, None, None, None])
|
||||
calls = []
|
||||
|
||||
def fake(brand, title, **k):
|
||||
calls.append(title)
|
||||
return next(answers)
|
||||
|
||||
monkeypatch.setattr(ollama_service, "fetch_product_details", fake)
|
||||
_run(["Product Name", "Brand"], [[name, "Amul"] for name in _AMUL_NAMES], use_llm=True)
|
||||
|
||||
# miss, miss, hit (reset), miss, miss, miss (trip) -> the 7th row is not asked.
|
||||
assert len(calls) == 6
|
||||
|
||||
|
||||
def test_stage_two_leaves_a_blank_description_blank_on_a_miss():
|
||||
"""The fallback lives in _to_storage_row; stage 2 must not pre-empt it."""
|
||||
row = {"brand": "Amul", "product_name": "Amul Butter 500g", "description": ""}
|
||||
breaker = pipeline.LlmBreaker()
|
||||
pipeline.stage_2_row_intake(row, use_llm=True, breaker=breaker)
|
||||
assert row["description"] == ""
|
||||
assert breaker.consecutive_misses == 1
|
||||
@@ -187,6 +187,45 @@ def test_template_headers_all_map_to_their_field():
|
||||
assert mapping.ignored == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("header, canonical", [
|
||||
# The camel-cased headers a store's export tool produces, which
|
||||
# _normalize_header collapses to one word - locked as exact aliases so
|
||||
# they do not depend on the keyword rules' ordering.
|
||||
("ProductName", "product_name"),
|
||||
("ProductBrand", "brand"),
|
||||
("Product Brand", "brand"),
|
||||
("ProductSKU", "product_sku"),
|
||||
("Product SKU", "product_sku"),
|
||||
# What the shop charges -> selling_price.
|
||||
("RetailPrice", "selling_price"),
|
||||
("Retail Price", "selling_price"),
|
||||
("Selling Price", "selling_price"),
|
||||
("Sale Price", "selling_price"),
|
||||
("SP", "selling_price"),
|
||||
("Price", "selling_price"),
|
||||
("Unit Price (Rs)", "selling_price"),
|
||||
# The tax-inclusive ceiling -> final_selling_price. "Final Selling Price"
|
||||
# contains "selling price" and must still win: rule order regression.
|
||||
("Final Selling Price", "final_selling_price"),
|
||||
("Final Price (₹)", "final_selling_price"),
|
||||
("MRP", "final_selling_price"),
|
||||
("Maximum MRP", "final_selling_price"),
|
||||
])
|
||||
def test_price_and_identity_headers_map_where_the_sheet_means_them(header, canonical):
|
||||
"""Before the selling_price canonical existed every price header landed in
|
||||
final_selling_price and the table's selling_price column stayed NULL."""
|
||||
mapping = user_products.map_spreadsheet_columns([header])
|
||||
assert mapping.columns == {canonical: header}
|
||||
|
||||
|
||||
def test_a_sheet_may_carry_both_a_retail_and_a_final_price():
|
||||
mapping = user_products.map_spreadsheet_columns(
|
||||
["Item Name", "Brand", "Retail Price", "MRP"])
|
||||
assert mapping.columns["selling_price"] == "Retail Price"
|
||||
assert mapping.columns["final_selling_price"] == "MRP"
|
||||
assert mapping.ignored == []
|
||||
|
||||
|
||||
def test_a_product_sku_column_is_not_mistaken_for_the_product_name():
|
||||
"""'Product SKU' contains 'product'. Matched loosely, it used to become a
|
||||
second product_name column, and duplicate columns are what turned a row
|
||||
|
||||
Reference in New Issue
Block a user