diff --git a/.env.example b/.env.example index 65286a1..5e6101f 100644 --- a/.env.example +++ b/.env.example @@ -267,6 +267,34 @@ BRAND_SYNC_INTERVAL_SECONDS=300 #ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever,Own Products +# --------------------------------------------------------------------------- +# Brand discovery (a brand NAME -> the 11-stage pipeline) +# --------------------------------------------------------------------------- +# POST /api/admin/brand-discovery/preview finds a brand's products, and +# /ingest stages the ones an admin approved as an ordinary catalog batch. +# Every value below has a working default; none of these need to be set. +# +# Open Food Facts is the primary source and the language model is the +# supplement. OFF returns real products with real barcodes and pack sizes; +# the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) will invent plausible ones, and +# nothing downstream can tell a well-formed fiction from a real product. Turn +# BRAND_DISCOVERY_USE_OFF off and the result rests on the model alone. +# +# NOTE: discovering a brand that is not in ACTIVE_BRANDS writes a complete +# catalog that no endpoint can read. The ingest route refuses with a 409 and +# names the line to add here; it is a config change plus a restart, never a +# re-ingest. +#BRAND_DISCOVERY_USE_OFF=true +#BRAND_DISCOVERY_USE_LLM=true +#BRAND_DISCOVERY_MAX_PRODUCTS=200 +# Pack sizes kept per product when only the language model offers any. Stage 6 +# runs an image search per exploded row, so this multiplies the slowest stage. +#BRAND_DISCOVERY_MAX_SIZES=3 +# Wall-clock ceiling on the language-model half, checked between prompts. Open +# Food Facts runs first and is never subject to it. +#BRAND_DISCOVERY_DEADLINE_SECONDS=300 + + USE_S3=true S3_ACCESS_KEY=your-do-spaces-key S3_SECRET_KEY=your-do-spaces-secret diff --git a/app/api/routers/brand_discovery.py b/app/api/routers/brand_discovery.py new file mode 100644 index 0000000..b09b680 --- /dev/null +++ b/app/api/routers/brand_discovery.py @@ -0,0 +1,253 @@ +"""Admin routes for brand discovery: a brand NAME into the 11-stage pipeline. + +Two steps on purpose, mirroring `batch_catalog.py`'s preview/ingest split. + +`/preview` discovers and returns; it writes nothing, anywhere. `/ingest` takes +the rows the admin kept, renders them as a CSV, and hands the bytes to the same +`batch_common.stage_and_queue` an uploaded spreadsheet goes through - so the +batch manifest, the stage timeline, Resume, Cancel and the nutrition +auto-enrichment that follows a batch all work here without a line of new code. + +The gap between the two steps is the point. Discovery's language-model half can +invent a product that nothing downstream is able to catch: a well-formed +fiction resolves a category, gets a price band and an internal SKU, and clears +`product_validator`'s "verified" threshold comfortably. `product_validator` was +built to reject MALFORMED rows, not false ones. A person looking at the list is +the check, so the list is shown before anything is written. +""" +from __future__ import annotations + +import logging +from typing import Any, Dict, List, Optional + +from fastapi import APIRouter, Depends, HTTPException, status +from pydantic import BaseModel, Field +from starlette.concurrency import run_in_threadpool + +from app.api import batch_common +from app.api.deps import require_admin +from app.core import store_catalog_pipeline as pipeline +from app.infrastructure.settings import ( + BATCH_MAX_FILES, + BATCH_MAX_TOTAL_BYTES, + BATCH_MAX_TOTAL_ROWS, + BRAND_DISCOVERY_DEADLINE_SECONDS, + BRAND_DISCOVERY_MAX_PRODUCTS, +) +from app.services import active_brands, brand_discovery + +logger = logging.getLogger(__name__) + +router = APIRouter(prefix="/admin/brand-discovery", tags=["admin", "catalog"]) + +# The same per-file ceilings the admin batch routes apply. Discovery emits one +# CSV row per product and pack-size explosion happens later, inside stage 4, so +# 200 products is 200 rows here - three orders of magnitude inside the limit. +MAX_UPLOAD_BYTES = 10 * 1024 * 1024 +MAX_UPLOAD_ROWS = 2000 + + +def _limits() -> batch_common.UploadLimits: + return batch_common.UploadLimits( + max_files=BATCH_MAX_FILES, + max_file_bytes=MAX_UPLOAD_BYTES, + max_file_rows=MAX_UPLOAD_ROWS, + max_total_bytes=BATCH_MAX_TOTAL_BYTES, + max_total_rows=BATCH_MAX_TOTAL_ROWS, + ) + + +# --------------------------------------------------------------------------- +# Request bodies +# --------------------------------------------------------------------------- +class DiscoveryPreviewRequest(BaseModel): + brand: str + max_products: int = Field(default=BRAND_DISCOVERY_MAX_PRODUCTS, ge=1, le=2000) + use_openfacts: bool = True + use_llm: bool = True + # Ungrounded language-model rows are dropped rather than shown by default. + # Turning this off is how an admin sees them - they arrive unticked. + require_evidence: bool = True + # Shorter than the service default: somebody is watching a spinner. + deadline_seconds: float = Field(default=90.0, ge=0.0, + le=BRAND_DISCOVERY_DEADLINE_SECONDS) + refresh_corpus: bool = False + + +class DiscoveredProductIn(BaseModel): + """One row the admin kept. Mirrors `DiscoveredProduct`'s written fields. + + Sent back rather than re-discovered so that what is ingested is exactly what + was reviewed - a second discovery pass could legitimately return something + different, and then the approval would have been of a different list. + """ + + product_name: str + title: Optional[str] = None + category: Optional[str] = None + description: Optional[str] = None + size_variants: List[str] = Field(default_factory=list) + providers: List[str] = Field(default_factory=list) + highlights: List[str] = Field(default_factory=list) + nutrients: List[str] = Field(default_factory=list) + fssai_license: Optional[str] = None + barcode: Optional[str] = None + image_url: Optional[str] = None + + +class DiscoveryIngestRequest(BaseModel): + brand: str + products: List[DiscoveredProductIn] + # Stage 2 only calls Ollama for a row whose description is blank, and + # discovery leaves most of them blank on purpose (see brand_discovery's note + # on the boilerplate generator). On an unreachable Ollama each such row + # costs up to OLLAMA_TIMEOUT_SECONDS, so this is worth being able to turn + # off for a large brand. + use_llm: bool = True + fetch_images: bool = True + # Ingesting a brand outside ACTIVE_BRANDS writes rows nothing can read. + # Refused unless the caller says they mean it - see the 409 below. + acknowledge_inactive: bool = False + + +# --------------------------------------------------------------------------- +# Routes +# --------------------------------------------------------------------------- +@router.post("/preview", dependencies=[Depends(require_admin)]) +async def preview_brand_discovery(payload: DiscoveryPreviewRequest) -> Dict[str, Any]: + """Discover a brand's products and return them. Writes nothing. + + Run on a worker thread: discovery does blocking HTTP to Open Food Facts and, + when the language model is enabled, a series of blocking Ollama calls. On + the event loop that would stall every other request for the duration. + """ + try: + result = await run_in_threadpool( + brand_discovery.discover_brand_products, + payload.brand, + max_products=payload.max_products, + deadline_seconds=payload.deadline_seconds, + use_openfacts=payload.use_openfacts, + use_llm=payload.use_llm, + require_evidence=payload.require_evidence, + refresh_corpus=payload.refresh_corpus, + ) + except ValueError as exc: + raise HTTPException(status_code=400, detail=str(exc)) from exc + except Exception as exc: # noqa: BLE001 - report the failure, do not 500 + logger.exception("Brand discovery failed for %r", payload.brand) + raise HTTPException( + status_code=502, + detail=f"Discovery failed for {payload.brand!r}: {exc}", + ) from exc + + body = result.as_dict() + # Served rather than duplicated in the frontend, exactly as BatchOut does, + # so the two cannot drift when a stage is added. + body["stages"] = list(pipeline.STAGE_NAMES) + if result.filtering_enabled and not result.brand_active: + body["warnings"] = list(body.get("warnings") or []) + [_inactive_message(result)] + return body + + +@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED, + dependencies=[Depends(require_admin)]) +async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_common.BatchOut: + """Stage the reviewed products as a catalog batch and return an id to poll.""" + brand = (payload.brand or "").strip() + if not brand: + raise HTTPException(status_code=400, detail="A brand name is required.") + if not payload.products: + raise HTTPException( + status_code=400, + detail="No products were selected, so there is nothing to ingest.", + ) + + # A green run over an unreadable catalog is the failure this project refuses + # to ship. The rows WOULD be written and fully enriched - nutrition + # auto-enrichment runs with include_inactive=True - but /api/brands, search, + # suggest and the category listing all filter the brand out, so the result + # looks like nothing happened. + if active_brands.filtering_enabled() and not active_brands.is_active_brand(brand): + if not payload.acknowledge_inactive: + raise HTTPException(status_code=409, detail=_inactive_detail(brand)) + + products = [ + brand_discovery.DiscoveredProduct( + brand=brand, + product_name=item.product_name, + title=item.title or item.product_name, + category=item.category or "", + category_hint="", + description=item.description or "", + size_variants=list(item.size_variants), + providers=list(item.providers), + highlights=list(item.highlights), + nutrients=list(item.nutrients), + fssai_license=item.fssai_license, + barcode=item.barcode, + image_url=item.image_url, + ) + for item in payload.products + ] + + filename = brand_discovery.synthetic_filename(brand) + contents = brand_discovery.rows_to_csv_bytes(products) + + # ONE FILE, NEVER CHUNKED. stage 11 groups by brand per file and reads the + # existing catalog per file, and its intra-file de-duplication + # ({image_id: row}) is per file too - so the same image_id split across two + # chunks would not be caught. A single file makes that de-duplication total. + valid, invalid, _rows = batch_common.parse_all([(filename, contents)], _limits()) + if not valid: + detail = "; ".join(f"{name}: {reason}" for name, reason in invalid) + raise HTTPException( + status_code=400, + detail=f"The discovered products could not be staged. {detail}", + ) + + manifest, started = batch_common.stage_and_queue( + valid, invalid, + use_llm=payload.use_llm, + fetch_images=payload.fetch_images, + submitted_by=f"brand-discovery: {brand}", + ) + if not started: + raise HTTPException( + status_code=429, + detail=( + "Too many batches are already queued. This one has been saved - " + "press Resume on it once the current batch finishes." + ), + ) + return batch_common.to_out(manifest) + + +# --------------------------------------------------------------------------- +# The ACTIVE_BRANDS message, in one place +# --------------------------------------------------------------------------- +def _env_line(brand: str) -> str: + names = active_brands.active_display_names() + return "ACTIVE_BRANDS=" + ",".join(list(names) + [brand]) + + +def _inactive_message(result: brand_discovery.DiscoveryResult) -> str: + return ( + f"{result.brand} is not in ACTIVE_BRANDS, so these products would be " + f"written to {result.table} and then filtered out of /api/brands, " + f"search, suggest and the category listing. The rows would be complete " + f"and correct, just unreadable. To make them visible, set " + f"'{_env_line(result.brand)}' in backend/.env and restart the API - " + f"settings are read once at import, so a restart is required." + ) + + +def _inactive_detail(brand: str) -> str: + return ( + f"{brand} is not in ACTIVE_BRANDS. Ingesting it now would write a " + f"complete catalog that no endpoint can read. Either set " + f"'{_env_line(brand)}' in backend/.env and restart the API first, or " + f"re-send with acknowledge_inactive=true to stage the data anyway - " + f"adding the brand later is a config change and a restart, not a " + f"re-ingest." + ) diff --git a/app/api/routers/catalog.py b/app/api/routers/catalog.py index 08ce2d8..1825cea 100644 --- a/app/api/routers/catalog.py +++ b/app/api/routers/catalog.py @@ -48,6 +48,20 @@ def generate_catalog(payload: CatalogGenerateRequest) -> CatalogJobOut: """Kick off brand catalog ingestion (discovery -> images -> embeddings -> pgvector) as a background daemon thread and return immediately with a job id. + PREFER /api/admin/brand-discovery/* FOR NEW WORK. This route is unchanged + and still supported, but it does not run the eleven stages in + `app/core/store_catalog_pipeline.py` - no title validation, no pack-size + explosion, no SKU resolution, no barcode, no HSN/GST, no validation gate - + and it mints `image_id` with `s3_service.generate_image_id()`, which appends + a random uuid4. Nothing it writes can ever match an existing row, so running + it twice for one brand produces two catalogs. It also calls + `upsert_brand_products(cleanup=True)`, which deletes every row not in the + batch it just built. + + The discovery routes do run all eleven stages, use a deterministic + `image_id`, write with `cleanup=False`, and show the products for approval + before anything is stored. See docs/BRAND_DISCOVERY.md. + NOTE: on an 8GB RAM / CPU-only machine, running ingestion (which loads the embeddings model and calls Ollama repeatedly) at the same time as heavy chat traffic will be slow. This is intended as an occasional diff --git a/app/core/ingestion.py b/app/core/ingestion.py index d8ea958..58ce935 100644 --- a/app/core/ingestion.py +++ b/app/core/ingestion.py @@ -8,6 +8,18 @@ product discovery (Ollama), per-product image search + S3 upload, pricing/description enrichment, embedding generation, and the pgvector upsert. This module exists only to give that pipeline one clear, reusable entry point and a consistent result shape for callers. + +NOT THE ELEVEN-STAGE PIPELINE, and prefer `app/services/brand_discovery.py` +for new work. "Full pipeline" above means this module's own sequence, not the +eleven stages in `app/core/store_catalog_pipeline.py`: there is no title +validation, no pack-size explosion, no SKU service, no barcode, no HSN/GST and +no validation gate here, and `image_id` carries a random uuid4 suffix, so a +second run for the same brand cannot match the first and duplicates it. + +This path is unchanged and still works. It also rewrites the brand's seed +catalog via `brand_sync.export_brand_to_seed_file` (below), which the discovery +path deliberately does not - so the two are not drop-in replacements for each +other. See docs/BRAND_DISCOVERY.md. """ from __future__ import annotations diff --git a/app/infrastructure/settings.py b/app/infrastructure/settings.py index ae4f064..09d049d 100644 --- a/app/infrastructure/settings.py +++ b/app/infrastructure/settings.py @@ -316,6 +316,37 @@ BRAND_SYNC_INTERVAL_SECONDS = int(os.getenv("BRAND_SYNC_INTERVAL_SECONDS", "300" # See app/services/active_brands.py. ACTIVE_BRANDS = os.getenv("ACTIVE_BRANDS", "") +# --------------------------------------------------------------------------- +# Brand discovery (brand name -> the 11-stage pipeline) +# --------------------------------------------------------------------------- +# Discovery turns a brand NAME into rows the ordinary catalog pipeline ingests. +# See app/services/brand_discovery.py. Every value below has a working default, +# so the feature needs no configuration to run. +# +# Open Food Facts is the primary source and the language model is the +# supplement, not the reverse: OFF returns real products carrying a real GTIN, +# while the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) invents plausible ones that +# nothing downstream can catch. Turning BRAND_DISCOVERY_USE_OFF off leaves the +# result resting on the model alone. +BRAND_DISCOVERY_USE_OFF = _bool("BRAND_DISCOVERY_USE_OFF", "true") +BRAND_DISCOVERY_USE_LLM = _bool("BRAND_DISCOVERY_USE_LLM", "true") + +# Products per discovery run. One CSV row per product; pack-size explosion +# happens later in stage 4, so this is well inside the 2000-row per-file cap. +BRAND_DISCOVERY_MAX_PRODUCTS = int(os.getenv("BRAND_DISCOVERY_MAX_PRODUCTS", "200")) + +# Pack sizes kept per product when only the language model offers any. Stage 6 +# runs an image search per exploded row, so this multiplies the slowest part of +# the run; 3 keeps a large brand inside a sane wall-clock. +BRAND_DISCOVERY_MAX_SIZES = int(os.getenv("BRAND_DISCOVERY_MAX_SIZES", "3")) + +# Wall-clock ceiling on the LLM half of a run, checked between prompts. Open +# Food Facts runs first and is never subject to it, so a run that hits this +# still returns the evidence-backed products. +BRAND_DISCOVERY_DEADLINE_SECONDS = float( + os.getenv("BRAND_DISCOVERY_DEADLINE_SECONDS", "300") +) + # --------------------------------------------------------------------------- # S3 / DigitalOcean Spaces (product image storage) - optional # --------------------------------------------------------------------------- diff --git a/app/main.py b/app/main.py index dc423d1..063d298 100644 --- a/app/main.py +++ b/app/main.py @@ -31,7 +31,7 @@ from app.api.routers import health, brands, search, suggest, chat, catalog, syst from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin from app.api.routers import nutrition, nutrition_admin, upload from app.api.routers import auth, user_products, admin_train, mcp_info -from app.api.routers import batch_catalog, uploads +from app.api.routers import batch_catalog, uploads, brand_discovery from app.services.store_db import ensure_store_intelligence_schema from app.services.nutrition_db import ensure_nutrition_schema @@ -307,6 +307,7 @@ app.include_router(nutrition_admin.router, prefix="/api") app.include_router(upload.router, prefix="/api") app.include_router(batch_catalog.router, prefix="/api") app.include_router(uploads.router, prefix="/api") +app.include_router(brand_discovery.router, prefix="/api") app.include_router(mcp_info.router, prefix="/api") # MCP lives outside /api on purpose: it is a protocol endpoint for AI clients, diff --git a/app/services/brand_discovery.py b/app/services/brand_discovery.py new file mode 100644 index 0000000..053f135 --- /dev/null +++ b/app/services/brand_discovery.py @@ -0,0 +1,1041 @@ +""" +Brand Discovery - turn a BRAND NAME into rows the 11-stage pipeline can ingest +============================================================================== + +WHY THIS FILE EXISTS +-------------------- +`app/core/store_catalog_pipeline.py` runs eleven stages over a SPREADSHEET and +does it well: deterministic `image_id`, fill-only-blanks merging, a validation +gate, `cleanup=False`. But it can only start from a file somebody uploads. + +The only way to start from a brand name was +`catalog_engine.generate_catalog()` (`POST /api/catalog/generate`), which skips +title validation, pack-size explosion, SKU resolution, barcode, HSN/GST and the +validation gate - and mints `image_id` via `s3_service.generate_image_id()`, +which appends a random uuid4 and therefore can never match an existing row. Run +it twice and you get two catalogs. + +This module closes that gap WITHOUT touching the pipeline. It discovers a +brand's products, then serialises them as a CSV whose headers are exactly the +canonical spreadsheet fields, and hands the bytes to the ordinary batch +machinery. `run_pipeline`, `row_to_request`, every stage function and the +worker are all untouched; they cannot tell a discovered file from an uploaded +one, which is the entire point. + + brand name + | + +-- off_bulk.fetch_brand_corpus() real names + GTIN + pack size + +-- ollama_service breadth, categories, prose + | + merge / dedupe / evidence + | + rows_to_csv_bytes() -> the existing batch pipeline, unchanged + +WHY OPEN FOOD FACTS IS THE PRIMARY SOURCE, NOT THE LLM +------------------------------------------------------ +`OLLAMA_MODEL_NAME` defaults to `qwen2.5:1.5b` on a CPU-only box. Asked to +enumerate a brand's catalog it invents plausible products, and there is no +downstream check that can catch a well-formed fiction: a made-up biscuit gets a +resolved category (+0.15), a synthesised price band, an internal SKU and a +plausible size, and clears `product_validator`'s 0.70 "verified" threshold +comfortably. The validator was built to catch MALFORMED rows, not false ones. + +Open Food Facts is a real database. `off_bulk.fetch_brand_corpus("Britannia")` +returns 218 actual products carrying a real GTIN and, for 135 of them, a real +pack size - in about five seconds, from a disk cache, with no LLM involved at +all. So OFF is ground truth and the LLM is the supplement, never the reverse. +A discovery run against a dead Ollama still produces a useful catalog. + +TWO BUGS THIS MODULE EXISTS TO PREVENT +-------------------------------------- +Both were found by round-tripping real Britannia data, and both are silent. + +1. **OFF pack sizes double into the image_id.** OFF writes quantities both ways + - the Britannia corpus has "100g" six times and "100 g" four times. + `build_image_id` appends the size unless `_slugify(size)` is a substring of + the slugified name, and `_slugify("200 g")` is `"200_g"` while the name + contains `"200g"`. So a raw OFF quantity produces + + britannia_britannia_good_day_200g_200_g + + and the same product discovered later as "200g" produces + + britannia_britannia_good_day_200g + + Two permanent rows for one product. `_canonical_size()` fixes this by + rebuilding every size through `category_units.parse_unit`. + +2. **List cells are shredded by the separator.** `user_products._string_list` + splits on `[,;|]`, and `catalog_engine.generate_product_highlights` emits + "Available in 3 sizes: 100g, 250g, 500g" and "Baked, Not Fried". Through the + bridge those two highlights become five fragments. `_safe_list_cell()` + replaces the separators inside each item before joining. + +THE THIRD RISK: CATALOGUE GROWTH BY RE-RUN +------------------------------------------ +Discovery is not byte-identical between runs. Run 1 yields "Good Day Cashew +Cookies", run 2 "Britannia Good Day Cashew" - different slug, different +`image_id`, a second row. Over a few runs the table becomes a swamp of +near-duplicates while every run reports green. + +`_existing_title_index()` is the guard: a discovered title whose normalised +form already exists in the brand table is re-emitted under the STORED +`product_name`, never the newly discovered wording. That makes `image_id` +stable across runs by construction and turns every re-run into pure backfill. +It is the single most important function in this module. +""" +from __future__ import annotations + +import csv +import io +import logging +import re +import time +from dataclasses import dataclass, field +from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple + +from app.core.catalog_engine import ( + generate_nutrients_info, + generate_product_highlights, +) +from app.infrastructure.settings import ( + BRAND_DISCOVERY_DEADLINE_SECONDS, + BRAND_DISCOVERY_MAX_PRODUCTS, + BRAND_DISCOVERY_MAX_SIZES, + BRAND_DISCOVERY_USE_LLM, + BRAND_DISCOVERY_USE_OFF, +) +from app.services import active_brands, ollama_service +from app.services.brand_registry import ( + get_fssai_license, + get_known_sub_brands, + resolve_parent_brand, +) +from app.services.category_registry import detect_category_from_text +from app.services.category_units import ( + COUNT_UNITS, + VOLUME_UNITS, + WEIGHT_UNITS, + parse_unit, +) +from app.services.enrichment.barcode.sources import off_bulk +from app.services.vector_store import _sanitize_name, get_products_by_brand + +logger = logging.getLogger(__name__) + +# The canonical spreadsheet headers. Every one of these resolves through +# `user_products._canonical_field` with no collisions and nothing unrecognised - +# `test_brand_discovery.py::test_the_csv_headers_all_map_onto_catalog_fields` +# asserts it against the real mapper, because the whole bridge rests on it. +# +# Deliberately ABSENT: `barcode_type` and `selling_price`. `row_to_request` +# (user_products.py:329-346) builds a fixed field list and carries neither, so a +# column for them would be silently dropped. Emitting one would be a lie in the +# file about what reaches the database. +CSV_HEADERS: Tuple[str, ...] = ( + "brand", "product_name", "title", "category", "description", + "price_range", "size_variants", "providers", "highlights", "nutrients", + "fssai_license", "product_sku", "hsn_code", "barcode", "image_url", +) + +# Units a pack size may legitimately carry. A quantity whose unit is not one of +# these is not a size: the Britannia corpus contains "India" and bare counts +# like "12" in the quantity field, and `_sizes_for` would drop them anyway with +# "a number with no unit is a quantity, not a size". +_SIZE_UNITS = WEIGHT_UNITS | VOLUME_UNITS | COUNT_UNITS + +# Anything that would split a list cell when it reaches `_string_list`. +_LIST_SEPARATORS = re.compile(r"[,;|]") + +# A pack size embedded in a product NAME. Deliberately NOT `off_bulk._SIZE_RE`, +# which anchors the number on a word boundary and so cannot see a size glued to +# a letter: the Britannia corpus contains "Jim Jam92 g", where there is no +# boundary between "m" and "9". The anchored form leaves that alone, and the +# unnormalised size then survives into the image_id as +# `britannia_jim_jam92_g_40g`. This form matches the digits wherever they sit, +# and the replacement re-inserts the missing space. +# +# LONGEST UNIT FIRST. Python alternation is first-match-wins, so a list with +# "g" ahead of "gms" reads "500gms" as "500g" and strands "ms" in the name. +# The trailing word boundary stops "1 litre" being cut down to "1 l". +_TITLE_SIZE_UNITS = ( + "kilograms|kilogram|kilos|kilo|kgs|kg|grams|gram|gms|gm|mg|" + "millilitres|milliliters|litres|liters|litre|liter|ltrs|ltr|mls|ml|" + "pieces|piece|pcs|pc|nos|g|l" +) +_TITLE_SIZE_RE = re.compile( + r"(\d+(?:[.,]\d+)?)\s*(" + _TITLE_SIZE_UNITS + r")\b", + re.IGNORECASE, +) + +# Marketplaces the seed catalogs already advertise. Used when the LLM offers no +# `provider_examples`, so the column is never empty for want of a guess. +_DEFAULT_PROVIDERS: Tuple[str, ...] = ( + "Amazon", "Flipkart", "BigBasket", "Jiomart", "Blinkit", "Zepto", +) + +# How far a title may drift and still be judged the same product. 0.85 is what +# `off_bulk` uses for barcode attachment, and the question here is the same one: +# "is this the product we already have?" +_TITLE_MATCH_THRESHOLD = 0.85 + +# Products asked for per LLM prompt. Deliberately small. The existing +# `fetch_brand_catalog_exhaustive` asks for "at least 50 items" in one +# completion - roughly 1500-2500 output tokens, which at the 10-20 tok/s a +# 1.5B model manages on a CPU-only box needs 100-250 seconds against an +# OLLAMA_TIMEOUT_SECONDS of 120. It times out, retries twice, and returns +# nothing. That is why brand tables seeded through the legacy path hold three +# products instead of two hundred. Breadth here comes from the NUMBER of +# prompts, not the size of one. +_ITEMS_PER_PROMPT = 12 + + +# --------------------------------------------------------------------------- +# Result types +# --------------------------------------------------------------------------- +@dataclass +class DiscoveredProduct: + """One product, ready to become a CSV row. + + Everything below `image_url` is preview metadata: it is reported to the + admin so they can judge a row, and is never written to the file. + """ + + brand: str + product_name: str + title: str + # Blank unless a source stated it. See `_build_product`: stage 3 owns + # category detection, and a value here is treated by it as authoritative. + category: str + # What stage 3 will most likely decide, computed with the same function it + # uses. Shown in the preview and fed to the highlight/nutrient generators; + # never written to the CSV. + category_hint: str + description: str + size_variants: List[str] = field(default_factory=list) + providers: List[str] = field(default_factory=list) + highlights: List[str] = field(default_factory=list) + nutrients: List[str] = field(default_factory=list) + fssai_license: Optional[str] = None + barcode: Optional[str] = None + image_url: Optional[str] = None + + sources: List[str] = field(default_factory=list) + evidence: Optional[str] = None + confidence: float = 0.0 + matches_existing: Optional[str] = None + notes: List[str] = field(default_factory=list) + + def as_preview(self) -> Dict[str, Any]: + return { + "brand": self.brand, + "product_name": self.product_name, + "title": self.title, + "category": self.category, + "category_hint": self.category_hint, + "description": self.description, + "size_variants": list(self.size_variants), + "providers": list(self.providers), + "highlights": list(self.highlights), + "nutrients": list(self.nutrients), + "fssai_license": self.fssai_license, + "barcode": self.barcode, + "image_url": self.image_url, + "sources": list(self.sources), + "evidence": self.evidence, + "confidence": round(self.confidence, 2), + "matches_existing": self.matches_existing, + "notes": list(self.notes), + "selected": self.confidence >= 0.5, + } + + +@dataclass +class DiscoveryResult: + brand: str + parent_brand: str + table: str + brand_active: bool + filtering_enabled: bool + products: List[DiscoveredProduct] = field(default_factory=list) + counts: Dict[str, int] = field(default_factory=dict) + warnings: List[str] = field(default_factory=list) + + def as_dict(self) -> Dict[str, Any]: + return { + "brand": self.brand, + "parent_brand": self.parent_brand, + "table": self.table, + "brand_active": self.brand_active, + "filtering_enabled": self.filtering_enabled, + "products": [p.as_preview() for p in self.products], + "counts": dict(self.counts), + "warnings": list(self.warnings), + } + + +# --------------------------------------------------------------------------- +# Normalisation - the three functions that keep re-runs idempotent +# --------------------------------------------------------------------------- +def _canonical_size(raw: Optional[str]) -> Optional[str]: + """Rebuild a pack size in the one form the pipeline hashes consistently. + + "200 g" and "200g" are the same pack, but `build_image_id` slugifies them to + "200_g" and "200g" and therefore produces two different rows. OFF writes + both forms for the same brand, so this is not hypothetical - it would have + split the majority of the 135 sized Britannia products. + + Returns None for anything that is not a pack size: a bare count ("12"), a + country name ("India"), or a unit this catalog does not measure in. The + caller then falls back to a size in the title, which is what a real product + name usually carries anyway. + """ + if not raw or not str(raw).strip(): + return None + value, unit = parse_unit(str(raw)) + if value is None or not unit or unit not in _SIZE_UNITS: + return None + number = str(int(value)) if float(value).is_integer() else str(value) + return f"{number}{unit}" + + +def _safe_list_cell(values: Iterable[str]) -> str: + """Join items for a cell `_string_list` will split back on `[,;|]`. + + Each item has those characters replaced with " / " first, so a highlight + reading "Available in 3 sizes: 100g, 250g, 500g" survives as ONE item + rather than becoming three. Without this the round trip silently multiplies + every list-valued column. + """ + cleaned: List[str] = [] + for value in values or (): + text = _LIST_SEPARATORS.sub(" / ", str(value)).strip() + text = re.sub(r"\s+", " ", text) + if text: + cleaned.append(text) + return ";".join(cleaned) + + +def _canonicalise_title_size(title: str) -> str: + """Rewrite a pack size embedded in a product NAME into canonical form. + + `_canonical_size` fixes the size COLUMN, but OFF also writes sizes into the + name itself, and unevenly: the Britannia corpus contains "Jim Jam92 g". + `build_image_id` folds the size in only when `_slugify(size)` is a substring + of the slugified name, so "Jim Jam92 g" + size "40g" produced + + britannia_jim_jam92_g_40g + + while the same product named "Jim Jam 92g" produces a different id. The name + is half of every image_id, so it has to be normalised too or the doubled + size simply moves from the size column into the title. + """ + def _fix(match: re.Match) -> str: + canonical = _canonical_size(match.group(0)) + return f" {canonical}" if canonical else match.group(0) + + fixed = _TITLE_SIZE_RE.sub(_fix, title or "") + return re.sub(r"\s+", " ", fixed).strip() + + +def _normalise_title(brand: str, title: Optional[str]) -> str: + """The key two titles are judged to be the same product by. + + Delegates to `off_bulk.normalize_for_match`, which strips pack sizes, pack + parentheticals and the brand's own tokens - the machinery that already + powers barcode attachment and is tested by `test_off_bulk_backfill.py`. + + Returns "" when nothing survives, and callers MUST treat that as "no + identity" rather than falling back to the raw title. A row named only by its + brand and a size ("Britannia 200g") normalises to nothing, and matching two + of those against each other would fold unrelated products together. + """ + return off_bulk.normalize_for_match(title, off_bulk.brand_tokens(brand)) + + +def _same_product(left: str, right: str, left_raw: str, right_raw: str) -> bool: + """Whether two normalised titles name the same retail product. + + The threshold stays at `off_bulk`'s 0.85 rather than being relaxed to catch + more re-run drift, and the cost of that is worth stating: "Good Day Cashew" + against a stored "Good Day Cashew Cookies" scores 0.75 and will NOT fold, so + it becomes a second row. Any threshold low enough to catch it also folds + "Dairy Milk Silk" onto "Dairy Milk Silk Minis" - a different product - + which is the case `symmetric_similarity` was written to reject. + + That limitation is acceptable because Open Food Facts, not the language + model, supplies the names: OFF titles are stable between runs, so the + wording does not drift in the first place. It is the LLM half that can + invent a new phrasing, and those rows are the minority and are shown in the + preview before anything is written. + """ + if not left or not right: + return False + if left == right: + return True + if off_bulk.has_extra_variant_conflict(left_raw, right_raw): + return False + return off_bulk.symmetric_similarity(left, right) >= _TITLE_MATCH_THRESHOLD + + +def _match_existing(normalised: str, title: str, + catalog_keys: Dict[str, str]) -> Optional[str]: + """The stored base name for this product, or None. + + Exact first, then the same similarity rule the source merge uses. Sharing + one predicate matters: if the catalog fold were stricter than the merge, two + OFF spellings could merge with each other and then fail to recognise the row + they had already been written as. + """ + if not normalised: + return None + exact = catalog_keys.get(normalised) + if exact: + return exact + for key, stored in catalog_keys.items(): + if _same_product(normalised, key, title, stored): + return stored + return None + + +# --------------------------------------------------------------------------- +# The existing catalog - read once, and the reason re-runs are a no-op +# --------------------------------------------------------------------------- +def _existing_title_index(brand: str) -> Dict[str, str]: + """{normalised title -> the stored product's name WITHOUT its pack size}. + + Discovery re-emits matched products under the stored wording, so a second + run produces the same `image_id` and `_merge_with_existing` records it as + `skipped_existing` instead of inserting a near-duplicate. Without this the + feature grows the catalog every time it runs. + + THE SIZE IS STRIPPED, and that is not cosmetic. A stored `product_name` is a + DISPLAY name that `_to_storage_row` has already had the pack size folded + into ("Jim Jam 25g"), while the key here is size-free by construction. Two + real packs therefore share one key, and re-emitting the first stored name + verbatim gave the 92g pack the 25g name - which `_to_storage_row` then + extended to "Jim Jam 25g 92g", a product that does not exist, under a brand + new image_id. Handing back the base name lets the size be re-applied per + variant, so each pack lands back on its own row. + + Differing capitalisation between the stored rows is harmless: `_slugify` + lowercases, so "Jim Jam" and "Jim jam" produce the same image_id. + + A failure to read is not fatal - it costs idempotency for this run, which is + worth a warning rather than refusing to discover anything at all. + """ + index: Dict[str, str] = {} + try: + rows = get_products_by_brand(brand) + except Exception as exc: # noqa: BLE001 - never block discovery on a read + logger.warning("Could not read existing products for %r: %s", brand, exc) + return index + + for row in rows or (): + name = (row.get("product_name") or "").strip() + if not name: + continue + key = _normalise_title(brand, name) + if not key: + continue + # Fall back to the full name when stripping leaves nothing - a row + # titled only by its size has no other identity to offer. + base = off_bulk.strip_sizes(name).strip() or name + index.setdefault(key, base) + return index + + +# --------------------------------------------------------------------------- +# Source A - Open Food Facts (ground truth) +# --------------------------------------------------------------------------- +def _from_open_facts(brand: str, *, refresh: bool = False) -> List[Dict[str, Any]]: + """Real products for `brand`, with a real GTIN and often a real pack size. + + `off_bulk.fetch_brand_corpus` is already disk-cached, retry-wrapped and + paced at 2s between pages, and `scripts/backfill_barcodes_from_off.py` + already imports it from outside the enrichment package, so reaching for it + here is an established pattern rather than a new one. + """ + try: + hits = off_bulk.fetch_brand_corpus(brand, refresh=refresh) + except Exception as exc: # noqa: BLE001 - a dead OFF must not kill discovery + logger.warning("Open Food Facts lookup failed for %r: %s", brand, exc) + return [] + + out: List[Dict[str, Any]] = [] + for hit in hits or (): + title = (hit.get("product_name_en") or hit.get("product_name") or "").strip() + if not title: + continue + barcode = None + accepted = off_bulk.accept_barcode(hit.get("code")) + if accepted: + barcode = accepted[0] + out.append({ + "title": title, + "barcode": barcode, + "size": _canonical_size(hit.get("quantity")), + "source": "off", + }) + return out + + +# --------------------------------------------------------------------------- +# Source B - Ollama (breadth and prose) +# --------------------------------------------------------------------------- +def _ask_for_products(brand: str, subject: str, context: str) -> List[Dict[str, Any]]: + """One small prompt. Returns [] on any failure, never raises. + + Small on purpose - see `_ITEMS_PER_PROMPT`. A prompt that does not fit + inside `OLLAMA_TIMEOUT_SECONDS` returns nothing at all, so twelve items that + arrive beat fifty that time out. + """ + prompt = ( + f"Brand: {brand}\n" + f"{subject}\n" + f"Return a JSON object with key 'products' (array of at most " + f"{_ITEMS_PER_PROMPT} objects). Each product: title, category, " + f"description (<=160 chars), size_variants (array of pack sizes such as " + f'"75g"), provider_examples (array of up to 4 retailers).\n' + f"Only list products this brand actually sells. Do not invent products." + f"{context}" + ) + try: + text = ollama_service._generate(ollama_service.SYSTEM_PROMPT, prompt) + except Exception as exc: # noqa: BLE001 + logger.warning("LLM discovery prompt failed for %r: %s", brand, exc) + return [] + if not text: + return [] + + parsed = ollama_service._extract_json(text) + if parsed is None: + return [] + raw = parsed if isinstance(parsed, list) else parsed.get("products", []) + if not isinstance(raw, list): + return [] + + out: List[Dict[str, Any]] = [] + for item in raw: + if not isinstance(item, dict): + continue + title = (item.get("title") or item.get("name") or "").strip() + if not title: + continue + sizes = item.get("size_variants") + out.append({ + "title": title, + "category": (item.get("category") or "").strip() or None, + "description": (item.get("description") or "").strip() or None, + "sizes": [s for s in (sizes if isinstance(sizes, list) else []) if s], + "providers": [ + p for p in (item.get("provider_examples") or []) if isinstance(p, str) + ], + "source": "llm", + }) + return out + + +def _from_llm(brand: str, *, deadline: float, budget: int) -> List[Dict[str, Any]]: + """Breadth from many small prompts rather than one large one. + + Two passes. Categories first, because `get_categories_for_brand` is already + grounded by the alias registry; then one prompt per known sub-brand, which + is where depth actually comes from - "list the variants of Britannia Good + Day" is a question a small model can answer well, unlike "list everything + Britannia sells". + + Serial by design: one Ollama process on an 8GB box, and concurrent requests + only make it swap. + """ + known = get_known_sub_brands(brand) + context = ( + f"\nKnown product lines under this brand: {', '.join(known)}." + if known else "" + ) + + subjects: List[str] = [] + try: + for category in ollama_service.get_categories_for_brand(brand) or (): + subjects.append(f"List products in the category: {category}") + except Exception as exc: # noqa: BLE001 + logger.warning("Category enumeration failed for %r: %s", brand, exc) + for sub in known: + subjects.append(f"List the pack variants of the product line: {brand} {sub}") + + collected: List[Dict[str, Any]] = [] + barren = 0 + for subject in subjects: + if len(collected) >= budget: + break + if time.monotonic() >= deadline: + logger.info("LLM discovery for %r stopped at the deadline", brand) + break + found = _ask_for_products(brand, subject, context) + collected.extend(found) + # Two silent prompts in a row means the model has nothing more to add or + # is not answering at all; either way further prompts are a waste of the + # remaining budget. + barren = 0 if found else barren + 1 + if barren >= 2: + logger.info("LLM discovery for %r stopped after two empty prompts", brand) + break + return collected + + +# --------------------------------------------------------------------------- +# Evidence +# --------------------------------------------------------------------------- +def _registry_terms(brand: str) -> List[str]: + """Distinctive tokens from the brand's known sub-brands.""" + terms: List[str] = [] + for sub in get_known_sub_brands(brand): + terms.extend(t for t in re.split(r"[^a-z0-9]+", sub.lower()) if len(t) > 2) + return terms + + +def _evidence_for(normalised: str, *, off_keys: Dict[str, Any], + registry_terms: Sequence[str], catalog_keys: Dict[str, str]) -> Optional[str]: + """What corroborates this product, cheapest source first. + + Three tiers, none of which cost a network call at this point: the OFF index + was built during the sweep, the registry terms are a Python constant, and + the catalog index was read once. + + Deliberately NOT a tier: "an image search returned a URL naming this + product". `catalog_engine._select_best_images` documents that CDN filenames + are frequently opaque hashes, so absence of a naming URL is weak evidence of + absence and would drop real products. Image agreement is reported as a note + elsewhere, never used to reject. + """ + if not normalised: + return None + if normalised in off_keys: + return "openfacts" + if normalised in catalog_keys: + return "catalog" + tokens = set(normalised.split()) + if tokens & set(registry_terms): + return "registry" + return None + + +def _score(sources: Sequence[str], evidence: Optional[str]) -> float: + """Confidence, and therefore whether the row is ticked by default. + + An LLM-only row with nothing corroborating it scores below the 0.5 the + preview ticks at, so it is shown and explained but never silently ingested. + """ + has_off = "off" in sources + has_llm = "llm" in sources + if has_off and has_llm: + return 1.0 + if has_off: + return 0.9 + if evidence == "catalog": + return 0.7 + if evidence == "registry": + return 0.55 + return 0.25 + + +# --------------------------------------------------------------------------- +# Assembly +# --------------------------------------------------------------------------- +def _resolve_sizes(candidate: Dict[str, Any], title: str, category: str, + max_sizes: int) -> List[str]: + """Pack sizes for one product, real ones preferred. + + THE TITLE'S OWN SIZE WINS, ahead of the OFF `quantity` column. Both are real + data and they do disagree - OFF holds "Britannia Toastea 200g" with a + quantity of "250g", and "Britannia 5050 Potazos 71.5g" with "29g", because + the quantity field often describes a multipack while the name describes the + unit. Taking the quantity there produced the name "Britannia Toastea 200g + 250g": two sizes in one product name, for a pack that does not exist. + + Preferring the title also keeps the name and the size column consistent, + which is what makes `image_id` reproducible - the name is half of it. + + The LLM's list is a guess and comes last. Only when all three are empty is + the cell left blank, which hands the decision to stage 4's + `default_size_variants` - and that invents 100g/250g/500g, which + `_sizes_for`'s own comment identifies as the mechanism behind observed + catalogue drift. Leaving it blank is the last resort, not the default. + """ + in_title = _TITLE_SIZE_RE.search(title or "") + if in_title: + canonical = _canonical_size(in_title.group(0)) + if canonical: + return [canonical] + + off_size = candidate.get("size") + if off_size: + return [off_size] + + sizes: List[str] = [] + for raw in candidate.get("sizes") or (): + canonical = _canonical_size(raw) + if canonical and canonical not in sizes: + sizes.append(canonical) + if len(sizes) >= max_sizes: + break + return sizes + + +def _build_product(brand: str, candidate: Dict[str, Any], *, max_sizes: int, + catalog_keys: Dict[str, str], off_keys: Dict[str, Any], + registry_terms: Sequence[str], + fssai: Optional[str]) -> Optional[DiscoveredProduct]: + raw_title = _canonicalise_title_size((candidate.get("title") or "").strip()) + if not raw_title: + return None + normalised = _normalise_title(brand, raw_title) + + # THE PACK SIZE LIVES IN ONE PLACE: the size column. `_to_storage_row` + # appends the size to the name to build the display name, so a size left in + # the title is applied twice - OFF's "Britannia Toastea 200g" with a + # quantity of 250g became the product "Britannia Toastea 200g 250g". + # + # `strip_sizes` also removes OFF's case-pack parentheticals ("Brit 50-50 + # maska chaska 40g (20)", "Good Day Butter Cookies (25)"). Those are how + # many units are in a carton, not part of the product's name, and leaving + # them in put the carton count into the image_id. + # + # Both mattered on the very first run, not only on a re-run: seven Britannia + # products were stored under a name no shelf has ever carried. + # + # Keep the original when stripping empties the string - a product named only + # by its size has no other identity to offer. + title = off_bulk.strip_sizes(raw_title).strip() or raw_title + + # Re-emit under the stored base name when we already hold this product, so + # the image_id matches and stage 11 backfills instead of inserting. See the + # module docstring: this is what keeps re-runs idempotent. + matches_existing = _match_existing(normalised, raw_title, catalog_keys) + product_name = matches_existing or title + + # A category is written to the CSV only when a SOURCE stated one. Stage 3 + # treats a supplied category as authoritative - it sets + # `_category_deterministic` and then runs `validate_and_fix_title` against + # it, so a wrong guess here does not merely mislabel the row, it rewrites + # the product name. The keyword detector is genuinely wrong on real data + # ("50 50 Sweet & Salty" -> "Salt & Staples", "50 50 Gol Maal" -> "Spices & + # Masalas"), and stage 3 computes the same value from the same function with + # MORE signal than we have here, because it sees the description too. So the + # detector's answer is carried as a hint for the preview and for the two + # generators below, and the column is left blank for stage 3 to own. + stated_category = (candidate.get("category") or "").strip() + hint = detect_category_from_text(f"{title} {candidate.get('description') or ''}") + category_hint = stated_category or hint or "General" + # `raw_title`, not `title`: the size has just been stripped out of the + # latter, and it is the size we are trying to read off it. + sizes = _resolve_sizes(candidate, raw_title, category_hint, max_sizes) + + # NO GENERATED DESCRIPTION. `catalog_engine.generate_detailed_description` + # is available and was used here at first, and it is actively harmful: + # + # * Its boilerplate contains the word "taste", `detect_category_from_text` + # matches keywords FUZZILY, and "taste" is one edit from the Oral Care + # keyword "paste". Since stage 3 detects from title AND description, + # every product whose title carries no strong keyword was classified + # Oral Care - 127 of 245 Britannia rows - and stage 9 then stamped them + # with HSN 3306, the tax code for dentifrices. A wrong tax code is worse + # than a missing one, which is a rule this pipeline already states. + # * The same ~1200 characters land in `search_query`, which is what gets + # embedded. Hundreds of rows sharing one dominant paragraph degrades + # semantic search. + # + # A blank description is honest and already handled: `_to_storage_row` + # substitutes " from ." exactly as it does for a store + # sheet that left the column empty. + description = (candidate.get("description") or "").strip() + + # A BARCODE IDENTIFIES ONE PACK, SO IT ONLY SURVIVES A KNOWN PACK SIZE. + # With no real size, stage 4 falls back to `default_size_variants` and + # invents 100g/250g/500g - and because every column is copied into each + # exploded variant, one real GTIN would be stamped onto three packs, two of + # which do not exist. That is a worse error than a blank barcode: it is + # wrong data that looks authoritative, and `nutrition_by_barcode` would + # happily resolve all three to the same product. + barcode = candidate.get("barcode") + notes: List[str] = [] + if barcode and not sizes: + notes.append( + "barcode dropped: no real pack size is known, and stage 4 will " + "invent several - a GTIN names one pack, not three" + ) + barcode = None + + shape = {"title": title, "category": category_hint, "size_variants": sizes, + "description": description} + evidence = _evidence_for(normalised, off_keys=off_keys, + registry_terms=registry_terms, catalog_keys=catalog_keys) + sources = list(dict.fromkeys(candidate.get("sources") or [candidate.get("source")])) + sources = [s for s in sources if s] + + return DiscoveredProduct( + brand=brand, + product_name=product_name, + title=title, + category=stated_category, + category_hint=category_hint, + description=description, + size_variants=sizes, + providers=list(candidate.get("providers") or _DEFAULT_PROVIDERS), + highlights=generate_product_highlights(shape, brand) or [], + nutrients=generate_nutrients_info(shape, brand) or [], + fssai_license=fssai, + barcode=barcode, + image_url=candidate.get("image_url"), + notes=notes, + sources=sources, + evidence=evidence, + confidence=_score(sources, evidence), + matches_existing=matches_existing, + ) + + +def discover_brand_products( + brand: str, + *, + max_products: int = BRAND_DISCOVERY_MAX_PRODUCTS, + max_sizes_per_product: int = BRAND_DISCOVERY_MAX_SIZES, + deadline_seconds: float = BRAND_DISCOVERY_DEADLINE_SECONDS, + use_openfacts: bool = BRAND_DISCOVERY_USE_OFF, + use_llm: bool = BRAND_DISCOVERY_USE_LLM, + require_evidence: bool = True, + refresh_corpus: bool = False, +) -> DiscoveryResult: + """Discover a brand's products. Reads the catalog; writes nothing. + + Open Food Facts runs first and unconditionally, so a slow or dead Ollama + degrades the result rather than emptying it. + """ + brand = (brand or "").strip() + if not brand: + raise ValueError("a brand name is required") + + parent = resolve_parent_brand(brand) + deadline = time.monotonic() + max(0.0, deadline_seconds) + warnings: List[str] = [] + + catalog_keys = _existing_title_index(brand) + registry_terms = _registry_terms(brand) + fssai = get_fssai_license(brand) + + off_candidates = _from_open_facts(brand, refresh=refresh_corpus) if use_openfacts else [] + if use_openfacts and not off_candidates: + warnings.append( + "Open Food Facts returned nothing for this brand. Every row below " + "rests on the language model alone - review them individually." + ) + off_keys = {} + for candidate in off_candidates: + key = _normalise_title(brand, candidate["title"]) + if key: + off_keys.setdefault(key, candidate) + + llm_candidates: List[Dict[str, Any]] = [] + if use_llm: + llm_candidates = _from_llm(brand, deadline=deadline, budget=max_products * 2) + if not llm_candidates: + warnings.append( + "The language model returned no products. This is usually " + "Ollama being unreachable or too slow; Open Food Facts results " + "below are unaffected." + ) + + # Merge. OFF wins every field it has, because it is the source that is real. + # + # TWO CANDIDATES MERGE ONLY IF THEIR PACK SIZES AGREE. A pack size is what + # distinguishes one retail SKU from another, and the title alone does not: + # OFF lists "Jim Jam" at 25g and "Jim jam" at 92g, which normalise to the + # same size-free key. Merging on the title collapsed them into a single + # product and silently discarded a real pack - the opposite of the coverage + # this feature exists to provide. A candidate carrying no size still merges, + # and adopts the size of the one it joins. + merged: List[Dict[str, Any]] = [] + keys: List[str] = [] + for candidate in off_candidates + llm_candidates: + title = candidate["title"] + key = _normalise_title(brand, title) + target = None + if key: + for index, existing in enumerate(keys): + if not _same_product(key, existing, title, merged[index]["title"]): + continue + held = merged[index].get("size") + offered = candidate.get("size") + if held and offered and held != offered: + continue + target = index + break + if target is None: + entry = dict(candidate) + entry["sources"] = [candidate.get("source")] + merged.append(entry) + keys.append(key) + continue + + entry = merged[target] + entry["sources"] = list(dict.fromkeys( + (entry.get("sources") or []) + [candidate.get("source")] + )) + for column in ("category", "description", "barcode", "size", "image_url"): + if not entry.get(column) and candidate.get(column): + entry[column] = candidate[column] + if not entry.get("sizes") and candidate.get("sizes"): + entry["sizes"] = candidate["sizes"] + if not entry.get("providers") and candidate.get("providers"): + entry["providers"] = candidate["providers"] + + # EVERY PACK OF ONE PRODUCT MUST BE NAMED THE SAME WAY. + # + # OFF carries a single product under many spellings, one per pack: "Marie + # Gold" 117g, "marie gold" 39g, "Marie GOLD" 300g, "Britannia Marie Gold" + # 1kg. Left alone, each pack keeps its own wording, so the first run stores + # six differently-named rows - and the second run, which folds them all onto + # one stored name, then writes five brand-new rows. That is the catalogue + # growth this module exists to prevent, arriving by a different door. + # + # Casing alone is harmless (`_slugify` lowercases, so the image_id is + # unaffected); it is the varying brand prefix and wording that split a + # product. The pick is deterministic and independent of corpus order, so two + # runs over the same data choose the same name even if OFF reorders its + # pages: fewest brand tokens first, so the redundant "Britannia " prefix + # loses inside `brand_britannia`, then shortest, then alphabetical. + tokens_of_brand = off_bulk.brand_tokens(brand) + + def _name_rank(name: str) -> Tuple[int, int, int, str]: + words = [w for w in re.split(r"[^a-z0-9]+", name.lower()) if w] + brand_words = sum(1 for w in words if w in tokens_of_brand) + # Prefer conventional casing over shouting. OFF holds both "Marie Gold" + # and "Marie GOLD"; the two are the same length, so without this the + # tie broke on ASCII order and "GOLD" won, putting an all-caps name on + # the shelf. Casing does not affect the image_id - `_slugify` + # lowercases - so this is purely what the customer reads. + shouty = sum(1 for w in name.split() if len(w) > 1 and w.isupper()) + return (brand_words, shouty, len(name), name) + + # Pin each pack's OWN size before any title is rewritten. A size read off + # the title has to be read off the title this candidate arrived with: once + # the group shares one name, "Britannia 5050 69.2g" would otherwise take its + # size from whichever spelling won, and every pack in the group would + # collapse onto one. + for entry in merged: + own_title = _canonicalise_title_size(entry["title"]) + in_title = _TITLE_SIZE_RE.search(own_title) + # The title's own size OVERWRITES the OFF quantity rather than merely + # filling a gap - the two disagree ("Britannia Toastea 200g" is listed + # with a quantity of 250g) and the title is the one that decides the + # product's name, so letting the quantity win would put a size in the + # name that contradicts the size column. + if in_title: + canonical = _canonical_size(in_title.group(0)) + if canonical: + entry["size"] = canonical + continue + if not entry.get("size"): + own = _resolve_sizes(entry, own_title, "", max_sizes_per_product) + if own: + entry["size"] = own[0] + + canonical_titles: Dict[str, str] = {} + for key, entry in zip(keys, merged): + if not key: + continue + current = canonical_titles.get(key) + title = entry["title"] + if current is None or _name_rank(title) < _name_rank(current): + canonical_titles[key] = title + for key, entry in zip(keys, merged): + if key and key in canonical_titles: + # Size-free, so the winning spelling cannot lend its pack size to + # the rest of the group. + chosen = _canonicalise_title_size(canonical_titles[key]) + entry["title"] = off_bulk.strip_sizes(chosen).strip() or chosen + + products: List[DiscoveredProduct] = [] + dropped = 0 + for candidate in merged: + product = _build_product( + brand, candidate, max_sizes=max_sizes_per_product, + catalog_keys=catalog_keys, off_keys=off_keys, + registry_terms=registry_terms, fssai=fssai, + ) + if product is None: + continue + if require_evidence and "off" not in product.sources and not product.evidence: + dropped += 1 + continue + products.append(product) + if len(products) >= max_products: + break + + products.sort(key=lambda p: (-p.confidence, p.product_name.lower())) + + counts = { + "discovered": len(products), + "from_openfacts": sum(1 for p in products if "off" in p.sources), + "from_llm_only": sum(1 for p in products if p.sources == ["llm"]), + "corroborated": sum(1 for p in products if len(p.sources) > 1), + "already_in_catalog": sum(1 for p in products if p.matches_existing), + "with_barcode": sum(1 for p in products if p.barcode), + "dropped_without_evidence": dropped, + } + + return DiscoveryResult( + brand=brand, + parent_brand=parent, + table=f"brand_{_sanitize_name(parent)}", + brand_active=active_brands.is_active_brand(brand), + filtering_enabled=active_brands.filtering_enabled(), + products=products, + counts=counts, + warnings=warnings, + ) + + +# --------------------------------------------------------------------------- +# Serialisation - the bridge itself +# --------------------------------------------------------------------------- +def rows_to_csv_bytes(products: Sequence[DiscoveredProduct]) -> bytes: + """Render discovered products as a spreadsheet the pipeline already reads. + + UTF-8 with `\\r\\n` line endings and `csv`'s own quoting, so a description + containing a comma or a newline survives. `read_products_dataframe` reads + everything `dtype=str`, so nothing is lost by going through text. + """ + buffer = io.StringIO(newline="") + writer = csv.DictWriter(buffer, fieldnames=list(CSV_HEADERS), + extrasaction="ignore", lineterminator="\r\n") + writer.writeheader() + for product in products: + writer.writerow({ + "brand": product.brand, + "product_name": product.product_name, + "title": product.title, + "category": product.category, + "description": product.description, + "price_range": "", + "size_variants": _safe_list_cell(product.size_variants), + "providers": _safe_list_cell(product.providers), + "highlights": _safe_list_cell(product.highlights), + "nutrients": _safe_list_cell(product.nutrients), + "fssai_license": product.fssai_license or "", + "product_sku": "", + "hsn_code": "", + "barcode": product.barcode or "", + "image_url": product.image_url or "", + }) + return buffer.getvalue().encode("utf-8") + + +def synthetic_filename(brand: str) -> str: + """A real, inspectable filename for the staged file. + + The bytes are written to the batch volume like any upload, so this names an + artefact that actually exists and can be re-run - not a placeholder. + """ + stamp = time.strftime("%Y%m%d-%H%M%S") + return f"discovered-{_sanitize_name(brand) or 'brand'}-{stamp}.csv" diff --git a/data/cache/off_brand_corpus/britannia.json b/data/cache/off_brand_corpus/britannia.json new file mode 100644 index 0000000..f9b6b8a --- /dev/null +++ b/data/cache/off_brand_corpus/britannia.json @@ -0,0 +1,2534 @@ +{ + "brand": "Britannia", + "country": "india", + "fetched_at": 1788778878.2242186, + "fetched_at_human": "2026-09-07 16:31:18", + "hits": [ + { + "code": "8901063031807", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Treat" + }, + { + "code": "8901063014206", + "brands": [ + "Britannia" + ], + "quantity": "400g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Vita Marie Gold", + "product_name_en": "Britannia Vita Marie Gold" + }, + { + "code": "8901063138469", + "brands": [ + "Britannia" + ], + "quantity": "150 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Thin Arrowroot", + "product_name_en": "Britannia Thin Arrowroot" + }, + { + "code": "8901063017627", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Brit 50-50 maska chaska 40g (20)", + "product_name_en": "Brit 50-50 maska chaska 40g (20)" + }, + { + "code": "8901063142442", + "brands": [ + "Britannia" + ], + "quantity": "75 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri Choice Seeds", + "product_name_en": "Nutri Choice Seeds" + }, + { + "code": "8901063019027", + "brands": [ + "Britannia" + ], + "quantity": "75g", + "countries_tags": [ + "en:india" + ], + "product_name": "Little Hearts Classic", + "product_name_en": "Little Hearts Classic" + }, + { + "code": "8901063012516", + "brands": [ + "Britannia" + ], + "quantity": "100g", + "countries_tags": [ + "en:india" + ], + "product_name": "Milk Bikis", + "product_name_en": "Milk Bikis" + }, + { + "code": "8901063033399", + "brands": [ + "Britannia" + ], + "quantity": "60 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Treat Vanilla Creme", + "product_name_en": "Treat Vanilla Creme" + }, + { + "code": "8901063093287", + "brands": [ + "Britannia" + ], + "quantity": "200 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Good Day Cashew Cookies", + "product_name_en": "Good Day Cashew Cookies" + }, + { + "code": "8901063016934", + "brands": [ + "BRITANNIA" + ], + "quantity": "69.2g", + "countries_tags": [ + "en:india" + ], + "product_name": "BRITANNIA 5050", + "product_name_en": "BRITANNIA 5050" + }, + { + "code": "8901063166257", + "brands": [ + "Britannia Tiger Kreemz Choco" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Chocolate Bar", + "product_name_en": "Chocolate Bar" + }, + { + "code": "8901063338623", + "brands": [ + "Britannia" + ], + "quantity": "400 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Multi Grain Bread", + "product_name_en": "Multi Grain Bread" + }, + { + "code": "0694189138724", + "brands": [ + "BRITANNIA" + ], + "quantity": "50g g", + "countries_tags": [ + "en:india" + ], + "product_name": "BOURBON", + "product_name_en": "BOURBON" + }, + { + "code": "8901063004122", + "brands": [ + "Britannia" + ], + "quantity": "100 g", + "countries_tags": [ + "en:india" + ], + "product_name": "GoodDay Choco Chip", + "product_name_en": "GoodDay Choco Chip" + }, + { + "code": "8901063163331", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Tiger glucose biscuit", + "product_name_en": "Tiger glucose biscuit" + }, + { + "code": "8901063092686", + "brands": [ + "Britannia" + ], + "quantity": "60g", + "countries_tags": [ + "en:india" + ], + "product_name": "Good Day Butter Cookies (25)", + "product_name_en": "Good Day Butter Cookies (25)" + }, + { + "code": "8901063093522", + "brands": [ + "Britannia" + ], + "quantity": "India", + "countries_tags": [ + "en:india" + ], + "product_name": "Good Day - Cashew Cookies", + "product_name_en": "Good Day - Cashew Cookies" + }, + { + "code": "8901063162242", + "brands": [ + "Britannia" + ], + "quantity": "1 kg", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Marie Gold", + "product_name_en": "Britannia Marie Gold" + }, + { + "code": "8901063016859", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "50 50 Sweet & Salty", + "product_name_en": "50 50 Sweet & Salty" + }, + { + "code": "8901063362871", + "brands": [ + "britannia" + ], + "quantity": "60g", + "countries_tags": [ + "en:india" + ], + "product_name": "cake nut & raisin", + "product_name_en": "cake nut & raisin" + }, + { + "code": "3948063139190", + "brands": [ + "britannia" + ], + "quantity": "60g", + "countries_tags": [ + "en:india" + ], + "product_name": "bourbon", + "product_name_en": "bourbon" + }, + { + "code": "8901063363793", + "brands": [ + "Britannia" + ], + "quantity": "50 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britania cake pineapple", + "product_name_en": "Britania cake pineapple" + }, + { + "code": "8901063146204", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Winkin cow milk shek", + "product_name_en": "Winkin cow milk shek" + }, + { + "code": "8901063094246", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia gd pista badam cookies 600g", + "product_name_en": "Britannia gd pista badam cookies 600g" + }, + { + "code": "8901063325760", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Toastea 200g", + "product_name_en": "Britannia Toastea 200g" + }, + { + "code": "8901063092631", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Good day (60g)", + "product_name_en": "Britannia Good day (60g)" + }, + { + "code": "8901063363281", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Roll Yo!" + }, + { + "code": "8901063017504", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "potazos", + "product_name_en": "potazos" + }, + { + "code": "8901063138162", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri choice", + "product_name_en": "Nutri choice" + }, + { + "code": "8901063092495", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "good day butter cookiew", + "product_name_en": "good day butter cookiew" + }, + { + "code": "8901063094185", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "good day", + "product_name_en": "good day" + }, + { + "code": "4901465098334", + "brands": [ + "britannia" + ], + "quantity": "39g", + "countries_tags": [ + "en:india" + ], + "product_name": "goodat", + "product_name_en": "goodat" + }, + { + "code": "8901063029163", + "brands": [ + "Britannia" + ], + "quantity": "100 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Brianna treat jimjam", + "product_name_en": "Brianna treat jimjam" + }, + { + "code": "8901063004092", + "brands": [ + "Britannia" + ], + "quantity": "1", + "countries_tags": [ + "en:india" + ], + "product_name": "Good day (chocochip)", + "product_name_en": "Good day (chocochip)" + }, + { + "code": "8901063401457", + "brands": [ + "Britannia" + ], + "quantity": "250gm", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Cheese", + "product_name_en": "Britannia Cheese" + }, + { + "code": "8901063401136", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Cheese Slices", + "product_name_en": "Cheese Slices" + }, + { + "code": "8901063004207", + "brands": [ + "Britannia" + ], + "quantity": "111g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Good Day Chocochip Cookies", + "product_name_en": "Britannia Good Day Chocochip Cookies" + }, + { + "code": "8901063032446", + "brands": [ + "Britannia" + ], + "quantity": "45g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Treat Croissant Cocoa Crème", + "product_name_en": "Britannia Treat Croissant Cocoa Crème" + }, + { + "code": "8901063028210", + "brands": [ + "Britannia Industries Ltd" + ], + "quantity": "59.4g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Nice", + "product_name_en": "Britannia Nice" + }, + { + "code": "8901063093355", + "brands": [ + "Britannia" + ], + "quantity": "600 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Good Day Cashew Cookies Medium Pack 600g", + "product_name_en": "Good Day Cashew Cookies Medium Pack 600g" + }, + { + "code": "8901063093393", + "brands": [ + "Britannia" + ], + "quantity": "200", + "countries_tags": [ + "en:india" + ], + "product_name": "Good day cashew cookies", + "product_name_en": "Good day cashew cookies" + }, + { + "code": "8901063162648", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Marie Gold", + "product_name_en": "Britannia Marie Gold" + }, + { + "code": "8901063162600", + "brands": [ + "Britannia" + ], + "quantity": "117g", + "countries_tags": [ + "en:india" + ], + "product_name": "Marie Gold", + "product_name_en": "Marie Gold" + }, + { + "code": "8901063029217", + "brands": [ + "Britannia" + ], + "quantity": "25g", + "countries_tags": [ + "en:india" + ], + "product_name": "Jim Jam", + "product_name_en": "Jim Jam" + }, + { + "code": "8901063341241", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Atta Kulcha Bread", + "product_name_en": "Atta Kulcha Bread" + }, + { + "code": "8901063026162", + "brands": [ + "Britannia" + ], + "quantity": "300 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri choice crakers", + "product_name_en": "Nutri choice crakers" + }, + { + "code": "8901063014299", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Vita maríe", + "product_name_en": "Vita maríe" + }, + { + "code": "8901063146297", + "brands": [ + "Britannia" + ], + "quantity": "180ml", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Winkin cow thick lassi", + "product_name_en": "Britannia Winkin cow thick lassi" + }, + { + "code": "8901063365384", + "brands": [ + "Britannia" + ], + "quantity": "35g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Cake Gobbles 35g", + "product_name_en": "Britannia Cake Gobbles 35g" + }, + { + "code": "8901063162327", + "brands": [ + "Britannia" + ], + "quantity": "78g", + "countries_tags": [ + "en:india" + ], + "product_name": "Br Marie gold 10", + "product_name_en": "Br Marie gold 10" + }, + { + "code": "3948063160293", + "brands": [ + "Britannia" + ], + "quantity": "100", + "countries_tags": [ + "en:india" + ], + "product_name": "Pure magic chocolush", + "product_name_en": "Pure magic chocolush" + }, + { + "code": "8901063338494", + "brands": [ + "Britannia" + ], + "quantity": "200g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Pav", + "product_name_en": "Britannia Pav" + }, + { + "code": "8901063016606", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "50-50 sweet and salt", + "product_name_en": "50-50 sweet and salt" + }, + { + "code": "8901063019140", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Little hearts", + "product_name_en": "Little hearts" + }, + { + "code": "8901063146334", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Winkin Cow" + }, + { + "code": "8901063032323", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Treat Cream Wafers" + }, + { + "code": "8901063093485", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Good Day" + }, + { + "code": "8901063028166", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britania Nice time", + "product_name_en": "Britania Nice time" + }, + { + "code": "8901063014152", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "vita marie gold", + "product_name_en": "vita marie gold" + }, + { + "code": "8901063325586", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "toastea bilk rusk", + "product_name_en": "toastea bilk rusk" + }, + { + "code": "8901063098657", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "good day", + "product_name_en": "good day" + }, + { + "code": "8901063142299", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Nutrichoice", + "product_name_en": "Nutrichoice" + }, + { + "code": "8901063012332", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "milk bikies", + "product_name_en": "milk bikies" + }, + { + "code": "8901063032484", + "brands": [ + "Britannia" + ], + "quantity": "47g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia treat Croissant Mixed Fruit" + }, + { + "code": "8901063023956", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Marie Gold 400g (1)" + }, + { + "code": "8901063160088", + "brands": [ + "Britannia" + ], + "quantity": "75g", + "countries_tags": [ + "en:india" + ], + "product_name": "Puremagic chocolush", + "product_name_en": "Puremagic chocolush" + }, + { + "code": "8901063146181", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Brtn wnkn cow vanila milk shake", + "product_name_en": "Brtn wnkn cow vanila milk shake" + }, + { + "code": "8901063325036", + "brands": [ + "Britannia" + ], + "quantity": "250g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia toastea", + "product_name_en": "Britannia toastea" + }, + { + "code": "8901063014312", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Vita Marie GOLD", + "product_name_en": "Vita Marie GOLD" + }, + { + "code": "8901063325746", + "brands": [ + "Britannia" + ], + "quantity": "6", + "countries_tags": [ + "en:india" + ], + "product_name": "toastea premium rusk", + "product_name_en": "toastea premium rusk" + }, + { + "code": "8901063362086", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Golddes", + "product_name_en": "Britannia Golddes" + }, + { + "code": "8901063162297", + "brands": [ + "Britannia Industries Ltd" + ], + "quantity": "39g", + "countries_tags": [ + "en:india" + ], + "product_name": "marie gold", + "product_name_en": "marie gold" + }, + { + "code": "8901063139206", + "brands": [ + "Britannia" + ], + "quantity": "120g", + "countries_tags": [ + "en:india" + ], + "product_name": "Bourbon", + "product_name_en": "Bourbon" + }, + { + "code": "8901063029323", + "brands": [ + "BRITANNIA" + ], + "quantity": "68.5g", + "countries_tags": [ + "en:india" + ], + "product_name": "Jim jam Britannia biscuit", + "product_name_en": "Jim jam Britannia biscuit" + }, + { + "code": "8901063146488", + "brands": [ + "Winkin' Cow", + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Winkin' cow Vanillicious Thick Shake", + "product_name_en": "Britannia Winkin' cow Vanillicious Thick Shake" + }, + { + "code": "8901063142619", + "brands": [ + "Britannia" + ], + "quantity": "120g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia nutri choice", + "product_name_en": "Britannia nutri choice" + }, + { + "code": "8901063012608", + "brands": [ + "Britannia" + ], + "quantity": "33.5g", + "countries_tags": [ + "en:india" + ], + "product_name": "Milk Bikis", + "product_name_en": "Milk Bikis" + }, + { + "code": "8901063016880", + "brands": [ + "Britannia" + ], + "quantity": "76", + "countries_tags": [ + "en:india" + ], + "product_name": "50 50", + "product_name_en": "50 50" + }, + { + "code": "8901063325784", + "brands": [ + "Britannia" + ], + "quantity": "275 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Toastea Rusk", + "product_name_en": "Toastea Rusk" + }, + { + "code": "8901063093089", + "brands": [ + "Britannia" + ], + "quantity": "120 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Good day Cashew", + "product_name_en": "Good day Cashew" + }, + { + "code": "8901063162549", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Marie gold", + "product_name_en": "Marie gold" + }, + { + "code": "8901063026278", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri Choice Sugar Free Crackers", + "product_name_en": "Nutri Choice Sugar Free Crackers" + }, + { + "code": "8901063016682", + "brands": [ + "Britannia" + ], + "quantity": "38g", + "countries_tags": [ + "en:india" + ], + "product_name": "5050", + "product_name_en": "5050" + }, + { + "code": "8901063092501", + "brands": [ + "Britannia" + ], + "quantity": "600 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Good Day Butter 600g", + "product_name_en": "Good Day Butter 600g" + }, + { + "code": "8901063325364", + "brands": [ + "Britannia" + ], + "quantity": "273 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Toastea Rusk", + "product_name_en": "Toastea Rusk" + }, + { + "code": "8901063343276", + "brands": [ + "Britannia" + ], + "quantity": "450g", + "countries_tags": [ + "en:india" + ], + "product_name": "Multi Grain Bread", + "product_name_en": "Multi Grain Bread" + }, + { + "code": "8901063033160", + "brands": [ + "Britannia" + ], + "quantity": "28", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia treat rs 10", + "product_name_en": "Britannia treat rs 10" + }, + { + "code": "8901063365032", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia cake gobbles 30g", + "product_name_en": "Britannia cake gobbles 30g" + }, + { + "code": "8901063325883", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Toastea (217g)", + "product_name_en": "Britannia Toastea (217g)" + }, + { + "code": "0901063139213", + "brands": [ + "Britannia", + "chocolate" + ], + "quantity": "150g", + "countries_tags": [ + "en:india" + ], + "product_name": "the Original bourbon", + "product_name_en": "the Original bourbon" + }, + { + "code": "8901063364462", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Time pass Funsticks Tomato Twist", + "product_name_en": "Britannia Time pass Funsticks Tomato Twist" + }, + { + "code": "8901063343528", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Vitarich Sandwich Bread", + "product_name_en": "Vitarich Sandwich Bread" + }, + { + "code": "3948063154025", + "brands": [ + "Britannia" + ], + "quantity": "150 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri choise essentials", + "product_name_en": "Nutri choise essentials" + }, + { + "code": "8901063017887", + "brands": [ + "Britannia", + "5050" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "5050 Golmaal Butter Garlic", + "product_name_en": "5050 Golmaal Butter Garlic" + }, + { + "code": "8901063402089", + "brands": [ + "Britannia" + ], + "quantity": "1kg", + "countries_tags": [ + "en:india" + ], + "product_name": "Dairy whitener", + "product_name_en": "Dairy whitener" + }, + { + "code": "8901063017399", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia 5050 maska chaska (300g)", + "product_name_en": "Britannia 5050 maska chaska (300g)" + }, + { + "code": "8901063365216", + "brands": [ + "britannia" + ], + "quantity": "29.5g", + "countries_tags": [ + "en:india" + ], + "product_name": "biscafe", + "product_name_en": "biscafe" + }, + { + "code": "8901063164321", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Tiger Kreemz Orange", + "product_name_en": "Tiger Kreemz Orange" + }, + { + "code": "8901063012349", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Milk Bikis", + "product_name_en": "Milk Bikis" + }, + { + "code": "8901063162143", + "brands": [ + "Britannia" + ], + "quantity": "89g", + "countries_tags": [ + "en:india" + ], + "product_name": "marie Gold", + "product_name_en": "marie Gold" + }, + { + "code": "8901063017719", + "brands": [ + "Britannia" + ], + "quantity": "110g", + "countries_tags": [ + "en:india" + ], + "product_name": "50 50 Golmaal", + "product_name_en": "50 50 Golmaal" + }, + { + "code": "8901063029286", + "brands": [ + "Britannia" + ], + "quantity": "460 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Jimjam 460g", + "product_name_en": "Jimjam 460g" + }, + { + "code": "8901063091023", + "brands": [ + "Britannia", + "Good Day" + ], + "quantity": "120 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Good Day Harmony", + "product_name_en": "Good Day Harmony" + }, + { + "code": "8901063012615", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Milk bikies biscuit", + "product_name_en": "Milk bikies biscuit" + }, + { + "code": "8901063342934", + "brands": [ + "Britannia" + ], + "quantity": "450 g", + "countries_tags": [ + "en:india" + ], + "product_name": "100% Whole Wheat Sandwich Bread", + "product_name_en": "100% Whole Wheat Sandwich Bread" + }, + { + "code": "8901063146464", + "brands": [ + "Britannia" + ], + "quantity": "1", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Winkin' Cow pet", + "product_name_en": "Britannia Winkin' Cow pet" + }, + { + "code": "8901063338685", + "brands": [ + "Britannia" + ], + "quantity": "450g", + "countries_tags": [ + "en:india" + ], + "product_name": "Brown bread", + "product_name_en": "Brown bread" + }, + { + "code": "8901063029309", + "brands": [ + "Britannia" + ], + "quantity": "70 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Jimjam Pops", + "product_name_en": "Jimjam Pops" + }, + { + "code": "8901063023901", + "brands": [ + "Britannia" + ], + "quantity": "1 Packet", + "countries_tags": [ + "en:india" + ], + "product_name": "Marie Gold", + "product_name_en": "Marie Gold" + }, + { + "code": "8901063035102", + "brands": [ + "Britannia" + ], + "quantity": "100 g", + "countries_tags": [ + "en:india" + ], + "product_name": "milk bikis", + "product_name_en": "milk bikis" + }, + { + "code": "8901063363779", + "brands": [ + "Britannia" + ], + "quantity": "50g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia cake fruity fun", + "product_name_en": "Britannia cake fruity fun" + }, + { + "code": "8901063325609", + "brands": [ + "Britannia" + ], + "quantity": "200g", + "countries_tags": [ + "en:india" + ], + "product_name": "Toastea premium rusk", + "product_name_en": "Toastea premium rusk" + }, + { + "code": "8901063004139", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Good day Chocolate Cookies 400g" + }, + { + "code": "8901063029316", + "brands": [ + "Britannia" + ], + "quantity": "350g", + "countries_tags": [ + "en:india" + ], + "product_name": "Jimjam pops", + "product_name_en": "Jimjam pops" + }, + { + "code": "8901063028227", + "brands": [ + "Britannia" + ], + "quantity": "4", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia nice time (8)", + "product_name_en": "Britannia nice time (8)" + }, + { + "code": "8901063017795", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "5050 Golmaal", + "product_name_en": "5050 Golmaal" + }, + { + "code": "8901063365018", + "brands": [ + "Britannia" + ], + "quantity": "100 gm", + "countries_tags": [ + "en:india" + ], + "product_name": "Biscafe Biscuit 🍪", + "product_name_en": "Biscafe Biscuit 🍪" + }, + { + "code": "8901063142015", + "brands": [ + "Britannia Digestive" + ], + "quantity": "100 g", + "countries_tags": [ + "en:france", + "en:india" + ], + "product_name": "Nutri Choice", + "product_name_en": "Nutri Choice" + }, + { + "code": "8901063410312", + "brands": [ + "BRITANNIA" + ], + "quantity": "120 g", + "countries_tags": [ + "en:india" + ], + "product_name": "The Laughing Cow", + "product_name_en": "The Laughing Cow" + }, + { + "code": "8901063032453", + "brands": [ + "Britannia" + ], + "quantity": "1 croissant, 45g", + "countries_tags": [ + "en:india" + ], + "product_name": "Treat Croissant Vanilla Créme", + "product_name_en": "Treat Croissant Vanilla Créme" + }, + { + "code": "8901063362857", + "brands": [ + "britannia" + ], + "quantity": "120g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia cake", + "product_name_en": "Britannia cake" + }, + { + "code": "94164466", + "brands": [ + "BRITANNIA" + ], + "quantity": "76 g", + "countries_tags": [ + "en:india" + ], + "product_name": "50 50", + "product_name_en": "50 50" + }, + { + "code": "8901063342910", + "brands": [ + "Britannia" + ], + "quantity": "400g", + "countries_tags": [ + "en:india" + ], + "product_name": "Brown bread", + "product_name_en": "Brown bread" + }, + { + "code": "8901063012530", + "brands": [ + "Britannia" + ], + "quantity": "66.8", + "countries_tags": [ + "en:india" + ], + "product_name": "Milk Bikis", + "product_name_en": "Milk Bikis" + }, + { + "code": "8901063401198", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Cheese Block", + "product_name_en": "Cheese Block" + }, + { + "code": "8901063142114", + "brands": [ + "Britannia" + ], + "quantity": "75 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Nutri Choice Oats", + "product_name_en": "Britannia Nutri Choice Oats" + }, + { + "code": "8901063017481", + "brands": [ + "Britannia" + ], + "quantity": "45.5g", + "countries_tags": [ + "en:india" + ], + "product_name": "50-50 Maska Chaska", + "product_name_en": "50-50 Maska Chaska" + }, + { + "code": "8901063012622", + "brands": [ + "Britannia" + ], + "quantity": "67 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Milk Bikis", + "product_name_en": "Milk Bikis" + }, + { + "code": "8901063012431", + "brands": [ + "Britannia" + ], + "quantity": "500g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia milk bikis (500g)", + "product_name_en": "Britannia milk bikis (500g)" + }, + { + "code": "8901063016842", + "brands": [ + "britannia", + "" + ], + "quantity": "76g", + "countries_tags": [ + "en:india" + ], + "product_name": "50 50", + "product_name_en": "50 50" + }, + { + "code": "8901063023949", + "brands": [ + "Britannia" + ], + "quantity": "300g", + "countries_tags": [ + "en:india" + ], + "product_name": "Marie GOLD", + "product_name_en": "Marie GOLD" + }, + { + "code": "8901063031937", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "britannia treat croissant vanilla creme roll", + "product_name_en": "britannia treat croissant vanilla creme roll" + }, + { + "code": "8901063142466", + "brands": [ + "Britannia" + ], + "quantity": "50 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri Choice Digestive", + "product_name_en": "Nutri Choice Digestive" + }, + { + "code": "8901063032828", + "brands": [ + "Britannia" + ], + "quantity": "60g", + "countries_tags": [ + "en:india" + ], + "product_name": "Treat", + "product_name_en": "Treat" + }, + { + "code": "8901063325074", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Toastea" + }, + { + "code": "8901063029262", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Jim jam92 g", + "product_name_en": "Jim jam92 g" + }, + { + "code": "8901063401082", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia cheese cube", + "product_name_en": "Britannia cheese cube" + }, + { + "code": "8901063325616", + "brands": [ + "Britannia" + ], + "quantity": "12", + "countries_tags": [ + "en:india" + ], + "product_name": "Bri toastea 700g", + "product_name_en": "Bri toastea 700g" + }, + { + "code": "8901063028197", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia mibe time (300g)", + "product_name_en": "Britannia mibe time (300g)" + }, + { + "code": "8901063138186", + "brands": [ + "Britannia" + ], + "quantity": "52g", + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri Choice", + "product_name_en": "Nutri Choice" + }, + { + "code": "8901063029170", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Treat Jimjam" + }, + { + "code": "8901063032309", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "treat orange", + "product_name_en": "treat orange" + }, + { + "code": "8901063016873", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "50-50", + "product_name_en": "50-50" + }, + { + "code": "8901063094154", + "brands": [ + "Britannia" + ], + "quantity": "200g", + "countries_tags": [ + "en:india" + ], + "product_name": "good day pista badam", + "product_name_en": "good day pista badam" + }, + { + "code": "8901063094147", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "good day", + "product_name_en": "good day" + }, + { + "code": "8901063091016", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "harmony", + "product_name_en": "harmony" + }, + { + "code": "8901063032330", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britania treat", + "product_name_en": "Britania treat" + }, + { + "code": "8901063032514", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Traet Croissant" + }, + { + "code": "8901063162525", + "brands": [ + "Britannia" + ], + "quantity": "4", + "countries_tags": [ + "en:india" + ], + "product_name": "Marie gold 5*117g (4)", + "product_name_en": "Marie gold 5*117g (4)" + }, + { + "code": "8901063342033", + "brands": [ + "Britannia" + ], + "quantity": "250g", + "countries_tags": [ + "en:india" + ], + "product_name": "britannia bread", + "product_name_en": "britannia bread" + }, + { + "code": "8901063021037", + "brands": [ + "Britannia" + ], + "quantity": "150 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Tiger Brita Biscuits", + "product_name_en": "Tiger Brita Biscuits" + }, + { + "code": "8901063029255", + "brands": [ + "Britannia" + ], + "quantity": "57 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Jimjam 57g (57)", + "product_name_en": "Jimjam 57g (57)" + }, + { + "code": "8901063001022", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia nutri choice", + "product_name_en": "Britannia nutri choice" + }, + { + "code": "8901063032927", + "brands": [ + "Britannia" + ], + "quantity": "14g", + "countries_tags": [ + "en:india" + ], + "product_name": "treat burst", + "product_name_en": "treat burst" + }, + { + "code": "8901063092402", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "good day blue 200g (3)", + "product_name_en": "good day blue 200g (3)" + }, + { + "code": "8901063012547", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Milk bikis", + "product_name_en": "Milk bikis" + }, + { + "code": "8901063017221", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Brittania Maska chaska", + "product_name_en": "Brittania Maska chaska" + }, + { + "code": "8901063363748", + "brands": [ + "Britannia" + ], + "quantity": "64g", + "countries_tags": [ + "en:india" + ], + "product_name": "CAKE 100% VEG Gobbles fruity fun", + "product_name_en": "CAKE 100% VEG Gobbles fruity fun" + }, + { + "code": "8901063016781", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia 5050(72g)", + "product_name_en": "Britannia 5050(72g)" + }, + { + "code": "8901063146457", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Winkin' cow thick shake", + "product_name_en": "Britannia Winkin' cow thick shake" + }, + { + "code": "8901063363977", + "brands": [ + "Britannia" + ], + "quantity": "110g", + "countries_tags": [ + "en:india" + ], + "product_name": "BRITANNIA Cake Gobbles Fruity Fun" + }, + { + "code": "8901063012660", + "brands": [ + "Britannia" + ], + "quantity": "500 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Milk Bikis", + "product_name_en": "Milk Bikis" + }, + { + "code": "8901063162426", + "brands": [ + "Britannia" + ], + "quantity": "73 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Marie gold", + "product_name_en": "Marie gold" + }, + { + "code": "8901063019171", + "brands": [ + "Britannia" + ], + "quantity": "13g", + "countries_tags": [ + "en:india" + ], + "product_name": "Little Hearts", + "product_name_en": "Little Hearts" + }, + { + "code": "8901063162303", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Marie gold", + "product_name_en": "Marie gold" + }, + { + "code": "8901063019089", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Little hearts" + }, + { + "code": "8901063139343", + "brands": [ + "Britannia" + ], + "quantity": "5 x 100 g", + "countries_tags": [ + "en:india" + ], + "product_name": "BRITANNIA BOURBON4+1", + "product_name_en": "BRITANNIA BOURBON4+1" + }, + { + "code": "8901063363786", + "brands": [ + "Britannia" + ], + "quantity": "50 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Gobbles Orange Bites cake", + "product_name_en": "Gobbles Orange Bites cake" + }, + { + "code": "8901063365087", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Brit 100% veg cake", + "product_name_en": "Brit 100% veg cake" + }, + { + "code": "8901063033306", + "brands": [ + "Britannia" + ], + "quantity": "51g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Treat Orange Creme", + "product_name_en": "Britannia Treat Orange Creme" + }, + { + "code": "8901063017559", + "brands": [ + "Britannia" + ], + "quantity": "29.0g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia 5050 Potazos 71.5g", + "product_name_en": "Britannia 5050 Potazos 71.5g" + }, + { + "code": "8901063155497", + "brands": [ + "Britannia" + ], + "quantity": "75G", + "countries_tags": [ + "en:india" + ], + "product_name": "TIGER KRUNCH COCONUT", + "product_name_en": "TIGER KRUNCH COCONUT" + }, + { + "code": "8901063004146", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "good day", + "product_name_en": "good day" + }, + { + "code": "8901063142367", + "brands": [ + "Britannia" + ], + "quantity": "75g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia NutriChoice oats chocolate & almond", + "product_name_en": "Britannia NutriChoice oats chocolate & almond" + }, + { + "code": "8901063155329", + "brands": [ + "Britannia" + ], + "quantity": "400g", + "countries_tags": [ + "en:india" + ], + "product_name": "Tiger Crunch biscuit 400g", + "product_name_en": "Tiger Crunch biscuit 400g" + }, + { + "code": "8901063142541", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia nutri choice 200g(sys)", + "product_name_en": "Britannia nutri choice 200g(sys)" + }, + { + "code": "8901063363809", + "brands": [ + "Britannia" + ], + "quantity": "50 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Gobbless Choco Chill Cake", + "product_name_en": "Gobbless Choco Chill Cake" + }, + { + "code": "8901063012554", + "brands": [ + "Britannia" + ], + "quantity": "115g", + "countries_tags": [ + "en:india" + ], + "product_name": "Milk BIKIS CLASSIC", + "product_name_en": "Milk BIKIS CLASSIC" + }, + { + "code": "8901063146280", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Winkin cow thick Lassi classic", + "product_name_en": "Winkin cow thick Lassi classic" + }, + { + "code": "8901063163287", + "brands": [ + "Britannia" + ], + "quantity": "49.7 grams", + "countries_tags": [ + "en:india" + ], + "product_name": "Tiger glucose biscuits", + "product_name_en": "Tiger glucose biscuits" + }, + { + "code": "9908463001088", + "brands": [ + "BRITANNIA" + ], + "quantity": "200g", + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri Choice", + "product_name_en": "Nutri Choice" + }, + { + "code": "8901063342354", + "brands": [ + "Britannia" + ], + "quantity": "450g", + "countries_tags": [ + "en:india" + ], + "product_name": "sandwich white bread", + "product_name_en": "sandwich white bread" + }, + { + "code": "8901063028821", + "brands": [ + "Britannia" + ], + "quantity": "143g", + "countries_tags": [ + "en:india" + ], + "product_name": "Nice time", + "product_name_en": "Nice time" + }, + { + "code": "8901063342026", + "brands": [ + "Britannia" + ], + "quantity": "400g", + "countries_tags": [ + "en:india" + ], + "product_name": "Vitarich Sandwich Premium White Bread", + "product_name_en": "Vitarich Sandwich Premium White Bread" + }, + { + "code": "8901063136434", + "brands": [ + "Britannia" + ], + "quantity": "100g", + "countries_tags": [ + "en:india" + ], + "product_name": "Milk Bikis", + "product_name_en": "Milk Bikis" + }, + { + "code": "8901063325357", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Toastea Rusk", + "product_name_en": "Toastea Rusk" + }, + { + "code": "8901063026209", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Sugar Free Cracker Nature spice Jeera & Ajwain", + "product_name_en": "Sugar Free Cracker Nature spice Jeera & Ajwain" + }, + { + "code": "8901063155459", + "brands": [ + "Britannia" + ], + "quantity": "63 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Tiger Krunch ChocoChips", + "product_name_en": "Tiger Krunch ChocoChips" + }, + { + "code": "8901063017702", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "50 50 Gol Maal", + "product_name_en": "50 50 Gol Maal" + }, + { + "code": "901063139213", + "brands": [ + "Britannia", + "chocolate" + ], + "quantity": "150g", + "countries_tags": [ + "en:india" + ], + "product_name": "the Original bourbon", + "product_name_en": "the Original bourbon" + }, + { + "code": "8901063018297", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "time pass", + "product_name_en": "time pass" + }, + { + "code": "8901063142244", + "brands": [ + "Britannia" + ], + "quantity": "100g", + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri choice zero" + }, + { + "code": "8901063092433", + "brands": [ + "britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Good day cookies", + "product_name_en": "Good day cookies" + }, + { + "code": "8901063154025", + "brands": [ + "Britannia Nutrichoice" + ], + "quantity": "1", + "countries_tags": [ + "en:india" + ], + "product_name": "Ragi cookies" + }, + { + "code": "8901063016613", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "SWEET & SALTY", + "product_name_en": "SWEET & SALTY" + }, + { + "code": "8901063017252", + "brands": [ + "Britannia" + ], + "quantity": "50g", + "countries_tags": [ + "en:india" + ], + "product_name": "50-50", + "product_name_en": "50-50" + }, + { + "code": "8901063139329", + "brands": [ + "britannia", + "" + ], + "quantity": "50g", + "countries_tags": [ + "en:india" + ], + "product_name": "bourbon", + "product_name_en": "bourbon" + }, + { + "code": "8901063017566", + "brands": [ + "Britannia" + ], + "quantity": "1", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia patazos", + "product_name_en": "Britannia patazos" + }, + { + "code": "8901063142459", + "brands": [ + "Britannia" + ], + "quantity": "100g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia nutri choice herbs", + "product_name_en": "Britannia nutri choice herbs" + }, + { + "code": "8901063092464", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Good day cashew 200g (8)", + "product_name_en": "Good day cashew 200g (8)" + }, + { + "code": "8901063365162", + "brands": [ + "Britannia" + ], + "quantity": "27 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Cake Roll Yo!", + "product_name_en": "Britannia Cake Roll Yo!" + }, + { + "code": "8901063032804", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia treat (50g)", + "product_name_en": "Britannia treat (50g)" + }, + { + "code": "8901063018341", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia 50 50 time pass (63g)", + "product_name_en": "Britannia 50 50 time pass (63g)" + }, + { + "code": "8901063142305", + "brands": [ + "Britannia" + ], + "quantity": "1kg", + "countries_tags": [ + "en:india" + ], + "product_name": "Nutri Choice" + }, + { + "code": "4901265161429", + "brands": [ + "Britannia" + ], + "quantity": "300g", + "countries_tags": [ + "en:india" + ], + "product_name": "Pure Magic Chocolush", + "product_name_en": "Pure Magic Chocolush" + }, + { + "code": "8901063363595", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Muffills Choco Vanilla Flavoured", + "product_name_en": "Muffills Choco Vanilla Flavoured" + }, + { + "code": "8901063092617", + "brands": [ + "Britannia" + ], + "quantity": "68g", + "countries_tags": [ + "en:india" + ], + "product_name": "Good Day Butter Cookies", + "product_name_en": "Good Day Butter Cookies" + }, + { + "code": "8901063139190", + "brands": [ + "Britannia" + ], + "quantity": "60g", + "countries_tags": [ + "en:india" + ], + "product_name": "Bourbon", + "product_name_en": "Bourbon" + }, + { + "code": "8901063026018", + "brands": [ + "Britannia", + "Nutrichoice" + ], + "quantity": "100g", + "countries_tags": [ + "en:india" + ], + "product_name": "Cracker", + "product_name_en": "Cracker" + }, + { + "code": "8901063405011", + "brands": [ + "Britannia" + ], + "quantity": "1L", + "countries_tags": [ + "en:india", + "en:united-states" + ], + "product_name": "Britannia Cow's ghee", + "product_name_en": "Britannia Cow's ghee" + }, + { + "code": "8901063142336", + "brands": [ + "Britannia" + ], + "quantity": "75g", + "countries_tags": [ + "en:india" + ], + "product_name": "nutri choice", + "product_name_en": "nutri choice" + }, + { + "code": "8901063092747", + "brands": [ + "Britannia" + ], + "countries_tags": [ + "en:india" + ], + "product_name": "Good day", + "product_name_en": "Good day" + }, + { + "code": "8901063032729", + "brands": [ + "Britannia" + ], + "quantity": "55 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Treat coconut wafers", + "product_name_en": "Treat coconut wafers" + }, + { + "code": "8901063365100", + "brands": [ + "Britannia" + ], + "quantity": "110g", + "countries_tags": [ + "en:india" + ], + "product_name": "Gobble marble cake", + "product_name_en": "Gobble marble cake" + }, + { + "code": "8901063139374", + "brands": [ + "Britannia" + ], + "quantity": "60g", + "countries_tags": [ + "en:india" + ], + "product_name": "BOURBON", + "product_name_en": "BOURBON" + }, + { + "code": "8901063363359", + "brands": [ + "Britannia" + ], + "quantity": "120 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia Fudge It - Chocolate Brownie", + "product_name_en": "Britannia Fudge It - Chocolate Brownie" + }, + { + "code": "8901063029279", + "brands": [ + "Britannia" + ], + "quantity": "138 g", + "countries_tags": [ + "en:india" + ], + "product_name": "Jimjam" + }, + { + "code": "8901063139336", + "brands": [ + "Britannia" + ], + "quantity": "100g", + "countries_tags": [ + "en:india" + ], + "product_name": "Britannia bourbon 100g", + "product_name_en": "Britannia bourbon 100g" + } + ] +} \ No newline at end of file diff --git a/data/sku_sequences/britannia.json b/data/sku_sequences/britannia.json index e6b9989..b6e1bf0 100644 --- a/data/sku_sequences/britannia.json +++ b/data/sku_sequences/britannia.json @@ -1,21 +1,254 @@ { "_sku_sequences": { - "BRITAN-GOO-100": 10, - "BRITAN-GOO-200": 12, + "BRITAN-GOO-100": 97, + "BRITAN-GOO-200": 153, "BRITAN-GOO-500": 9, "BRITAN-GOO-250": 1, "BRITAN-MAR-100": 2, "BRITAN-MAR-250": 2, "BRITAN-MAR-500": 1, - "BRITAN-MIL-200": 2, - "BRITAN-MIL-500": 1, - "BRITAN-MIL-1": 1, - "BRITAN-GOO-375": 1, + "BRITAN-MIL-200": 58, + "BRITAN-MIL-500": 42, + "BRITAN-MIL-1": 28, + "BRITAN-GOO-375": 55, "BRITAN-MAR-200": 1, "BRITAN-MAR-375": 1, - "BRITAN-MIL-100": 1, - "BRITAN-MIL-375": 1, + "BRITAN-MIL-100": 58, + "BRITAN-MIL-375": 28, "BRITAN-MIL-150": 1, - "BRITAN-50-200": 1 + "BRITAN-50-200": 1, + "BRITAN-100-450": 29, + "BRITAN-50-76": 29, + "BRITAN-50-100": 29, + "BRITAN-50-250": 27, + "BRITAN-50-500": 27, + "BRITAN-50-110": 29, + "BRITAN-50-1": 58, + "BRITAN-50-455": 29, + "BRITAN-505-40": 1, + "BRITAN-505-80": 1, + "BRITAN-505-150": 1, + "BRITAN-505-200": 29, + "BRITAN-505-500": 53, + "BRITAN-505-1": 27, + "BRITAN-ATT-200": 29, + "BRITAN-ATT-400": 27, + "BRITAN-ATT-600": 27, + "BRITAN-BIS-295": 29, + "BRITAN-BIS-100": 29, + "BRITAN-BOU-50": 29, + "BRITAN-BR-78": 29, + "BRITAN-BRI-700": 29, + "BRITAN-BRI-100": 142, + "BRITAN-BRI-250": 105, + "BRITAN-BRI-400": 27, + "BRITAN-BRI-40": 32, + "BRITAN-BRI-50": 29, + "BRITAN-BRI-80": 3, + "BRITAN-BRI-150": 3, + "BRITAN-50-63": 29, + "BRITAN-505-692": 29, + "BRITAN-505-300": 29, + "BRITAN-505-29": 17, + "BRITAN-BOU-40": 1, + "BRITAN-BOU-80": 1, + "BRITAN-BOU-150": 30, + "BRITAN-BRE-250": 29, + "BRITAN-CAK-120": 29, + "BRITAN-CAK-50": 29, + "BRITAN-CAK-35": 29, + "BRITAN-CAK-110": 29, + "BRITAN-CAK-27": 29, + "BRITAN-CHE-250": 29, + "BRITAN-CHE-200": 87, + "BRITAN-CHE-500": 81, + "BRITAN-CHE-1": 81, + "BRITAN-COW-1": 29, + "BRITAN-FUD-120": 29, + "BRITAN-GD-600": 29, + "BRITAN-GOL-40": 1, + "BRITAN-GOL-80": 1, + "BRITAN-GOL-150": 1, + "BRITAN-GOO-60": 58, + "BRITAN-GOO-111": 29, + "BRITAN-GOO-400": 29, + "BRITAN-MAR-1": 29, + "BRITAN-MIB-300": 29, + "BRITAN-NIC-594": 29, + "BRITAN-NIC-143": 29, + "BRITAN-NUT-100": 129, + "BRITAN-NUT-75": 101, + "BRITAN-PAT-40": 1, + "BRITAN-PAT-80": 1, + "BRITAN-PAT-150": 1, + "BRITAN-PAV-200": 29, + "BRITAN-THI-150": 29, + "BRITAN-TIM-40": 2, + "BRITAN-TIM-80": 2, + "BRITAN-TIM-150": 2, + "BRITAN-TOA-250": 17, + "BRITAN-TRA-40": 1, + "BRITAN-TRA-80": 1, + "BRITAN-TRA-150": 1, + "BRITAN-TRE-60": 58, + "BRITAN-TRE-45": 29, + "BRITAN-TRE-47": 29, + "BRITAN-TRE-40": 5, + "BRITAN-TRE-80": 5, + "BRITAN-TRE-150": 5, + "BRITAN-TRE-51": 29, + "BRITAN-VIT-400": 85, + "BRITAN-WIN-180": 29, + "BRITAN-WIN-40": 5, + "BRITAN-WIN-80": 5, + "BRITAN-WIN-150": 5, + "BRITAN-BRO-450": 29, + "BRITAN-BRT-200": 29, + "BRITAN-BRT-500": 27, + "BRITAN-BRT-1": 27, + "BRITAN-CAK-64": 29, + "BRITAN-CAK-60": 29, + "BRITAN-CHO-20": 29, + "BRITAN-CHO-55": 27, + "BRITAN-CHO-150": 27, + "BRITAN-CRA-100": 29, + "BRITAN-DAI-1": 29, + "BRITAN-GOB-110": 29, + "BRITAN-GOB-50": 58, + "BRITAN-GOO-600": 57, + "BRITAN-GOO-120": 58, + "BRITAN-GOO-39": 29, + "BRITAN-HAR-40": 1, + "BRITAN-HAR-80": 1, + "BRITAN-HAR-150": 1, + "BRITAN-JIM-25": 29, + "BRITAN-JIM-685": 29, + "BRITAN-JIM-40": 1, + "BRITAN-JIM-80": 1, + "BRITAN-JIM-150": 1, + "BRITAN-JIM-460": 29, + "BRITAN-JIM-70": 29, + "BRITAN-LIT-13": 29, + "BRITAN-LIT-75": 29, + "BRITAN-MAR-117": 43, + "BRITAN-MIL-115": 29, + "BRITAN-MUF-100": 29, + "BRITAN-MUF-250": 27, + "BRITAN-MUF-400": 27, + "BRITAN-MUL-400": 29, + "BRITAN-NUT-120": 29, + "BRITAN-NUT-300": 56, + "BRITAN-NUT-50": 29, + "BRITAN-NUT-200": 41, + "BRITAN-NUT-150": 30, + "BRITAN-NUT-40": 1, + "BRITAN-NUT-80": 1, + "BRITAN-POT-40": 1, + "BRITAN-POT-80": 1, + "BRITAN-POT-150": 1, + "BRITAN-PUR-300": 29, + "BRITAN-PUR-75": 29, + "BRITAN-RAG-100": 29, + "BRITAN-RAG-200": 27, + "BRITAN-RAG-375": 27, + "BRITAN-ROL-40": 1, + "BRITAN-ROL-80": 1, + "BRITAN-ROL-150": 1, + "BRITAN-SAN-450": 29, + "BRITAN-SUG-100": 29, + "BRITAN-SUG-200": 27, + "BRITAN-SUG-300": 27, + "BRITAN-SWE-1": 29, + "BRITAN-LAU-120": 29, + "BRITAN-TIG-150": 30, + "BRITAN-TIG-400": 29, + "BRITAN-TIG-100": 57, + "BRITAN-TIG-200": 27, + "BRITAN-TIG-375": 27, + "BRITAN-TIG-497": 29, + "BRITAN-TIG-40": 1, + "BRITAN-TIG-80": 1, + "BRITAN-TIG-63": 29, + "BRITAN-TIG-75": 29, + "BRITAN-TOA-180": 28, + "BRITAN-TOA-300": 27, + "BRITAN-TOA-600": 27, + "BRITAN-TOA-200": 41, + "BRITAN-TOA-275": 29, + "BRITAN-TRE-14": 29, + "BRITAN-TRE-55": 29, + "BRITAN-TRE-75": 29, + "BRITAN-TRE-100": 167, + "BRITAN-TRE-125": 27, + "BRITAN-VIT-40": 1, + "BRITAN-VIT-80": 1, + "BRITAN-VIT-150": 1, + "BRITAN-VIT-200": 29, + "BRITAN-VIT-600": 27, + "BRITAN-WIN-200": 29, + "BRITAN-WIN-500": 157, + "BRITAN-WIN-1": 27, + "BRITAN-505-100": 28, + "BRITAN-505-250": 26, + "BRITAN-BRI-500": 78, + "BRITAN-BOU-100": 42, + "BRITAN-BOU-250": 26, + "BRITAN-BOU-500": 26, + "BRITAN-GOL-100": 28, + "BRITAN-GOL-250": 26, + "BRITAN-GOL-500": 26, + "BRITAN-PAT-100": 28, + "BRITAN-PAT-250": 26, + "BRITAN-PAT-500": 26, + "BRITAN-TIM-100": 56, + "BRITAN-TIM-250": 52, + "BRITAN-TIM-500": 52, + "BRITAN-TRA-100": 28, + "BRITAN-TRA-250": 26, + "BRITAN-TRA-500": 26, + "BRITAN-TRE-250": 130, + "BRITAN-TRE-500": 130, + "BRITAN-WIN-100": 140, + "BRITAN-WIN-250": 130, + "BRITAN-HAR-100": 28, + "BRITAN-HAR-250": 26, + "BRITAN-HAR-500": 26, + "BRITAN-JIM-100": 1, + "BRITAN-JIM-250": 1, + "BRITAN-JIM-500": 1, + "BRITAN-NUT-250": 26, + "BRITAN-NUT-500": 26, + "BRITAN-POT-100": 28, + "BRITAN-POT-250": 26, + "BRITAN-POT-500": 26, + "BRITAN-ROL-100": 28, + "BRITAN-ROL-250": 26, + "BRITAN-ROL-500": 26, + "BRITAN-TIG-250": 26, + "BRITAN-TIG-500": 26, + "BRITAN-VIT-100": 28, + "BRITAN-VIT-250": 26, + "BRITAN-VIT-500": 26, + "BRITAN-JIM-92": 27, + "BRITAN-505-715": 12, + "BRITAN-50-50": 14, + "BRITAN-505-38": 14, + "BRITAN-BOU-60": 14, + "BRITAN-BOU-120": 14, + "BRITAN-BRO-400": 14, + "BRITAN-GOO-68": 14, + "BRITAN-JIM-57": 14, + "BRITAN-JIM-138": 14, + "BRITAN-JIM-350": 14, + "BRITAN-MAR-39": 14, + "BRITAN-MAR-89": 14, + "BRITAN-MAR-300": 13, + "BRITAN-MAR-73": 14, + "BRITAN-MIL-335": 14, + "BRITAN-MIL-67": 14, + "BRITAN-MUL-450": 14, + "BRITAN-NUT-52": 13, + "BRITAN-NUT-1": 14, + "BRITAN-TOA-273": 14 } } \ No newline at end of file diff --git a/tests/test_brand_discovery.py b/tests/test_brand_discovery.py new file mode 100644 index 0000000..f524b57 --- /dev/null +++ b/tests/test_brand_discovery.py @@ -0,0 +1,551 @@ +"""Tests for brand discovery - the brand-name -> 11-stage-pipeline bridge. + +Follows the pattern in test_store_catalog_pipeline.py: monkeypatch the network +and storage boundary *on the module object* (brand_discovery imports those names +directly), and assert on the rows that come out rather than on a status code. + +Nothing here reaches Open Food Facts, Ollama or a database. The two source +functions are stubbed and every pipeline run uses `use_llm=False, +fetch_images=False`. + +Several of these tests pin behaviour that was WRONG in the first working +version of this module and was found by round-tripping the real Britannia +corpus. They are regression tests with a known failure, not speculative ones - +each names the defect it prevents. +""" +from __future__ import annotations + +import pytest + +from app.api.routers.user_products import map_spreadsheet_columns, row_to_request +from app.core import store_catalog_pipeline as pipeline +from app.services import brand_discovery as bd + + +# --------------------------------------------------------------------------- +# Fixtures +# --------------------------------------------------------------------------- +@pytest.fixture(autouse=True) +def _isolate_sku_counter(tmp_path, monkeypatch): + """Keep the SKU sequence counter out of the repo - see the same fixture in + test_store_catalog_pipeline.py.""" + from app.services import sku_service + monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences") + + +@pytest.fixture(autouse=True) +def _no_network(monkeypatch): + """Neither source may reach the outside world by default. + + A test that wants products stubs one of these explicitly. Without this an + accidental real call would hit Open Food Facts from the suite. + """ + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: []) + monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: []) + monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: []) + + +@pytest.fixture +def store(monkeypatch): + """A fake brand table. Returns the dict of image_id -> stored row.""" + table: dict = {} + + def fake_upsert(brand, rows, cleanup=False): + assert cleanup is False, "cleanup=True would delete the brand's existing catalog" + for row in rows: + table[row["image_id"]] = row + return len(rows) + + monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert) + monkeypatch.setattr(pipeline, "get_products_by_brand", + lambda brand, **kw: list(table.values())) + monkeypatch.setattr(pipeline, "embed_texts", + lambda texts: [[0.0] * 384 for _ in texts]) + return table + + +def _off(title, *, code=None, quantity=None): + """One Open Food Facts corpus hit, in the shape `_from_open_facts` returns.""" + return {"title": title, "barcode": code, "size": bd._canonical_size(quantity), + "source": "off"} + + +def _run(products, filename="discovered.csv"): + """Discovered products -> CSV -> the real 11 stages.""" + return pipeline.run_pipeline(filename, bd.rows_to_csv_bytes(products), + use_llm=False, fetch_images=False) + + +# --------------------------------------------------------------------------- +# The contract the whole CSV bridge rests on +# --------------------------------------------------------------------------- +def test_the_csv_headers_all_map_onto_catalog_fields(): + """Every emitted header must be understood by the real column mapper. + + This is THE contract: discovery writes a spreadsheet and the pipeline reads + it with `map_spreadsheet_columns`. A header that does not resolve is dropped + in silence, so the column simply never arrives and the run still reports + success. Asserting it here means a rename on either side fails loudly. + """ + mapping = map_spreadsheet_columns(bd.CSV_HEADERS) + + assert mapping.unrecognised == [] + assert mapping.ignored == [] + for header in bd.CSV_HEADERS: + assert header in mapping.columns, f"{header!r} did not resolve to a field" + + +def test_the_emitted_csv_parses_with_the_real_spreadsheet_reader(monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", + lambda brand, refresh=False: [_off("Marie Gold", quantity="250 g")]) + + result = bd.discover_brand_products("Britannia", use_llm=False) + df, mapping = pipeline.parse_spreadsheet("d.csv", bd.rows_to_csv_bytes(result.products)) + + assert len(df) == 1 + assert mapping.unrecognised == [] + request = row_to_request(df.to_dict(orient="records")[0], mapping) + assert request.brand == "Britannia" + assert request.product_name == "Marie Gold" + assert request.size_variants == ["250g"] + + +# --------------------------------------------------------------------------- +# Regression: list cells were shredded by the separator +# --------------------------------------------------------------------------- +def test_a_list_item_containing_a_separator_survives_the_round_trip(): + """`_string_list` splits on [,;|], and the highlight generator emits commas. + + "Available in 3 sizes: 100g, 250g, 500g" and "Baked, Not Fried" became five + fragments instead of two highlights. The damage is invisible in the database + - the column is populated, just wrong - so it needs a test. + """ + cell = bd._safe_list_cell([ + "Available in 3 sizes: 100g, 250g, 500g", + "Baked, Not Fried", + "Pipe | separated | too", + ]) + mapping = map_spreadsheet_columns(("highlights",)) + from app.api.routers.user_products import _string_list + + recovered = _string_list({"highlights": cell}, mapping, "highlights") + + assert len(recovered) == 3 + assert recovered[0] == "Available in 3 sizes: 100g / 250g / 500g" + assert recovered[1] == "Baked / Not Fried" + + +# --------------------------------------------------------------------------- +# Regression: pack sizes doubled into the image_id +# --------------------------------------------------------------------------- +@pytest.mark.parametrize("raw, expected", [ + ("200 g", "200g"), + ("200g", "200g"), + ("1.5 kg", "1.5kg"), + ("75 g", "75g"), +]) +def test_a_spaced_pack_size_is_canonicalised(raw, expected): + """OFF writes both "100g" and "100 g" for the same brand. + + `build_image_id` slugifies "200 g" to "200_g", which is not a substring of a + name containing "200g", so the size is appended anyway and the same product + lands on two permanent rows depending on which spelling was discovered. + """ + assert bd._canonical_size(raw) == expected + + +@pytest.mark.parametrize("junk", ["India", "12", "200", "", None, "6"]) +def test_a_quantity_that_is_not_a_pack_size_is_rejected(junk): + """The Britannia corpus carries "India" and bare counts in `quantity`.""" + assert bd._canonical_size(junk) is None + + +def test_a_size_glued_to_a_letter_in_the_title_is_separated(): + """OFF holds "Jim Jam92 g", where there is no word boundary before the 92. + + `off_bulk._SIZE_RE` anchors on one and so cannot see it, which left the + unnormalised size in the product name and produced the image_id + `britannia_jim_jam92_g_40g`. + """ + assert bd._canonicalise_title_size("Jim Jam92 g") == "Jim Jam 92g" + assert bd._canonicalise_title_size("Britannia Good Day 200 g") == "Britannia Good Day 200g" + + +def test_no_stored_image_id_carries_the_pack_size_twice(store): + result = _run([ + bd.DiscoveredProduct(brand="Britannia", product_name="Jim Jam", + title="Jim Jam", category="", category_hint="Biscuits & Cookies", + description="", size_variants=["92g"]), + ]) + assert result.inserted == 1 + assert list(store) == ["britannia_jim_jam_92g"] + + +# --------------------------------------------------------------------------- +# Regression: the pack size belongs in ONE place +# --------------------------------------------------------------------------- +def test_a_size_in_the_title_is_not_applied_twice(monkeypatch): + """OFF holds "Britannia Toastea 200g" with a `quantity` of "250g". + + Taking the quantity and keeping the title produced the product name + "Britannia Toastea 200g 250g" - two sizes, one name, a pack that does not + exist. The title's own size wins and is removed from the name. + """ + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"), + ]) + + product = bd.discover_brand_products("Britannia", use_llm=False).products[0] + + assert product.product_name == "Britannia Toastea" + assert product.size_variants == ["200g"] + + +def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch): + """OFF titles carry carton counts: "Good Day Butter Cookies (25)". + + That is how many units ship in a box, not part of what the product is + called, and leaving it in put the carton count into the image_id. + """ + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Good Day Butter Cookies (25)", quantity="60 g"), + ]) + + product = bd.discover_brand_products("Britannia", use_llm=False).products[0] + + assert product.product_name == "Good Day Butter Cookies" + assert product.size_variants == ["60g"] + + +# --------------------------------------------------------------------------- +# Regression: one GTIN, one pack +# --------------------------------------------------------------------------- +def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch): + """A GTIN identifies one pack, and stage 4 invents three when it has none. + + Every column is copied into each exploded variant, so a surviving barcode + would be stamped onto two packs that do not exist - wrong data that looks + authoritative. + """ + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("50 50 Gol Maal", code="8901063017702"), + ]) + + product = bd.discover_brand_products("Britannia", use_llm=False).products[0] + + assert product.size_variants == [] + assert product.barcode is None + assert any("barcode dropped" in note for note in product.notes) + + +def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Milk Bikis", code="8901063012516", quantity="100 g"), + ]) + + product = bd.discover_brand_products("Britannia", use_llm=False).products[0] + + assert product.barcode == "8901063012516" + assert product.size_variants == ["100g"] + + +def test_no_gtin_is_written_to_more_than_one_row(store, monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Milk Bikis", code="8901063012516", quantity="100 g"), + _off("50 50 Gol Maal", code="8901063017702"), + ]) + + result = bd.discover_brand_products("Britannia", use_llm=False) + _run(result.products) + + barcodes = [row["barcode"] for row in store.values() if row["barcode"]] + assert len(barcodes) == len(set(barcodes)) + + +# --------------------------------------------------------------------------- +# Regression: the generated description poisoned the category and the tax code +# --------------------------------------------------------------------------- +def test_no_category_is_written_when_no_source_stated_one(monkeypatch): + """Stage 3 owns category detection, and treats a supplied value as final. + + It sets `_category_deterministic` and then rewrites the title against the + category, so a wrong guess here does not merely mislabel a row - it renames + the product. + """ + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("50 50 Gol Maal", quantity="100 g"), + ]) + + product = bd.discover_brand_products("Britannia", use_llm=False).products[0] + + assert product.category == "" + assert product.category_hint # still reported, for the preview + + +def test_a_stated_category_is_carried_through(monkeypatch): + monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [ + {"title": "Good Day Cashew", "category": "Biscuits & Cookies", + "description": "Cashew cookies", "sizes": ["75g"], "providers": [], + "source": "llm"}, + ]) + + result = bd.discover_brand_products("Britannia", require_evidence=False) + + assert result.products[0].category == "Biscuits & Cookies" + + +def test_no_boilerplate_description_is_generated(monkeypatch): + """`generate_detailed_description` contains the word "taste". + + `detect_category_from_text` matches keywords fuzzily and "taste" is one edit + from the Oral Care keyword "paste", so every weakly-titled product was + classified Oral Care and stage 9 stamped it with HSN 3306 - the tax code for + dentifrices. A blank description is honest, and `_to_storage_row` already + substitutes " from ." + """ + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("50 50 Gol Maal", quantity="100 g"), + ]) + + product = bd.discover_brand_products("Britannia", use_llm=False).products[0] + + assert product.description == "" + assert "taste" not in (product.description or "").lower() + + +def test_a_weakly_titled_product_is_not_classified_as_oral_care(store, monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("50 50 Gol Maal", quantity="100 g"), + ]) + + result = bd.discover_brand_products("Britannia", use_llm=False) + _run(result.products) + + stored = next(iter(store.values())) + assert stored["category"] != "Oral Care" + assert stored["hsn_code"] != "3306" + + +# --------------------------------------------------------------------------- +# Regression: catalogue growth by re-run +# --------------------------------------------------------------------------- +def test_a_known_product_is_reemitted_under_its_stored_name(monkeypatch): + """A brand prefix appearing or disappearing is the realistic drift. + + The stored name has no brand prefix and the discovered one does; both + normalise identically once `brand_tokens` are stripped, so the row folds and + the image_id stays put. + """ + monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [ + {"product_name": "Good Day Cashew Cookies 75g"}, + ]) + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Britannia Good Day Cashew Cookies", quantity="75 g"), + ]) + + product = bd.discover_brand_products("Britannia", use_llm=False).products[0] + + assert product.matches_existing == "Good Day Cashew Cookies" + assert product.product_name == "Good Day Cashew Cookies" + + +def test_a_shorter_title_does_not_fold_onto_a_longer_stored_one(monkeypatch): + """The documented limit of the 0.85 threshold, pinned so it is a decision + rather than a surprise. + + "Good Day Cashew" scores 0.75 against "Good Day Cashew Cookies" and stays a + separate product. Relaxing the threshold far enough to fold it would also + fold "Dairy Milk Silk" onto "Dairy Milk Silk Minis", which is a different + product - see `_same_product`. + """ + monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [ + {"product_name": "Good Day Cashew Cookies 75g"}, + ]) + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Good Day Cashew", quantity="75 g"), + ]) + + product = bd.discover_brand_products("Britannia", use_llm=False).products[0] + + assert product.matches_existing is None + + +def test_two_real_packs_sharing_a_name_stay_on_their_own_rows(monkeypatch): + """OFF holds "Jim Jam" at 25g and "Jim jam" at 92g. + + Both normalise to the same size-free key. Re-emitting the first stored + DISPLAY name gave the 92g pack the 25g name, which `_to_storage_row` then + extended to "Jim Jam 25g 92g" under a brand-new image_id. The index hands + back the base name so the size can be re-applied per pack. + """ + monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [ + {"product_name": "Jim Jam 25g"}, + {"product_name": "Jim jam 92g"}, + ]) + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Jim Jam", quantity="25 g"), + _off("Jim jam", quantity="92 g"), + ]) + + products = bd.discover_brand_products("Britannia", use_llm=False).products + names = {p.product_name for p in products} + + assert names == {"Jim Jam"} + assert sorted(s for p in products for s in p.size_variants) == ["25g", "92g"] + assert not any("25g 92g" in p.product_name for p in products) + + +def test_re_running_discovery_and_ingest_writes_nothing_new(store, monkeypatch): + """The whole feature's safety property, end to end. + + Deterministic image_id + fill-only-blanks + `cleanup=False` make a re-run a + no-op, but only if discovery re-emits the same names. This exercises the + real pipeline twice with the catalog fed back in between. + """ + corpus = [ + _off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"), + _off("Good Day Butter Cookies (25)", quantity="60 g"), + _off("Jim Jam92 g", code="8901063019027"), + _off("50 50 Gol Maal", code="8901063017702"), + ] + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: corpus) + monkeypatch.setattr(bd, "get_products_by_brand", + lambda brand, **kw: list(store.values())) + + first = _run(bd.discover_brand_products("Britannia", use_llm=False).products) + after_first = dict(store) + assert first.inserted > 0 + + second = _run(bd.discover_brand_products("Britannia", use_llm=False).products) + + assert second.inserted == 0 + assert second.backfilled == 0 + assert second.skipped_existing == len(after_first) + assert set(store) == set(after_first) + + +# --------------------------------------------------------------------------- +# Sources, evidence and scoring +# --------------------------------------------------------------------------- +def test_open_food_facts_results_survive_an_unreachable_ollama(monkeypatch): + """OFF runs first and unconditionally, so a dead LLM degrades the result + rather than emptying it - the common case on a CPU-only box.""" + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Marie Gold", code="8901063014206", quantity="250 g"), + ]) + monkeypatch.setattr(bd.ollama_service, "_generate", lambda *a, **k: "") + monkeypatch.setattr(bd.ollama_service, "get_categories_for_brand", lambda brand: []) + + result = bd.discover_brand_products("Britannia", use_llm=True) + + assert len(result.products) == 1 + assert result.products[0].sources == ["off"] + assert any("language model returned no products" in w for w in result.warnings) + + +def test_a_product_found_by_both_sources_is_one_row_and_scores_highest(monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Marie Gold", code="8901063014206", quantity="250 g"), + ]) + monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [ + {"title": "Marie Gold", "category": "Biscuits & Cookies", + "description": "Tea-time biscuit", "sizes": ["250g"], + "providers": ["Amazon"], "source": "llm"}, + ]) + + result = bd.discover_brand_products("Britannia") + + assert len(result.products) == 1 + product = result.products[0] + assert sorted(product.sources) == ["llm", "off"] + assert product.confidence == 1.0 + assert result.counts["corroborated"] == 1 + + +def test_an_uncorroborated_llm_product_is_dropped_when_evidence_is_required(monkeypatch): + monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [ + {"title": "Britannia Quantum Wafer", "category": None, "description": None, + "sizes": [], "providers": [], "source": "llm"}, + ]) + + result = bd.discover_brand_products("Britannia", require_evidence=True) + + assert result.products == [] + assert result.counts["dropped_without_evidence"] == 1 + + +def test_an_uncorroborated_llm_product_is_kept_but_unticked_when_evidence_is_optional(monkeypatch): + monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [ + {"title": "Britannia Quantum Wafer", "category": None, "description": None, + "sizes": [], "providers": [], "source": "llm"}, + ]) + + result = bd.discover_brand_products("Britannia", require_evidence=False) + + assert len(result.products) == 1 + product = result.products[0] + assert product.evidence is None + assert product.confidence < 0.5 + assert product.as_preview()["selected"] is False + + +def test_a_sub_brand_match_counts_as_registry_evidence(monkeypatch): + """"Good Day" is a registered Britannia sub-brand, so an LLM row naming it + is grounded without needing Open Food Facts.""" + monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [ + {"title": "Good Day Chocochip", "category": None, "description": None, + "sizes": ["100g"], "providers": [], "source": "llm"}, + ]) + + result = bd.discover_brand_products("Britannia", require_evidence=True) + + assert len(result.products) == 1 + assert result.products[0].evidence == "registry" + + +def test_discovery_stops_at_max_products(monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off(f"Product {n}", quantity="100 g") for n in range(50) + ]) + + result = bd.discover_brand_products("Britannia", max_products=10, use_llm=False) + + assert len(result.products) == 10 + + +def test_the_brand_table_and_active_state_are_reported(monkeypatch): + monkeypatch.setattr(bd.active_brands, "is_active_brand", lambda brand: False) + monkeypatch.setattr(bd.active_brands, "filtering_enabled", lambda: True) + + result = bd.discover_brand_products("Britannia", use_llm=False) + + assert result.table == "brand_britannia" + assert result.parent_brand == "britannia" + assert result.brand_active is False + assert result.filtering_enabled is True + + +def test_an_empty_brand_name_is_refused(): + with pytest.raises(ValueError): + bd.discover_brand_products(" ") + + +# --------------------------------------------------------------------------- +# Field completeness through the real stages +# --------------------------------------------------------------------------- +def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch): + monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [ + _off("Milk Bikis", code="8901063012516", quantity="100 g"), + ]) + + result = bd.discover_brand_products("Britannia", use_llm=False) + _run(result.products) + + row = next(iter(store.values())) + for column in ("product_name", "title", "description", "category", "image_id", + "price_range", "size_variants", "providers", "highlights", + "nutrients", "fssai_license", "product_sku", "sku_source", + "search_query"): + assert row.get(column), f"{column} was left empty" + assert row["barcode"] == "8901063012516" + assert row["fssai_license"] == "10012022000103" diff --git a/tests/test_brand_discovery_api.py b/tests/test_brand_discovery_api.py new file mode 100644 index 0000000..a56a9f8 --- /dev/null +++ b/tests/test_brand_discovery_api.py @@ -0,0 +1,361 @@ +"""HTTP tests for /api/admin/brand-discovery/*. + +Reuses the fixture set from test_batch_catalog_ingest.py verbatim - in +particular `batch_root` and `no_background_worker`, whose absence once wrote +batch manifests into the repository's own data directory. Read that fixture's +docstring before removing either from a test here. + +Discovery itself is stubbed at the module boundary; the point of this file is +the routes, the ACTIVE_BRANDS gate and the staging handoff, not the merge logic +(which test_brand_discovery.py covers). +""" +from __future__ import annotations + +import pytest + +from app.api import batch_common +from app.core import batch_ingest +from app.core import store_catalog_pipeline as pipeline +from app.services import brand_discovery as bd + +PREVIEW = "/api/admin/brand-discovery/preview" +INGEST = "/api/admin/brand-discovery/ingest" + + +# --------------------------------------------------------------------------- +# Fixtures - see test_batch_catalog_ingest.py for the rationale behind each +# --------------------------------------------------------------------------- +@pytest.fixture(autouse=True) +def _isolate_sku_counter(tmp_path, monkeypatch): + from app.services import sku_service + monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences") + + +@pytest.fixture(autouse=True) +def batch_root(tmp_path, monkeypatch): + root = tmp_path / "batch_uploads" + monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", root) + return root + + +@pytest.fixture(autouse=True) +def no_background_worker(monkeypatch): + """Stub the worker. Its absence once wrote manifests into the working tree - + the long docstring in test_batch_catalog_ingest.py explains how.""" + from app.core import batch_worker + + submitted: list = [] + monkeypatch.setattr(batch_worker, "submit", submitted.append) + return submitted + + +@pytest.fixture(autouse=True) +def _clean_job_store(): + """The job store is a module singleton and outlives a test - see the same + fixture in test_uploads_api.py.""" + from app.api.batch_job_store import batch_job_store + batch_job_store._batches.clear() + batch_job_store._cancelled.clear() + yield + batch_job_store._batches.clear() + batch_job_store._cancelled.clear() + + +@pytest.fixture(autouse=True) +def _active_brands_allow_everything(monkeypatch): + """Filtering off by default, so only the tests that are about the + ACTIVE_BRANDS gate have to think about it.""" + from app.services import active_brands + monkeypatch.setattr(active_brands, "filtering_enabled", lambda: False) + monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: True) + + +@pytest.fixture +def discovered(monkeypatch): + """A fixed two-product discovery result.""" + def fake_discover(brand, **kwargs): + return bd.DiscoveryResult( + brand=brand, + parent_brand="britannia", + table="brand_britannia", + brand_active=True, + filtering_enabled=False, + products=[ + bd.DiscoveredProduct( + brand=brand, product_name="Marie Gold", title="Marie Gold", + category="", category_hint="Biscuits & Cookies", description="", + size_variants=["250g"], barcode="8901063014206", + sources=["off"], evidence="openfacts", confidence=0.9, + ), + bd.DiscoveredProduct( + brand=brand, product_name="Milk Bikis", title="Milk Bikis", + category="", category_hint="Biscuits & Cookies", description="", + size_variants=["100g"], barcode="8901063012516", + sources=["off"], evidence="openfacts", confidence=0.9, + ), + ], + counts={"discovered": 2}, + ) + + from app.api.routers import brand_discovery as router_module + monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", + fake_discover) + return fake_discover + + +def _payload(**overrides): + body = { + "brand": "Britannia", + "products": [ + {"product_name": "Marie Gold", "size_variants": ["250g"], + "barcode": "8901063014206", "providers": ["Amazon"], + "highlights": ["Tea-time favourite"], "nutrients": ["Iron - Blood health"]}, + ], + } + body.update(overrides) + return body + + +# --------------------------------------------------------------------------- +# Auth +# --------------------------------------------------------------------------- +def test_preview_requires_admin(client): + assert client.post(PREVIEW, json={"brand": "Britannia"}).status_code in (401, 403) + + +def test_ingest_requires_admin(client): + assert client.post(INGEST, json=_payload()).status_code in (401, 403) + + +def test_a_plain_user_cannot_discover(client, user_headers): + resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=user_headers) + assert resp.status_code == 403 + + +# --------------------------------------------------------------------------- +# Preview writes nothing +# --------------------------------------------------------------------------- +def test_preview_returns_products_and_the_stage_list(client, admin_headers, discovered): + resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers) + + assert resp.status_code == 200 + body = resp.json() + assert len(body["products"]) == 2 + assert body["table"] == "brand_britannia" + assert body["stages"] == list(pipeline.STAGE_NAMES) + assert len(body["stages"]) == 11 + assert body["products"][0]["selected"] is True + + +def test_preview_stages_nothing(client, admin_headers, discovered, batch_root, + no_background_worker): + """The whole reason preview is a separate route.""" + client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers) + + assert no_background_worker == [] + assert not batch_root.exists() or list(batch_root.iterdir()) == [] + + +def test_preview_reports_a_bad_brand_name_as_client_error(client, admin_headers): + resp = client.post(PREVIEW, json={"brand": " "}, headers=admin_headers) + assert resp.status_code == 400 + + +def test_preview_surfaces_a_discovery_failure_rather_than_500(client, admin_headers, + monkeypatch): + from app.api.routers import brand_discovery as router_module + + def boom(brand, **kwargs): + raise RuntimeError("Open Food Facts is unreachable") + + monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", boom) + + resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers) + + assert resp.status_code == 502 + assert "unreachable" in resp.json()["detail"] + + +# --------------------------------------------------------------------------- +# The ACTIVE_BRANDS gate +# --------------------------------------------------------------------------- +def test_preview_warns_when_the_brand_is_not_active(client, admin_headers, discovered, + monkeypatch): + from app.services import active_brands + monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True) + monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False) + monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"]) + + from app.api.routers import brand_discovery as router_module + + def inactive(brand, **kwargs): + result = discovered(brand, **kwargs) + result.brand_active = False + result.filtering_enabled = True + return result + + monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", inactive) + + body = client.post(PREVIEW, json={"brand": "Britannia"}, + headers=admin_headers).json() + + assert body["brand_active"] is False + assert any("ACTIVE_BRANDS" in w for w in body["warnings"]) + + +def test_ingest_is_refused_for_an_inactive_brand(client, admin_headers, monkeypatch): + """A green run over a catalog no endpoint can read is not an acceptable + outcome to hand back silently.""" + from app.services import active_brands + monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True) + monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False) + monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul", "Cadbury"]) + + resp = client.post(INGEST, json=_payload(), headers=admin_headers) + + assert resp.status_code == 409 + detail = resp.json()["detail"] + assert "ACTIVE_BRANDS=Amul,Cadbury,Britannia" in detail + assert "restart" in detail.lower() + + +def test_ingest_proceeds_for_an_inactive_brand_when_acknowledged(client, admin_headers, + monkeypatch): + from app.services import active_brands + monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True) + monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False) + monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"]) + + resp = client.post(INGEST, json=_payload(acknowledge_inactive=True), + headers=admin_headers) + + assert resp.status_code == 202 + + +# --------------------------------------------------------------------------- +# Ingest +# --------------------------------------------------------------------------- +def test_ingest_stages_one_batch_and_returns_a_pollable_id(client, admin_headers, + no_background_worker): + resp = client.post(INGEST, json=_payload(), headers=admin_headers) + + assert resp.status_code == 202 + body = resp.json() + assert body["batch_id"] + assert body["files_total"] == 1 + assert len(body["stage_names"]) == 11 + assert no_background_worker == [body["batch_id"]] + + +def test_the_staged_file_is_a_real_csv_the_pipeline_can_read(client, admin_headers, + batch_root): + """Not a placeholder: the bytes on the batch volume are the exact input that + produced the rows, and they must parse with the same reader an upload uses.""" + batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"] + + manifest = batch_ingest.read_manifest(batch_id) + entry = manifest.files[0] + assert entry.filename.startswith("discovered-britannia-") + assert entry.filename.endswith(".csv") + + contents = (batch_ingest.batch_dir(batch_id) / entry.stored_name).read_bytes() + df, mapping = pipeline.parse_spreadsheet(entry.filename, contents) + + assert len(df) == 1 + assert mapping.unrecognised == [] + assert "product_name" in mapping.columns + + +def test_the_batch_records_where_it_came_from(client, admin_headers): + batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"] + + manifest = batch_ingest.read_manifest(batch_id) + + assert manifest.submitted_by == "brand-discovery: Britannia" + + +def test_ingest_runs_the_eleven_stages_end_to_end(client, admin_headers, store, + no_background_worker, monkeypatch): + """Drive the real worker function over the staged file. + + `use_llm` and `fetch_images` are turned OFF for this one, matching every + other pipeline test in the suite: with them on, `run_batch` really calls + Ollama once per row and really searches the open web for images. That is the + correct production default and a terrible test - slow, and it fails when the + machine is offline. + """ + body = _payload(use_llm=False, fetch_images=False) + batch_id = client.post(INGEST, json=body, headers=admin_headers).json()["batch_id"] + + batch_ingest.run_batch(batch_id) + + assert len(store) == 1 + row = next(iter(store.values())) + assert row["image_id"] == "britannia_marie_gold_250g" + assert row["barcode"] == "8901063014206" + assert row["product_sku"] + assert row["highlights"] == ["Tea-time favourite"] + + +def test_an_empty_selection_is_refused(client, admin_headers): + resp = client.post(INGEST, json=_payload(products=[]), headers=admin_headers) + + assert resp.status_code == 400 + assert "nothing to ingest" in resp.json()["detail"] + + +def test_a_blank_brand_is_refused(client, admin_headers): + resp = client.post(INGEST, json=_payload(brand=" "), headers=admin_headers) + assert resp.status_code == 400 + + +def test_a_full_queue_is_reported_as_429(client, admin_headers, monkeypatch): + import queue + + def full(batch_id): + raise queue.Full() + + from app.core import batch_worker + monkeypatch.setattr(batch_worker, "submit", full) + + resp = client.post(INGEST, json=_payload(), headers=admin_headers) + + assert resp.status_code == 429 + assert "Resume" in resp.json()["detail"] + + +def test_ingest_defaults_turn_on_images_and_the_per_row_llm(client, admin_headers, + monkeypatch): + """The two enrichment stages this feature runs with. Both are per-run + arguments already, so neither needs a settings change.""" + seen = {} + real = batch_common.stage_and_queue + + def spy(valid, invalid, *, use_llm, fetch_images, submitted_by=None): + seen["use_llm"] = use_llm + seen["fetch_images"] = fetch_images + return real(valid, invalid, use_llm=use_llm, fetch_images=fetch_images, + submitted_by=submitted_by) + + from app.api.routers import brand_discovery as router_module + monkeypatch.setattr(router_module.batch_common, "stage_and_queue", spy) + + client.post(INGEST, json=_payload(), headers=admin_headers) + + assert seen == {"use_llm": True, "fetch_images": True} + + +@pytest.fixture +def store(monkeypatch): + table: dict = {} + + def fake_upsert(brand, rows, cleanup=False): + assert cleanup is False, "cleanup=True would delete the brand's existing catalog" + for row in rows: + table[row["image_id"]] = dict(row) + return len(rows) + + monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert) + monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values())) + monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts]) + return table