Brand Ingestion

This commit is contained in:
sriram
2026-09-07 17:52:45 +05:30
parent 75dd3eb3ce
commit 2749bee1a3
11 changed files with 5069 additions and 10 deletions

View File

@@ -267,6 +267,34 @@ BRAND_SYNC_INTERVAL_SECONDS=300
#ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever,Own Products
# ---------------------------------------------------------------------------
# Brand discovery (a brand NAME -> the 11-stage pipeline)
# ---------------------------------------------------------------------------
# POST /api/admin/brand-discovery/preview finds a brand's products, and
# /ingest stages the ones an admin approved as an ordinary catalog batch.
# Every value below has a working default; none of these need to be set.
#
# Open Food Facts is the primary source and the language model is the
# supplement. OFF returns real products with real barcodes and pack sizes;
# the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) will invent plausible ones, and
# nothing downstream can tell a well-formed fiction from a real product. Turn
# BRAND_DISCOVERY_USE_OFF off and the result rests on the model alone.
#
# NOTE: discovering a brand that is not in ACTIVE_BRANDS writes a complete
# catalog that no endpoint can read. The ingest route refuses with a 409 and
# names the line to add here; it is a config change plus a restart, never a
# re-ingest.
#BRAND_DISCOVERY_USE_OFF=true
#BRAND_DISCOVERY_USE_LLM=true
#BRAND_DISCOVERY_MAX_PRODUCTS=200
# Pack sizes kept per product when only the language model offers any. Stage 6
# runs an image search per exploded row, so this multiplies the slowest stage.
#BRAND_DISCOVERY_MAX_SIZES=3
# Wall-clock ceiling on the language-model half, checked between prompts. Open
# Food Facts runs first and is never subject to it.
#BRAND_DISCOVERY_DEADLINE_SECONDS=300
USE_S3=true
S3_ACCESS_KEY=your-do-spaces-key
S3_SECRET_KEY=your-do-spaces-secret

View File

@@ -0,0 +1,253 @@
"""Admin routes for brand discovery: a brand NAME into the 11-stage pipeline.
Two steps on purpose, mirroring `batch_catalog.py`'s preview/ingest split.
`/preview` discovers and returns; it writes nothing, anywhere. `/ingest` takes
the rows the admin kept, renders them as a CSV, and hands the bytes to the same
`batch_common.stage_and_queue` an uploaded spreadsheet goes through - so the
batch manifest, the stage timeline, Resume, Cancel and the nutrition
auto-enrichment that follows a batch all work here without a line of new code.
The gap between the two steps is the point. Discovery's language-model half can
invent a product that nothing downstream is able to catch: a well-formed
fiction resolves a category, gets a price band and an internal SKU, and clears
`product_validator`'s "verified" threshold comfortably. `product_validator` was
built to reject MALFORMED rows, not false ones. A person looking at the list is
the check, so the list is shown before anything is written.
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List, Optional
from fastapi import APIRouter, Depends, HTTPException, status
from pydantic import BaseModel, Field
from starlette.concurrency import run_in_threadpool
from app.api import batch_common
from app.api.deps import require_admin
from app.core import store_catalog_pipeline as pipeline
from app.infrastructure.settings import (
BATCH_MAX_FILES,
BATCH_MAX_TOTAL_BYTES,
BATCH_MAX_TOTAL_ROWS,
BRAND_DISCOVERY_DEADLINE_SECONDS,
BRAND_DISCOVERY_MAX_PRODUCTS,
)
from app.services import active_brands, brand_discovery
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/admin/brand-discovery", tags=["admin", "catalog"])
# The same per-file ceilings the admin batch routes apply. Discovery emits one
# CSV row per product and pack-size explosion happens later, inside stage 4, so
# 200 products is 200 rows here - three orders of magnitude inside the limit.
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
MAX_UPLOAD_ROWS = 2000
def _limits() -> batch_common.UploadLimits:
return batch_common.UploadLimits(
max_files=BATCH_MAX_FILES,
max_file_bytes=MAX_UPLOAD_BYTES,
max_file_rows=MAX_UPLOAD_ROWS,
max_total_bytes=BATCH_MAX_TOTAL_BYTES,
max_total_rows=BATCH_MAX_TOTAL_ROWS,
)
# ---------------------------------------------------------------------------
# Request bodies
# ---------------------------------------------------------------------------
class DiscoveryPreviewRequest(BaseModel):
brand: str
max_products: int = Field(default=BRAND_DISCOVERY_MAX_PRODUCTS, ge=1, le=2000)
use_openfacts: bool = True
use_llm: bool = True
# Ungrounded language-model rows are dropped rather than shown by default.
# Turning this off is how an admin sees them - they arrive unticked.
require_evidence: bool = True
# Shorter than the service default: somebody is watching a spinner.
deadline_seconds: float = Field(default=90.0, ge=0.0,
le=BRAND_DISCOVERY_DEADLINE_SECONDS)
refresh_corpus: bool = False
class DiscoveredProductIn(BaseModel):
"""One row the admin kept. Mirrors `DiscoveredProduct`'s written fields.
Sent back rather than re-discovered so that what is ingested is exactly what
was reviewed - a second discovery pass could legitimately return something
different, and then the approval would have been of a different list.
"""
product_name: str
title: Optional[str] = None
category: Optional[str] = None
description: Optional[str] = None
size_variants: List[str] = Field(default_factory=list)
providers: List[str] = Field(default_factory=list)
highlights: List[str] = Field(default_factory=list)
nutrients: List[str] = Field(default_factory=list)
fssai_license: Optional[str] = None
barcode: Optional[str] = None
image_url: Optional[str] = None
class DiscoveryIngestRequest(BaseModel):
brand: str
products: List[DiscoveredProductIn]
# Stage 2 only calls Ollama for a row whose description is blank, and
# discovery leaves most of them blank on purpose (see brand_discovery's note
# on the boilerplate generator). On an unreachable Ollama each such row
# costs up to OLLAMA_TIMEOUT_SECONDS, so this is worth being able to turn
# off for a large brand.
use_llm: bool = True
fetch_images: bool = True
# Ingesting a brand outside ACTIVE_BRANDS writes rows nothing can read.
# Refused unless the caller says they mean it - see the 409 below.
acknowledge_inactive: bool = False
# ---------------------------------------------------------------------------
# Routes
# ---------------------------------------------------------------------------
@router.post("/preview", dependencies=[Depends(require_admin)])
async def preview_brand_discovery(payload: DiscoveryPreviewRequest) -> Dict[str, Any]:
"""Discover a brand's products and return them. Writes nothing.
Run on a worker thread: discovery does blocking HTTP to Open Food Facts and,
when the language model is enabled, a series of blocking Ollama calls. On
the event loop that would stall every other request for the duration.
"""
try:
result = await run_in_threadpool(
brand_discovery.discover_brand_products,
payload.brand,
max_products=payload.max_products,
deadline_seconds=payload.deadline_seconds,
use_openfacts=payload.use_openfacts,
use_llm=payload.use_llm,
require_evidence=payload.require_evidence,
refresh_corpus=payload.refresh_corpus,
)
except ValueError as exc:
raise HTTPException(status_code=400, detail=str(exc)) from exc
except Exception as exc: # noqa: BLE001 - report the failure, do not 500
logger.exception("Brand discovery failed for %r", payload.brand)
raise HTTPException(
status_code=502,
detail=f"Discovery failed for {payload.brand!r}: {exc}",
) from exc
body = result.as_dict()
# Served rather than duplicated in the frontend, exactly as BatchOut does,
# so the two cannot drift when a stage is added.
body["stages"] = list(pipeline.STAGE_NAMES)
if result.filtering_enabled and not result.brand_active:
body["warnings"] = list(body.get("warnings") or []) + [_inactive_message(result)]
return body
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
dependencies=[Depends(require_admin)])
async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_common.BatchOut:
"""Stage the reviewed products as a catalog batch and return an id to poll."""
brand = (payload.brand or "").strip()
if not brand:
raise HTTPException(status_code=400, detail="A brand name is required.")
if not payload.products:
raise HTTPException(
status_code=400,
detail="No products were selected, so there is nothing to ingest.",
)
# A green run over an unreadable catalog is the failure this project refuses
# to ship. The rows WOULD be written and fully enriched - nutrition
# auto-enrichment runs with include_inactive=True - but /api/brands, search,
# suggest and the category listing all filter the brand out, so the result
# looks like nothing happened.
if active_brands.filtering_enabled() and not active_brands.is_active_brand(brand):
if not payload.acknowledge_inactive:
raise HTTPException(status_code=409, detail=_inactive_detail(brand))
products = [
brand_discovery.DiscoveredProduct(
brand=brand,
product_name=item.product_name,
title=item.title or item.product_name,
category=item.category or "",
category_hint="",
description=item.description or "",
size_variants=list(item.size_variants),
providers=list(item.providers),
highlights=list(item.highlights),
nutrients=list(item.nutrients),
fssai_license=item.fssai_license,
barcode=item.barcode,
image_url=item.image_url,
)
for item in payload.products
]
filename = brand_discovery.synthetic_filename(brand)
contents = brand_discovery.rows_to_csv_bytes(products)
# ONE FILE, NEVER CHUNKED. stage 11 groups by brand per file and reads the
# existing catalog per file, and its intra-file de-duplication
# ({image_id: row}) is per file too - so the same image_id split across two
# chunks would not be caught. A single file makes that de-duplication total.
valid, invalid, _rows = batch_common.parse_all([(filename, contents)], _limits())
if not valid:
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
raise HTTPException(
status_code=400,
detail=f"The discovered products could not be staged. {detail}",
)
manifest, started = batch_common.stage_and_queue(
valid, invalid,
use_llm=payload.use_llm,
fetch_images=payload.fetch_images,
submitted_by=f"brand-discovery: {brand}",
)
if not started:
raise HTTPException(
status_code=429,
detail=(
"Too many batches are already queued. This one has been saved - "
"press Resume on it once the current batch finishes."
),
)
return batch_common.to_out(manifest)
# ---------------------------------------------------------------------------
# The ACTIVE_BRANDS message, in one place
# ---------------------------------------------------------------------------
def _env_line(brand: str) -> str:
names = active_brands.active_display_names()
return "ACTIVE_BRANDS=" + ",".join(list(names) + [brand])
def _inactive_message(result: brand_discovery.DiscoveryResult) -> str:
return (
f"{result.brand} is not in ACTIVE_BRANDS, so these products would be "
f"written to {result.table} and then filtered out of /api/brands, "
f"search, suggest and the category listing. The rows would be complete "
f"and correct, just unreadable. To make them visible, set "
f"'{_env_line(result.brand)}' in backend/.env and restart the API - "
f"settings are read once at import, so a restart is required."
)
def _inactive_detail(brand: str) -> str:
return (
f"{brand} is not in ACTIVE_BRANDS. Ingesting it now would write a "
f"complete catalog that no endpoint can read. Either set "
f"'{_env_line(brand)}' in backend/.env and restart the API first, or "
f"re-send with acknowledge_inactive=true to stage the data anyway - "
f"adding the brand later is a config change and a restart, not a "
f"re-ingest."
)

View File

@@ -48,6 +48,20 @@ def generate_catalog(payload: CatalogGenerateRequest) -> CatalogJobOut:
"""Kick off brand catalog ingestion (discovery -> images -> embeddings ->
pgvector) as a background daemon thread and return immediately with a job id.
PREFER /api/admin/brand-discovery/* FOR NEW WORK. This route is unchanged
and still supported, but it does not run the eleven stages in
`app/core/store_catalog_pipeline.py` - no title validation, no pack-size
explosion, no SKU resolution, no barcode, no HSN/GST, no validation gate -
and it mints `image_id` with `s3_service.generate_image_id()`, which appends
a random uuid4. Nothing it writes can ever match an existing row, so running
it twice for one brand produces two catalogs. It also calls
`upsert_brand_products(cleanup=True)`, which deletes every row not in the
batch it just built.
The discovery routes do run all eleven stages, use a deterministic
`image_id`, write with `cleanup=False`, and show the products for approval
before anything is stored. See docs/BRAND_DISCOVERY.md.
NOTE: on an 8GB RAM / CPU-only machine, running ingestion (which loads
the embeddings model and calls Ollama repeatedly) at the same time as
heavy chat traffic will be slow. This is intended as an occasional

View File

@@ -8,6 +8,18 @@ product discovery (Ollama), per-product image search + S3 upload,
pricing/description enrichment, embedding generation, and the pgvector
upsert. This module exists only to give that pipeline one clear, reusable
entry point and a consistent result shape for callers.
NOT THE ELEVEN-STAGE PIPELINE, and prefer `app/services/brand_discovery.py`
for new work. "Full pipeline" above means this module's own sequence, not the
eleven stages in `app/core/store_catalog_pipeline.py`: there is no title
validation, no pack-size explosion, no SKU service, no barcode, no HSN/GST and
no validation gate here, and `image_id` carries a random uuid4 suffix, so a
second run for the same brand cannot match the first and duplicates it.
This path is unchanged and still works. It also rewrites the brand's seed
catalog via `brand_sync.export_brand_to_seed_file` (below), which the discovery
path deliberately does not - so the two are not drop-in replacements for each
other. See docs/BRAND_DISCOVERY.md.
"""
from __future__ import annotations

View File

@@ -316,6 +316,37 @@ BRAND_SYNC_INTERVAL_SECONDS = int(os.getenv("BRAND_SYNC_INTERVAL_SECONDS", "300"
# See app/services/active_brands.py.
ACTIVE_BRANDS = os.getenv("ACTIVE_BRANDS", "")
# ---------------------------------------------------------------------------
# Brand discovery (brand name -> the 11-stage pipeline)
# ---------------------------------------------------------------------------
# Discovery turns a brand NAME into rows the ordinary catalog pipeline ingests.
# See app/services/brand_discovery.py. Every value below has a working default,
# so the feature needs no configuration to run.
#
# Open Food Facts is the primary source and the language model is the
# supplement, not the reverse: OFF returns real products carrying a real GTIN,
# while the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) invents plausible ones that
# nothing downstream can catch. Turning BRAND_DISCOVERY_USE_OFF off leaves the
# result resting on the model alone.
BRAND_DISCOVERY_USE_OFF = _bool("BRAND_DISCOVERY_USE_OFF", "true")
BRAND_DISCOVERY_USE_LLM = _bool("BRAND_DISCOVERY_USE_LLM", "true")
# Products per discovery run. One CSV row per product; pack-size explosion
# happens later in stage 4, so this is well inside the 2000-row per-file cap.
BRAND_DISCOVERY_MAX_PRODUCTS = int(os.getenv("BRAND_DISCOVERY_MAX_PRODUCTS", "200"))
# Pack sizes kept per product when only the language model offers any. Stage 6
# runs an image search per exploded row, so this multiplies the slowest part of
# the run; 3 keeps a large brand inside a sane wall-clock.
BRAND_DISCOVERY_MAX_SIZES = int(os.getenv("BRAND_DISCOVERY_MAX_SIZES", "3"))
# Wall-clock ceiling on the LLM half of a run, checked between prompts. Open
# Food Facts runs first and is never subject to it, so a run that hits this
# still returns the evidence-backed products.
BRAND_DISCOVERY_DEADLINE_SECONDS = float(
os.getenv("BRAND_DISCOVERY_DEADLINE_SECONDS", "300")
)
# ---------------------------------------------------------------------------
# S3 / DigitalOcean Spaces (product image storage) - optional
# ---------------------------------------------------------------------------

View File

@@ -31,7 +31,7 @@ from app.api.routers import health, brands, search, suggest, chat, catalog, syst
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
from app.api.routers import nutrition, nutrition_admin, upload
from app.api.routers import auth, user_products, admin_train, mcp_info
from app.api.routers import batch_catalog, uploads
from app.api.routers import batch_catalog, uploads, brand_discovery
from app.services.store_db import ensure_store_intelligence_schema
from app.services.nutrition_db import ensure_nutrition_schema
@@ -307,6 +307,7 @@ app.include_router(nutrition_admin.router, prefix="/api")
app.include_router(upload.router, prefix="/api")
app.include_router(batch_catalog.router, prefix="/api")
app.include_router(uploads.router, prefix="/api")
app.include_router(brand_discovery.router, prefix="/api")
app.include_router(mcp_info.router, prefix="/api")
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,

File diff suppressed because it is too large Load Diff

2534
data/cache/off_brand_corpus/britannia.json vendored Normal file

File diff suppressed because it is too large Load Diff

View File

@@ -1,21 +1,254 @@
{
"_sku_sequences": {
"BRITAN-GOO-100": 10,
"BRITAN-GOO-200": 12,
"BRITAN-GOO-100": 97,
"BRITAN-GOO-200": 153,
"BRITAN-GOO-500": 9,
"BRITAN-GOO-250": 1,
"BRITAN-MAR-100": 2,
"BRITAN-MAR-250": 2,
"BRITAN-MAR-500": 1,
"BRITAN-MIL-200": 2,
"BRITAN-MIL-500": 1,
"BRITAN-MIL-1": 1,
"BRITAN-GOO-375": 1,
"BRITAN-MIL-200": 58,
"BRITAN-MIL-500": 42,
"BRITAN-MIL-1": 28,
"BRITAN-GOO-375": 55,
"BRITAN-MAR-200": 1,
"BRITAN-MAR-375": 1,
"BRITAN-MIL-100": 1,
"BRITAN-MIL-375": 1,
"BRITAN-MIL-100": 58,
"BRITAN-MIL-375": 28,
"BRITAN-MIL-150": 1,
"BRITAN-50-200": 1
"BRITAN-50-200": 1,
"BRITAN-100-450": 29,
"BRITAN-50-76": 29,
"BRITAN-50-100": 29,
"BRITAN-50-250": 27,
"BRITAN-50-500": 27,
"BRITAN-50-110": 29,
"BRITAN-50-1": 58,
"BRITAN-50-455": 29,
"BRITAN-505-40": 1,
"BRITAN-505-80": 1,
"BRITAN-505-150": 1,
"BRITAN-505-200": 29,
"BRITAN-505-500": 53,
"BRITAN-505-1": 27,
"BRITAN-ATT-200": 29,
"BRITAN-ATT-400": 27,
"BRITAN-ATT-600": 27,
"BRITAN-BIS-295": 29,
"BRITAN-BIS-100": 29,
"BRITAN-BOU-50": 29,
"BRITAN-BR-78": 29,
"BRITAN-BRI-700": 29,
"BRITAN-BRI-100": 142,
"BRITAN-BRI-250": 105,
"BRITAN-BRI-400": 27,
"BRITAN-BRI-40": 32,
"BRITAN-BRI-50": 29,
"BRITAN-BRI-80": 3,
"BRITAN-BRI-150": 3,
"BRITAN-50-63": 29,
"BRITAN-505-692": 29,
"BRITAN-505-300": 29,
"BRITAN-505-29": 17,
"BRITAN-BOU-40": 1,
"BRITAN-BOU-80": 1,
"BRITAN-BOU-150": 30,
"BRITAN-BRE-250": 29,
"BRITAN-CAK-120": 29,
"BRITAN-CAK-50": 29,
"BRITAN-CAK-35": 29,
"BRITAN-CAK-110": 29,
"BRITAN-CAK-27": 29,
"BRITAN-CHE-250": 29,
"BRITAN-CHE-200": 87,
"BRITAN-CHE-500": 81,
"BRITAN-CHE-1": 81,
"BRITAN-COW-1": 29,
"BRITAN-FUD-120": 29,
"BRITAN-GD-600": 29,
"BRITAN-GOL-40": 1,
"BRITAN-GOL-80": 1,
"BRITAN-GOL-150": 1,
"BRITAN-GOO-60": 58,
"BRITAN-GOO-111": 29,
"BRITAN-GOO-400": 29,
"BRITAN-MAR-1": 29,
"BRITAN-MIB-300": 29,
"BRITAN-NIC-594": 29,
"BRITAN-NIC-143": 29,
"BRITAN-NUT-100": 129,
"BRITAN-NUT-75": 101,
"BRITAN-PAT-40": 1,
"BRITAN-PAT-80": 1,
"BRITAN-PAT-150": 1,
"BRITAN-PAV-200": 29,
"BRITAN-THI-150": 29,
"BRITAN-TIM-40": 2,
"BRITAN-TIM-80": 2,
"BRITAN-TIM-150": 2,
"BRITAN-TOA-250": 17,
"BRITAN-TRA-40": 1,
"BRITAN-TRA-80": 1,
"BRITAN-TRA-150": 1,
"BRITAN-TRE-60": 58,
"BRITAN-TRE-45": 29,
"BRITAN-TRE-47": 29,
"BRITAN-TRE-40": 5,
"BRITAN-TRE-80": 5,
"BRITAN-TRE-150": 5,
"BRITAN-TRE-51": 29,
"BRITAN-VIT-400": 85,
"BRITAN-WIN-180": 29,
"BRITAN-WIN-40": 5,
"BRITAN-WIN-80": 5,
"BRITAN-WIN-150": 5,
"BRITAN-BRO-450": 29,
"BRITAN-BRT-200": 29,
"BRITAN-BRT-500": 27,
"BRITAN-BRT-1": 27,
"BRITAN-CAK-64": 29,
"BRITAN-CAK-60": 29,
"BRITAN-CHO-20": 29,
"BRITAN-CHO-55": 27,
"BRITAN-CHO-150": 27,
"BRITAN-CRA-100": 29,
"BRITAN-DAI-1": 29,
"BRITAN-GOB-110": 29,
"BRITAN-GOB-50": 58,
"BRITAN-GOO-600": 57,
"BRITAN-GOO-120": 58,
"BRITAN-GOO-39": 29,
"BRITAN-HAR-40": 1,
"BRITAN-HAR-80": 1,
"BRITAN-HAR-150": 1,
"BRITAN-JIM-25": 29,
"BRITAN-JIM-685": 29,
"BRITAN-JIM-40": 1,
"BRITAN-JIM-80": 1,
"BRITAN-JIM-150": 1,
"BRITAN-JIM-460": 29,
"BRITAN-JIM-70": 29,
"BRITAN-LIT-13": 29,
"BRITAN-LIT-75": 29,
"BRITAN-MAR-117": 43,
"BRITAN-MIL-115": 29,
"BRITAN-MUF-100": 29,
"BRITAN-MUF-250": 27,
"BRITAN-MUF-400": 27,
"BRITAN-MUL-400": 29,
"BRITAN-NUT-120": 29,
"BRITAN-NUT-300": 56,
"BRITAN-NUT-50": 29,
"BRITAN-NUT-200": 41,
"BRITAN-NUT-150": 30,
"BRITAN-NUT-40": 1,
"BRITAN-NUT-80": 1,
"BRITAN-POT-40": 1,
"BRITAN-POT-80": 1,
"BRITAN-POT-150": 1,
"BRITAN-PUR-300": 29,
"BRITAN-PUR-75": 29,
"BRITAN-RAG-100": 29,
"BRITAN-RAG-200": 27,
"BRITAN-RAG-375": 27,
"BRITAN-ROL-40": 1,
"BRITAN-ROL-80": 1,
"BRITAN-ROL-150": 1,
"BRITAN-SAN-450": 29,
"BRITAN-SUG-100": 29,
"BRITAN-SUG-200": 27,
"BRITAN-SUG-300": 27,
"BRITAN-SWE-1": 29,
"BRITAN-LAU-120": 29,
"BRITAN-TIG-150": 30,
"BRITAN-TIG-400": 29,
"BRITAN-TIG-100": 57,
"BRITAN-TIG-200": 27,
"BRITAN-TIG-375": 27,
"BRITAN-TIG-497": 29,
"BRITAN-TIG-40": 1,
"BRITAN-TIG-80": 1,
"BRITAN-TIG-63": 29,
"BRITAN-TIG-75": 29,
"BRITAN-TOA-180": 28,
"BRITAN-TOA-300": 27,
"BRITAN-TOA-600": 27,
"BRITAN-TOA-200": 41,
"BRITAN-TOA-275": 29,
"BRITAN-TRE-14": 29,
"BRITAN-TRE-55": 29,
"BRITAN-TRE-75": 29,
"BRITAN-TRE-100": 167,
"BRITAN-TRE-125": 27,
"BRITAN-VIT-40": 1,
"BRITAN-VIT-80": 1,
"BRITAN-VIT-150": 1,
"BRITAN-VIT-200": 29,
"BRITAN-VIT-600": 27,
"BRITAN-WIN-200": 29,
"BRITAN-WIN-500": 157,
"BRITAN-WIN-1": 27,
"BRITAN-505-100": 28,
"BRITAN-505-250": 26,
"BRITAN-BRI-500": 78,
"BRITAN-BOU-100": 42,
"BRITAN-BOU-250": 26,
"BRITAN-BOU-500": 26,
"BRITAN-GOL-100": 28,
"BRITAN-GOL-250": 26,
"BRITAN-GOL-500": 26,
"BRITAN-PAT-100": 28,
"BRITAN-PAT-250": 26,
"BRITAN-PAT-500": 26,
"BRITAN-TIM-100": 56,
"BRITAN-TIM-250": 52,
"BRITAN-TIM-500": 52,
"BRITAN-TRA-100": 28,
"BRITAN-TRA-250": 26,
"BRITAN-TRA-500": 26,
"BRITAN-TRE-250": 130,
"BRITAN-TRE-500": 130,
"BRITAN-WIN-100": 140,
"BRITAN-WIN-250": 130,
"BRITAN-HAR-100": 28,
"BRITAN-HAR-250": 26,
"BRITAN-HAR-500": 26,
"BRITAN-JIM-100": 1,
"BRITAN-JIM-250": 1,
"BRITAN-JIM-500": 1,
"BRITAN-NUT-250": 26,
"BRITAN-NUT-500": 26,
"BRITAN-POT-100": 28,
"BRITAN-POT-250": 26,
"BRITAN-POT-500": 26,
"BRITAN-ROL-100": 28,
"BRITAN-ROL-250": 26,
"BRITAN-ROL-500": 26,
"BRITAN-TIG-250": 26,
"BRITAN-TIG-500": 26,
"BRITAN-VIT-100": 28,
"BRITAN-VIT-250": 26,
"BRITAN-VIT-500": 26,
"BRITAN-JIM-92": 27,
"BRITAN-505-715": 12,
"BRITAN-50-50": 14,
"BRITAN-505-38": 14,
"BRITAN-BOU-60": 14,
"BRITAN-BOU-120": 14,
"BRITAN-BRO-400": 14,
"BRITAN-GOO-68": 14,
"BRITAN-JIM-57": 14,
"BRITAN-JIM-138": 14,
"BRITAN-JIM-350": 14,
"BRITAN-MAR-39": 14,
"BRITAN-MAR-89": 14,
"BRITAN-MAR-300": 13,
"BRITAN-MAR-73": 14,
"BRITAN-MIL-335": 14,
"BRITAN-MIL-67": 14,
"BRITAN-MUL-450": 14,
"BRITAN-NUT-52": 13,
"BRITAN-NUT-1": 14,
"BRITAN-TOA-273": 14
}
}

View File

@@ -0,0 +1,551 @@
"""Tests for brand discovery - the brand-name -> 11-stage-pipeline bridge.
Follows the pattern in test_store_catalog_pipeline.py: monkeypatch the network
and storage boundary *on the module object* (brand_discovery imports those names
directly), and assert on the rows that come out rather than on a status code.
Nothing here reaches Open Food Facts, Ollama or a database. The two source
functions are stubbed and every pipeline run uses `use_llm=False,
fetch_images=False`.
Several of these tests pin behaviour that was WRONG in the first working
version of this module and was found by round-tripping the real Britannia
corpus. They are regression tests with a known failure, not speculative ones -
each names the defect it prevents.
"""
from __future__ import annotations
import pytest
from app.api.routers.user_products import map_spreadsheet_columns, row_to_request
from app.core import store_catalog_pipeline as pipeline
from app.services import brand_discovery as bd
# ---------------------------------------------------------------------------
# Fixtures
# ---------------------------------------------------------------------------
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
"""Keep the SKU sequence counter out of the repo - see the same fixture in
test_store_catalog_pipeline.py."""
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def _no_network(monkeypatch):
"""Neither source may reach the outside world by default.
A test that wants products stubs one of these explicitly. Without this an
accidental real call would hit Open Food Facts from the suite.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [])
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [])
@pytest.fixture
def store(monkeypatch):
"""A fake brand table. Returns the dict of image_id -> stored row."""
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
for row in rows:
table[row["image_id"]] = row
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand",
lambda brand, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts",
lambda texts: [[0.0] * 384 for _ in texts])
return table
def _off(title, *, code=None, quantity=None):
"""One Open Food Facts corpus hit, in the shape `_from_open_facts` returns."""
return {"title": title, "barcode": code, "size": bd._canonical_size(quantity),
"source": "off"}
def _run(products, filename="discovered.csv"):
"""Discovered products -> CSV -> the real 11 stages."""
return pipeline.run_pipeline(filename, bd.rows_to_csv_bytes(products),
use_llm=False, fetch_images=False)
# ---------------------------------------------------------------------------
# The contract the whole CSV bridge rests on
# ---------------------------------------------------------------------------
def test_the_csv_headers_all_map_onto_catalog_fields():
"""Every emitted header must be understood by the real column mapper.
This is THE contract: discovery writes a spreadsheet and the pipeline reads
it with `map_spreadsheet_columns`. A header that does not resolve is dropped
in silence, so the column simply never arrives and the run still reports
success. Asserting it here means a rename on either side fails loudly.
"""
mapping = map_spreadsheet_columns(bd.CSV_HEADERS)
assert mapping.unrecognised == []
assert mapping.ignored == []
for header in bd.CSV_HEADERS:
assert header in mapping.columns, f"{header!r} did not resolve to a field"
def test_the_emitted_csv_parses_with_the_real_spreadsheet_reader(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts",
lambda brand, refresh=False: [_off("Marie Gold", quantity="250 g")])
result = bd.discover_brand_products("Britannia", use_llm=False)
df, mapping = pipeline.parse_spreadsheet("d.csv", bd.rows_to_csv_bytes(result.products))
assert len(df) == 1
assert mapping.unrecognised == []
request = row_to_request(df.to_dict(orient="records")[0], mapping)
assert request.brand == "Britannia"
assert request.product_name == "Marie Gold"
assert request.size_variants == ["250g"]
# ---------------------------------------------------------------------------
# Regression: list cells were shredded by the separator
# ---------------------------------------------------------------------------
def test_a_list_item_containing_a_separator_survives_the_round_trip():
"""`_string_list` splits on [,;|], and the highlight generator emits commas.
"Available in 3 sizes: 100g, 250g, 500g" and "Baked, Not Fried" became five
fragments instead of two highlights. The damage is invisible in the database
- the column is populated, just wrong - so it needs a test.
"""
cell = bd._safe_list_cell([
"Available in 3 sizes: 100g, 250g, 500g",
"Baked, Not Fried",
"Pipe | separated | too",
])
mapping = map_spreadsheet_columns(("highlights",))
from app.api.routers.user_products import _string_list
recovered = _string_list({"highlights": cell}, mapping, "highlights")
assert len(recovered) == 3
assert recovered[0] == "Available in 3 sizes: 100g / 250g / 500g"
assert recovered[1] == "Baked / Not Fried"
# ---------------------------------------------------------------------------
# Regression: pack sizes doubled into the image_id
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("raw, expected", [
("200 g", "200g"),
("200g", "200g"),
("1.5 kg", "1.5kg"),
("75 g", "75g"),
])
def test_a_spaced_pack_size_is_canonicalised(raw, expected):
"""OFF writes both "100g" and "100 g" for the same brand.
`build_image_id` slugifies "200 g" to "200_g", which is not a substring of a
name containing "200g", so the size is appended anyway and the same product
lands on two permanent rows depending on which spelling was discovered.
"""
assert bd._canonical_size(raw) == expected
@pytest.mark.parametrize("junk", ["India", "12", "200", "", None, "6"])
def test_a_quantity_that_is_not_a_pack_size_is_rejected(junk):
"""The Britannia corpus carries "India" and bare counts in `quantity`."""
assert bd._canonical_size(junk) is None
def test_a_size_glued_to_a_letter_in_the_title_is_separated():
"""OFF holds "Jim Jam92 g", where there is no word boundary before the 92.
`off_bulk._SIZE_RE` anchors on one and so cannot see it, which left the
unnormalised size in the product name and produced the image_id
`britannia_jim_jam92_g_40g`.
"""
assert bd._canonicalise_title_size("Jim Jam92 g") == "Jim Jam 92g"
assert bd._canonicalise_title_size("Britannia Good Day 200 g") == "Britannia Good Day 200g"
def test_no_stored_image_id_carries_the_pack_size_twice(store):
result = _run([
bd.DiscoveredProduct(brand="Britannia", product_name="Jim Jam",
title="Jim Jam", category="", category_hint="Biscuits & Cookies",
description="", size_variants=["92g"]),
])
assert result.inserted == 1
assert list(store) == ["britannia_jim_jam_92g"]
# ---------------------------------------------------------------------------
# Regression: the pack size belongs in ONE place
# ---------------------------------------------------------------------------
def test_a_size_in_the_title_is_not_applied_twice(monkeypatch):
"""OFF holds "Britannia Toastea 200g" with a `quantity` of "250g".
Taking the quantity and keeping the title produced the product name
"Britannia Toastea 200g 250g" - two sizes, one name, a pack that does not
exist. The title's own size wins and is removed from the name.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.product_name == "Britannia Toastea"
assert product.size_variants == ["200g"]
def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch):
"""OFF titles carry carton counts: "Good Day Butter Cookies (25)".
That is how many units ship in a box, not part of what the product is
called, and leaving it in put the carton count into the image_id.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Good Day Butter Cookies (25)", quantity="60 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.product_name == "Good Day Butter Cookies"
assert product.size_variants == ["60g"]
# ---------------------------------------------------------------------------
# Regression: one GTIN, one pack
# ---------------------------------------------------------------------------
def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
"""A GTIN identifies one pack, and stage 4 invents three when it has none.
Every column is copied into each exploded variant, so a surviving barcode
would be stamped onto two packs that do not exist - wrong data that looks
authoritative.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", code="8901063017702"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.size_variants == []
assert product.barcode is None
assert any("barcode dropped" in note for note in product.notes)
def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.barcode == "8901063012516"
assert product.size_variants == ["100g"]
def test_no_gtin_is_written_to_more_than_one_row(store, monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
_off("50 50 Gol Maal", code="8901063017702"),
])
result = bd.discover_brand_products("Britannia", use_llm=False)
_run(result.products)
barcodes = [row["barcode"] for row in store.values() if row["barcode"]]
assert len(barcodes) == len(set(barcodes))
# ---------------------------------------------------------------------------
# Regression: the generated description poisoned the category and the tax code
# ---------------------------------------------------------------------------
def test_no_category_is_written_when_no_source_stated_one(monkeypatch):
"""Stage 3 owns category detection, and treats a supplied value as final.
It sets `_category_deterministic` and then rewrites the title against the
category, so a wrong guess here does not merely mislabel a row - it renames
the product.
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", quantity="100 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.category == ""
assert product.category_hint # still reported, for the preview
def test_a_stated_category_is_carried_through(monkeypatch):
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Good Day Cashew", "category": "Biscuits & Cookies",
"description": "Cashew cookies", "sizes": ["75g"], "providers": [],
"source": "llm"},
])
result = bd.discover_brand_products("Britannia", require_evidence=False)
assert result.products[0].category == "Biscuits & Cookies"
def test_no_boilerplate_description_is_generated(monkeypatch):
"""`generate_detailed_description` contains the word "taste".
`detect_category_from_text` matches keywords fuzzily and "taste" is one edit
from the Oral Care keyword "paste", so every weakly-titled product was
classified Oral Care and stage 9 stamped it with HSN 3306 - the tax code for
dentifrices. A blank description is honest, and `_to_storage_row` already
substitutes "<name> <size> from <brand>."
"""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", quantity="100 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.description == ""
assert "taste" not in (product.description or "").lower()
def test_a_weakly_titled_product_is_not_classified_as_oral_care(store, monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("50 50 Gol Maal", quantity="100 g"),
])
result = bd.discover_brand_products("Britannia", use_llm=False)
_run(result.products)
stored = next(iter(store.values()))
assert stored["category"] != "Oral Care"
assert stored["hsn_code"] != "3306"
# ---------------------------------------------------------------------------
# Regression: catalogue growth by re-run
# ---------------------------------------------------------------------------
def test_a_known_product_is_reemitted_under_its_stored_name(monkeypatch):
"""A brand prefix appearing or disappearing is the realistic drift.
The stored name has no brand prefix and the discovered one does; both
normalise identically once `brand_tokens` are stripped, so the row folds and
the image_id stays put.
"""
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
{"product_name": "Good Day Cashew Cookies 75g"},
])
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Britannia Good Day Cashew Cookies", quantity="75 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.matches_existing == "Good Day Cashew Cookies"
assert product.product_name == "Good Day Cashew Cookies"
def test_a_shorter_title_does_not_fold_onto_a_longer_stored_one(monkeypatch):
"""The documented limit of the 0.85 threshold, pinned so it is a decision
rather than a surprise.
"Good Day Cashew" scores 0.75 against "Good Day Cashew Cookies" and stays a
separate product. Relaxing the threshold far enough to fold it would also
fold "Dairy Milk Silk" onto "Dairy Milk Silk Minis", which is a different
product - see `_same_product`.
"""
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
{"product_name": "Good Day Cashew Cookies 75g"},
])
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Good Day Cashew", quantity="75 g"),
])
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
assert product.matches_existing is None
def test_two_real_packs_sharing_a_name_stay_on_their_own_rows(monkeypatch):
"""OFF holds "Jim Jam" at 25g and "Jim jam" at 92g.
Both normalise to the same size-free key. Re-emitting the first stored
DISPLAY name gave the 92g pack the 25g name, which `_to_storage_row` then
extended to "Jim Jam 25g 92g" under a brand-new image_id. The index hands
back the base name so the size can be re-applied per pack.
"""
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
{"product_name": "Jim Jam 25g"},
{"product_name": "Jim jam 92g"},
])
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Jim Jam", quantity="25 g"),
_off("Jim jam", quantity="92 g"),
])
products = bd.discover_brand_products("Britannia", use_llm=False).products
names = {p.product_name for p in products}
assert names == {"Jim Jam"}
assert sorted(s for p in products for s in p.size_variants) == ["25g", "92g"]
assert not any("25g 92g" in p.product_name for p in products)
def test_re_running_discovery_and_ingest_writes_nothing_new(store, monkeypatch):
"""The whole feature's safety property, end to end.
Deterministic image_id + fill-only-blanks + `cleanup=False` make a re-run a
no-op, but only if discovery re-emits the same names. This exercises the
real pipeline twice with the catalog fed back in between.
"""
corpus = [
_off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"),
_off("Good Day Butter Cookies (25)", quantity="60 g"),
_off("Jim Jam92 g", code="8901063019027"),
_off("50 50 Gol Maal", code="8901063017702"),
]
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: corpus)
monkeypatch.setattr(bd, "get_products_by_brand",
lambda brand, **kw: list(store.values()))
first = _run(bd.discover_brand_products("Britannia", use_llm=False).products)
after_first = dict(store)
assert first.inserted > 0
second = _run(bd.discover_brand_products("Britannia", use_llm=False).products)
assert second.inserted == 0
assert second.backfilled == 0
assert second.skipped_existing == len(after_first)
assert set(store) == set(after_first)
# ---------------------------------------------------------------------------
# Sources, evidence and scoring
# ---------------------------------------------------------------------------
def test_open_food_facts_results_survive_an_unreachable_ollama(monkeypatch):
"""OFF runs first and unconditionally, so a dead LLM degrades the result
rather than emptying it - the common case on a CPU-only box."""
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Marie Gold", code="8901063014206", quantity="250 g"),
])
monkeypatch.setattr(bd.ollama_service, "_generate", lambda *a, **k: "")
monkeypatch.setattr(bd.ollama_service, "get_categories_for_brand", lambda brand: [])
result = bd.discover_brand_products("Britannia", use_llm=True)
assert len(result.products) == 1
assert result.products[0].sources == ["off"]
assert any("language model returned no products" in w for w in result.warnings)
def test_a_product_found_by_both_sources_is_one_row_and_scores_highest(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Marie Gold", code="8901063014206", quantity="250 g"),
])
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Marie Gold", "category": "Biscuits & Cookies",
"description": "Tea-time biscuit", "sizes": ["250g"],
"providers": ["Amazon"], "source": "llm"},
])
result = bd.discover_brand_products("Britannia")
assert len(result.products) == 1
product = result.products[0]
assert sorted(product.sources) == ["llm", "off"]
assert product.confidence == 1.0
assert result.counts["corroborated"] == 1
def test_an_uncorroborated_llm_product_is_dropped_when_evidence_is_required(monkeypatch):
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Britannia Quantum Wafer", "category": None, "description": None,
"sizes": [], "providers": [], "source": "llm"},
])
result = bd.discover_brand_products("Britannia", require_evidence=True)
assert result.products == []
assert result.counts["dropped_without_evidence"] == 1
def test_an_uncorroborated_llm_product_is_kept_but_unticked_when_evidence_is_optional(monkeypatch):
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Britannia Quantum Wafer", "category": None, "description": None,
"sizes": [], "providers": [], "source": "llm"},
])
result = bd.discover_brand_products("Britannia", require_evidence=False)
assert len(result.products) == 1
product = result.products[0]
assert product.evidence is None
assert product.confidence < 0.5
assert product.as_preview()["selected"] is False
def test_a_sub_brand_match_counts_as_registry_evidence(monkeypatch):
""""Good Day" is a registered Britannia sub-brand, so an LLM row naming it
is grounded without needing Open Food Facts."""
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
{"title": "Good Day Chocochip", "category": None, "description": None,
"sizes": ["100g"], "providers": [], "source": "llm"},
])
result = bd.discover_brand_products("Britannia", require_evidence=True)
assert len(result.products) == 1
assert result.products[0].evidence == "registry"
def test_discovery_stops_at_max_products(monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off(f"Product {n}", quantity="100 g") for n in range(50)
])
result = bd.discover_brand_products("Britannia", max_products=10, use_llm=False)
assert len(result.products) == 10
def test_the_brand_table_and_active_state_are_reported(monkeypatch):
monkeypatch.setattr(bd.active_brands, "is_active_brand", lambda brand: False)
monkeypatch.setattr(bd.active_brands, "filtering_enabled", lambda: True)
result = bd.discover_brand_products("Britannia", use_llm=False)
assert result.table == "brand_britannia"
assert result.parent_brand == "britannia"
assert result.brand_active is False
assert result.filtering_enabled is True
def test_an_empty_brand_name_is_refused():
with pytest.raises(ValueError):
bd.discover_brand_products(" ")
# ---------------------------------------------------------------------------
# Field completeness through the real stages
# ---------------------------------------------------------------------------
def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch):
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
])
result = bd.discover_brand_products("Britannia", use_llm=False)
_run(result.products)
row = next(iter(store.values()))
for column in ("product_name", "title", "description", "category", "image_id",
"price_range", "size_variants", "providers", "highlights",
"nutrients", "fssai_license", "product_sku", "sku_source",
"search_query"):
assert row.get(column), f"{column} was left empty"
assert row["barcode"] == "8901063012516"
assert row["fssai_license"] == "10012022000103"

View File

@@ -0,0 +1,361 @@
"""HTTP tests for /api/admin/brand-discovery/*.
Reuses the fixture set from test_batch_catalog_ingest.py verbatim - in
particular `batch_root` and `no_background_worker`, whose absence once wrote
batch manifests into the repository's own data directory. Read that fixture's
docstring before removing either from a test here.
Discovery itself is stubbed at the module boundary; the point of this file is
the routes, the ACTIVE_BRANDS gate and the staging handoff, not the merge logic
(which test_brand_discovery.py covers).
"""
from __future__ import annotations
import pytest
from app.api import batch_common
from app.core import batch_ingest
from app.core import store_catalog_pipeline as pipeline
from app.services import brand_discovery as bd
PREVIEW = "/api/admin/brand-discovery/preview"
INGEST = "/api/admin/brand-discovery/ingest"
# ---------------------------------------------------------------------------
# Fixtures - see test_batch_catalog_ingest.py for the rationale behind each
# ---------------------------------------------------------------------------
@pytest.fixture(autouse=True)
def _isolate_sku_counter(tmp_path, monkeypatch):
from app.services import sku_service
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
@pytest.fixture(autouse=True)
def batch_root(tmp_path, monkeypatch):
root = tmp_path / "batch_uploads"
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", root)
return root
@pytest.fixture(autouse=True)
def no_background_worker(monkeypatch):
"""Stub the worker. Its absence once wrote manifests into the working tree -
the long docstring in test_batch_catalog_ingest.py explains how."""
from app.core import batch_worker
submitted: list = []
monkeypatch.setattr(batch_worker, "submit", submitted.append)
return submitted
@pytest.fixture(autouse=True)
def _clean_job_store():
"""The job store is a module singleton and outlives a test - see the same
fixture in test_uploads_api.py."""
from app.api.batch_job_store import batch_job_store
batch_job_store._batches.clear()
batch_job_store._cancelled.clear()
yield
batch_job_store._batches.clear()
batch_job_store._cancelled.clear()
@pytest.fixture(autouse=True)
def _active_brands_allow_everything(monkeypatch):
"""Filtering off by default, so only the tests that are about the
ACTIVE_BRANDS gate have to think about it."""
from app.services import active_brands
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: False)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: True)
@pytest.fixture
def discovered(monkeypatch):
"""A fixed two-product discovery result."""
def fake_discover(brand, **kwargs):
return bd.DiscoveryResult(
brand=brand,
parent_brand="britannia",
table="brand_britannia",
brand_active=True,
filtering_enabled=False,
products=[
bd.DiscoveredProduct(
brand=brand, product_name="Marie Gold", title="Marie Gold",
category="", category_hint="Biscuits & Cookies", description="",
size_variants=["250g"], barcode="8901063014206",
sources=["off"], evidence="openfacts", confidence=0.9,
),
bd.DiscoveredProduct(
brand=brand, product_name="Milk Bikis", title="Milk Bikis",
category="", category_hint="Biscuits & Cookies", description="",
size_variants=["100g"], barcode="8901063012516",
sources=["off"], evidence="openfacts", confidence=0.9,
),
],
counts={"discovered": 2},
)
from app.api.routers import brand_discovery as router_module
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products",
fake_discover)
return fake_discover
def _payload(**overrides):
body = {
"brand": "Britannia",
"products": [
{"product_name": "Marie Gold", "size_variants": ["250g"],
"barcode": "8901063014206", "providers": ["Amazon"],
"highlights": ["Tea-time favourite"], "nutrients": ["Iron - Blood health"]},
],
}
body.update(overrides)
return body
# ---------------------------------------------------------------------------
# Auth
# ---------------------------------------------------------------------------
def test_preview_requires_admin(client):
assert client.post(PREVIEW, json={"brand": "Britannia"}).status_code in (401, 403)
def test_ingest_requires_admin(client):
assert client.post(INGEST, json=_payload()).status_code in (401, 403)
def test_a_plain_user_cannot_discover(client, user_headers):
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=user_headers)
assert resp.status_code == 403
# ---------------------------------------------------------------------------
# Preview writes nothing
# ---------------------------------------------------------------------------
def test_preview_returns_products_and_the_stage_list(client, admin_headers, discovered):
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
assert resp.status_code == 200
body = resp.json()
assert len(body["products"]) == 2
assert body["table"] == "brand_britannia"
assert body["stages"] == list(pipeline.STAGE_NAMES)
assert len(body["stages"]) == 11
assert body["products"][0]["selected"] is True
def test_preview_stages_nothing(client, admin_headers, discovered, batch_root,
no_background_worker):
"""The whole reason preview is a separate route."""
client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
assert no_background_worker == []
assert not batch_root.exists() or list(batch_root.iterdir()) == []
def test_preview_reports_a_bad_brand_name_as_client_error(client, admin_headers):
resp = client.post(PREVIEW, json={"brand": " "}, headers=admin_headers)
assert resp.status_code == 400
def test_preview_surfaces_a_discovery_failure_rather_than_500(client, admin_headers,
monkeypatch):
from app.api.routers import brand_discovery as router_module
def boom(brand, **kwargs):
raise RuntimeError("Open Food Facts is unreachable")
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", boom)
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
assert resp.status_code == 502
assert "unreachable" in resp.json()["detail"]
# ---------------------------------------------------------------------------
# The ACTIVE_BRANDS gate
# ---------------------------------------------------------------------------
def test_preview_warns_when_the_brand_is_not_active(client, admin_headers, discovered,
monkeypatch):
from app.services import active_brands
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"])
from app.api.routers import brand_discovery as router_module
def inactive(brand, **kwargs):
result = discovered(brand, **kwargs)
result.brand_active = False
result.filtering_enabled = True
return result
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", inactive)
body = client.post(PREVIEW, json={"brand": "Britannia"},
headers=admin_headers).json()
assert body["brand_active"] is False
assert any("ACTIVE_BRANDS" in w for w in body["warnings"])
def test_ingest_is_refused_for_an_inactive_brand(client, admin_headers, monkeypatch):
"""A green run over a catalog no endpoint can read is not an acceptable
outcome to hand back silently."""
from app.services import active_brands
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul", "Cadbury"])
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
assert resp.status_code == 409
detail = resp.json()["detail"]
assert "ACTIVE_BRANDS=Amul,Cadbury,Britannia" in detail
assert "restart" in detail.lower()
def test_ingest_proceeds_for_an_inactive_brand_when_acknowledged(client, admin_headers,
monkeypatch):
from app.services import active_brands
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"])
resp = client.post(INGEST, json=_payload(acknowledge_inactive=True),
headers=admin_headers)
assert resp.status_code == 202
# ---------------------------------------------------------------------------
# Ingest
# ---------------------------------------------------------------------------
def test_ingest_stages_one_batch_and_returns_a_pollable_id(client, admin_headers,
no_background_worker):
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
assert resp.status_code == 202
body = resp.json()
assert body["batch_id"]
assert body["files_total"] == 1
assert len(body["stage_names"]) == 11
assert no_background_worker == [body["batch_id"]]
def test_the_staged_file_is_a_real_csv_the_pipeline_can_read(client, admin_headers,
batch_root):
"""Not a placeholder: the bytes on the batch volume are the exact input that
produced the rows, and they must parse with the same reader an upload uses."""
batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"]
manifest = batch_ingest.read_manifest(batch_id)
entry = manifest.files[0]
assert entry.filename.startswith("discovered-britannia-")
assert entry.filename.endswith(".csv")
contents = (batch_ingest.batch_dir(batch_id) / entry.stored_name).read_bytes()
df, mapping = pipeline.parse_spreadsheet(entry.filename, contents)
assert len(df) == 1
assert mapping.unrecognised == []
assert "product_name" in mapping.columns
def test_the_batch_records_where_it_came_from(client, admin_headers):
batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"]
manifest = batch_ingest.read_manifest(batch_id)
assert manifest.submitted_by == "brand-discovery: Britannia"
def test_ingest_runs_the_eleven_stages_end_to_end(client, admin_headers, store,
no_background_worker, monkeypatch):
"""Drive the real worker function over the staged file.
`use_llm` and `fetch_images` are turned OFF for this one, matching every
other pipeline test in the suite: with them on, `run_batch` really calls
Ollama once per row and really searches the open web for images. That is the
correct production default and a terrible test - slow, and it fails when the
machine is offline.
"""
body = _payload(use_llm=False, fetch_images=False)
batch_id = client.post(INGEST, json=body, headers=admin_headers).json()["batch_id"]
batch_ingest.run_batch(batch_id)
assert len(store) == 1
row = next(iter(store.values()))
assert row["image_id"] == "britannia_marie_gold_250g"
assert row["barcode"] == "8901063014206"
assert row["product_sku"]
assert row["highlights"] == ["Tea-time favourite"]
def test_an_empty_selection_is_refused(client, admin_headers):
resp = client.post(INGEST, json=_payload(products=[]), headers=admin_headers)
assert resp.status_code == 400
assert "nothing to ingest" in resp.json()["detail"]
def test_a_blank_brand_is_refused(client, admin_headers):
resp = client.post(INGEST, json=_payload(brand=" "), headers=admin_headers)
assert resp.status_code == 400
def test_a_full_queue_is_reported_as_429(client, admin_headers, monkeypatch):
import queue
def full(batch_id):
raise queue.Full()
from app.core import batch_worker
monkeypatch.setattr(batch_worker, "submit", full)
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
assert resp.status_code == 429
assert "Resume" in resp.json()["detail"]
def test_ingest_defaults_turn_on_images_and_the_per_row_llm(client, admin_headers,
monkeypatch):
"""The two enrichment stages this feature runs with. Both are per-run
arguments already, so neither needs a settings change."""
seen = {}
real = batch_common.stage_and_queue
def spy(valid, invalid, *, use_llm, fetch_images, submitted_by=None):
seen["use_llm"] = use_llm
seen["fetch_images"] = fetch_images
return real(valid, invalid, use_llm=use_llm, fetch_images=fetch_images,
submitted_by=submitted_by)
from app.api.routers import brand_discovery as router_module
monkeypatch.setattr(router_module.batch_common, "stage_and_queue", spy)
client.post(INGEST, json=_payload(), headers=admin_headers)
assert seen == {"use_llm": True, "fetch_images": True}
@pytest.fixture
def store(monkeypatch):
table: dict = {}
def fake_upsert(brand, rows, cleanup=False):
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
for row in rows:
table[row["image_id"]] = dict(row)
return len(rows)
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
return table