Brand Ingestion
This commit is contained in:
28
.env.example
28
.env.example
@@ -267,6 +267,34 @@ BRAND_SYNC_INTERVAL_SECONDS=300
|
||||
#ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever,Own Products
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Brand discovery (a brand NAME -> the 11-stage pipeline)
|
||||
# ---------------------------------------------------------------------------
|
||||
# POST /api/admin/brand-discovery/preview finds a brand's products, and
|
||||
# /ingest stages the ones an admin approved as an ordinary catalog batch.
|
||||
# Every value below has a working default; none of these need to be set.
|
||||
#
|
||||
# Open Food Facts is the primary source and the language model is the
|
||||
# supplement. OFF returns real products with real barcodes and pack sizes;
|
||||
# the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) will invent plausible ones, and
|
||||
# nothing downstream can tell a well-formed fiction from a real product. Turn
|
||||
# BRAND_DISCOVERY_USE_OFF off and the result rests on the model alone.
|
||||
#
|
||||
# NOTE: discovering a brand that is not in ACTIVE_BRANDS writes a complete
|
||||
# catalog that no endpoint can read. The ingest route refuses with a 409 and
|
||||
# names the line to add here; it is a config change plus a restart, never a
|
||||
# re-ingest.
|
||||
#BRAND_DISCOVERY_USE_OFF=true
|
||||
#BRAND_DISCOVERY_USE_LLM=true
|
||||
#BRAND_DISCOVERY_MAX_PRODUCTS=200
|
||||
# Pack sizes kept per product when only the language model offers any. Stage 6
|
||||
# runs an image search per exploded row, so this multiplies the slowest stage.
|
||||
#BRAND_DISCOVERY_MAX_SIZES=3
|
||||
# Wall-clock ceiling on the language-model half, checked between prompts. Open
|
||||
# Food Facts runs first and is never subject to it.
|
||||
#BRAND_DISCOVERY_DEADLINE_SECONDS=300
|
||||
|
||||
|
||||
USE_S3=true
|
||||
S3_ACCESS_KEY=your-do-spaces-key
|
||||
S3_SECRET_KEY=your-do-spaces-secret
|
||||
|
||||
253
app/api/routers/brand_discovery.py
Normal file
253
app/api/routers/brand_discovery.py
Normal file
@@ -0,0 +1,253 @@
|
||||
"""Admin routes for brand discovery: a brand NAME into the 11-stage pipeline.
|
||||
|
||||
Two steps on purpose, mirroring `batch_catalog.py`'s preview/ingest split.
|
||||
|
||||
`/preview` discovers and returns; it writes nothing, anywhere. `/ingest` takes
|
||||
the rows the admin kept, renders them as a CSV, and hands the bytes to the same
|
||||
`batch_common.stage_and_queue` an uploaded spreadsheet goes through - so the
|
||||
batch manifest, the stage timeline, Resume, Cancel and the nutrition
|
||||
auto-enrichment that follows a batch all work here without a line of new code.
|
||||
|
||||
The gap between the two steps is the point. Discovery's language-model half can
|
||||
invent a product that nothing downstream is able to catch: a well-formed
|
||||
fiction resolves a category, gets a price band and an internal SKU, and clears
|
||||
`product_validator`'s "verified" threshold comfortably. `product_validator` was
|
||||
built to reject MALFORMED rows, not false ones. A person looking at the list is
|
||||
the check, so the list is shown before anything is written.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException, status
|
||||
from pydantic import BaseModel, Field
|
||||
from starlette.concurrency import run_in_threadpool
|
||||
|
||||
from app.api import batch_common
|
||||
from app.api.deps import require_admin
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.infrastructure.settings import (
|
||||
BATCH_MAX_FILES,
|
||||
BATCH_MAX_TOTAL_BYTES,
|
||||
BATCH_MAX_TOTAL_ROWS,
|
||||
BRAND_DISCOVERY_DEADLINE_SECONDS,
|
||||
BRAND_DISCOVERY_MAX_PRODUCTS,
|
||||
)
|
||||
from app.services import active_brands, brand_discovery
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/admin/brand-discovery", tags=["admin", "catalog"])
|
||||
|
||||
# The same per-file ceilings the admin batch routes apply. Discovery emits one
|
||||
# CSV row per product and pack-size explosion happens later, inside stage 4, so
|
||||
# 200 products is 200 rows here - three orders of magnitude inside the limit.
|
||||
MAX_UPLOAD_BYTES = 10 * 1024 * 1024
|
||||
MAX_UPLOAD_ROWS = 2000
|
||||
|
||||
|
||||
def _limits() -> batch_common.UploadLimits:
|
||||
return batch_common.UploadLimits(
|
||||
max_files=BATCH_MAX_FILES,
|
||||
max_file_bytes=MAX_UPLOAD_BYTES,
|
||||
max_file_rows=MAX_UPLOAD_ROWS,
|
||||
max_total_bytes=BATCH_MAX_TOTAL_BYTES,
|
||||
max_total_rows=BATCH_MAX_TOTAL_ROWS,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Request bodies
|
||||
# ---------------------------------------------------------------------------
|
||||
class DiscoveryPreviewRequest(BaseModel):
|
||||
brand: str
|
||||
max_products: int = Field(default=BRAND_DISCOVERY_MAX_PRODUCTS, ge=1, le=2000)
|
||||
use_openfacts: bool = True
|
||||
use_llm: bool = True
|
||||
# Ungrounded language-model rows are dropped rather than shown by default.
|
||||
# Turning this off is how an admin sees them - they arrive unticked.
|
||||
require_evidence: bool = True
|
||||
# Shorter than the service default: somebody is watching a spinner.
|
||||
deadline_seconds: float = Field(default=90.0, ge=0.0,
|
||||
le=BRAND_DISCOVERY_DEADLINE_SECONDS)
|
||||
refresh_corpus: bool = False
|
||||
|
||||
|
||||
class DiscoveredProductIn(BaseModel):
|
||||
"""One row the admin kept. Mirrors `DiscoveredProduct`'s written fields.
|
||||
|
||||
Sent back rather than re-discovered so that what is ingested is exactly what
|
||||
was reviewed - a second discovery pass could legitimately return something
|
||||
different, and then the approval would have been of a different list.
|
||||
"""
|
||||
|
||||
product_name: str
|
||||
title: Optional[str] = None
|
||||
category: Optional[str] = None
|
||||
description: Optional[str] = None
|
||||
size_variants: List[str] = Field(default_factory=list)
|
||||
providers: List[str] = Field(default_factory=list)
|
||||
highlights: List[str] = Field(default_factory=list)
|
||||
nutrients: List[str] = Field(default_factory=list)
|
||||
fssai_license: Optional[str] = None
|
||||
barcode: Optional[str] = None
|
||||
image_url: Optional[str] = None
|
||||
|
||||
|
||||
class DiscoveryIngestRequest(BaseModel):
|
||||
brand: str
|
||||
products: List[DiscoveredProductIn]
|
||||
# Stage 2 only calls Ollama for a row whose description is blank, and
|
||||
# discovery leaves most of them blank on purpose (see brand_discovery's note
|
||||
# on the boilerplate generator). On an unreachable Ollama each such row
|
||||
# costs up to OLLAMA_TIMEOUT_SECONDS, so this is worth being able to turn
|
||||
# off for a large brand.
|
||||
use_llm: bool = True
|
||||
fetch_images: bool = True
|
||||
# Ingesting a brand outside ACTIVE_BRANDS writes rows nothing can read.
|
||||
# Refused unless the caller says they mean it - see the 409 below.
|
||||
acknowledge_inactive: bool = False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Routes
|
||||
# ---------------------------------------------------------------------------
|
||||
@router.post("/preview", dependencies=[Depends(require_admin)])
|
||||
async def preview_brand_discovery(payload: DiscoveryPreviewRequest) -> Dict[str, Any]:
|
||||
"""Discover a brand's products and return them. Writes nothing.
|
||||
|
||||
Run on a worker thread: discovery does blocking HTTP to Open Food Facts and,
|
||||
when the language model is enabled, a series of blocking Ollama calls. On
|
||||
the event loop that would stall every other request for the duration.
|
||||
"""
|
||||
try:
|
||||
result = await run_in_threadpool(
|
||||
brand_discovery.discover_brand_products,
|
||||
payload.brand,
|
||||
max_products=payload.max_products,
|
||||
deadline_seconds=payload.deadline_seconds,
|
||||
use_openfacts=payload.use_openfacts,
|
||||
use_llm=payload.use_llm,
|
||||
require_evidence=payload.require_evidence,
|
||||
refresh_corpus=payload.refresh_corpus,
|
||||
)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=400, detail=str(exc)) from exc
|
||||
except Exception as exc: # noqa: BLE001 - report the failure, do not 500
|
||||
logger.exception("Brand discovery failed for %r", payload.brand)
|
||||
raise HTTPException(
|
||||
status_code=502,
|
||||
detail=f"Discovery failed for {payload.brand!r}: {exc}",
|
||||
) from exc
|
||||
|
||||
body = result.as_dict()
|
||||
# Served rather than duplicated in the frontend, exactly as BatchOut does,
|
||||
# so the two cannot drift when a stage is added.
|
||||
body["stages"] = list(pipeline.STAGE_NAMES)
|
||||
if result.filtering_enabled and not result.brand_active:
|
||||
body["warnings"] = list(body.get("warnings") or []) + [_inactive_message(result)]
|
||||
return body
|
||||
|
||||
|
||||
@router.post("/ingest", status_code=status.HTTP_202_ACCEPTED,
|
||||
dependencies=[Depends(require_admin)])
|
||||
async def ingest_brand_discovery(payload: DiscoveryIngestRequest) -> batch_common.BatchOut:
|
||||
"""Stage the reviewed products as a catalog batch and return an id to poll."""
|
||||
brand = (payload.brand or "").strip()
|
||||
if not brand:
|
||||
raise HTTPException(status_code=400, detail="A brand name is required.")
|
||||
if not payload.products:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail="No products were selected, so there is nothing to ingest.",
|
||||
)
|
||||
|
||||
# A green run over an unreadable catalog is the failure this project refuses
|
||||
# to ship. The rows WOULD be written and fully enriched - nutrition
|
||||
# auto-enrichment runs with include_inactive=True - but /api/brands, search,
|
||||
# suggest and the category listing all filter the brand out, so the result
|
||||
# looks like nothing happened.
|
||||
if active_brands.filtering_enabled() and not active_brands.is_active_brand(brand):
|
||||
if not payload.acknowledge_inactive:
|
||||
raise HTTPException(status_code=409, detail=_inactive_detail(brand))
|
||||
|
||||
products = [
|
||||
brand_discovery.DiscoveredProduct(
|
||||
brand=brand,
|
||||
product_name=item.product_name,
|
||||
title=item.title or item.product_name,
|
||||
category=item.category or "",
|
||||
category_hint="",
|
||||
description=item.description or "",
|
||||
size_variants=list(item.size_variants),
|
||||
providers=list(item.providers),
|
||||
highlights=list(item.highlights),
|
||||
nutrients=list(item.nutrients),
|
||||
fssai_license=item.fssai_license,
|
||||
barcode=item.barcode,
|
||||
image_url=item.image_url,
|
||||
)
|
||||
for item in payload.products
|
||||
]
|
||||
|
||||
filename = brand_discovery.synthetic_filename(brand)
|
||||
contents = brand_discovery.rows_to_csv_bytes(products)
|
||||
|
||||
# ONE FILE, NEVER CHUNKED. stage 11 groups by brand per file and reads the
|
||||
# existing catalog per file, and its intra-file de-duplication
|
||||
# ({image_id: row}) is per file too - so the same image_id split across two
|
||||
# chunks would not be caught. A single file makes that de-duplication total.
|
||||
valid, invalid, _rows = batch_common.parse_all([(filename, contents)], _limits())
|
||||
if not valid:
|
||||
detail = "; ".join(f"{name}: {reason}" for name, reason in invalid)
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=f"The discovered products could not be staged. {detail}",
|
||||
)
|
||||
|
||||
manifest, started = batch_common.stage_and_queue(
|
||||
valid, invalid,
|
||||
use_llm=payload.use_llm,
|
||||
fetch_images=payload.fetch_images,
|
||||
submitted_by=f"brand-discovery: {brand}",
|
||||
)
|
||||
if not started:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=(
|
||||
"Too many batches are already queued. This one has been saved - "
|
||||
"press Resume on it once the current batch finishes."
|
||||
),
|
||||
)
|
||||
return batch_common.to_out(manifest)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The ACTIVE_BRANDS message, in one place
|
||||
# ---------------------------------------------------------------------------
|
||||
def _env_line(brand: str) -> str:
|
||||
names = active_brands.active_display_names()
|
||||
return "ACTIVE_BRANDS=" + ",".join(list(names) + [brand])
|
||||
|
||||
|
||||
def _inactive_message(result: brand_discovery.DiscoveryResult) -> str:
|
||||
return (
|
||||
f"{result.brand} is not in ACTIVE_BRANDS, so these products would be "
|
||||
f"written to {result.table} and then filtered out of /api/brands, "
|
||||
f"search, suggest and the category listing. The rows would be complete "
|
||||
f"and correct, just unreadable. To make them visible, set "
|
||||
f"'{_env_line(result.brand)}' in backend/.env and restart the API - "
|
||||
f"settings are read once at import, so a restart is required."
|
||||
)
|
||||
|
||||
|
||||
def _inactive_detail(brand: str) -> str:
|
||||
return (
|
||||
f"{brand} is not in ACTIVE_BRANDS. Ingesting it now would write a "
|
||||
f"complete catalog that no endpoint can read. Either set "
|
||||
f"'{_env_line(brand)}' in backend/.env and restart the API first, or "
|
||||
f"re-send with acknowledge_inactive=true to stage the data anyway - "
|
||||
f"adding the brand later is a config change and a restart, not a "
|
||||
f"re-ingest."
|
||||
)
|
||||
@@ -48,6 +48,20 @@ def generate_catalog(payload: CatalogGenerateRequest) -> CatalogJobOut:
|
||||
"""Kick off brand catalog ingestion (discovery -> images -> embeddings ->
|
||||
pgvector) as a background daemon thread and return immediately with a job id.
|
||||
|
||||
PREFER /api/admin/brand-discovery/* FOR NEW WORK. This route is unchanged
|
||||
and still supported, but it does not run the eleven stages in
|
||||
`app/core/store_catalog_pipeline.py` - no title validation, no pack-size
|
||||
explosion, no SKU resolution, no barcode, no HSN/GST, no validation gate -
|
||||
and it mints `image_id` with `s3_service.generate_image_id()`, which appends
|
||||
a random uuid4. Nothing it writes can ever match an existing row, so running
|
||||
it twice for one brand produces two catalogs. It also calls
|
||||
`upsert_brand_products(cleanup=True)`, which deletes every row not in the
|
||||
batch it just built.
|
||||
|
||||
The discovery routes do run all eleven stages, use a deterministic
|
||||
`image_id`, write with `cleanup=False`, and show the products for approval
|
||||
before anything is stored. See docs/BRAND_DISCOVERY.md.
|
||||
|
||||
NOTE: on an 8GB RAM / CPU-only machine, running ingestion (which loads
|
||||
the embeddings model and calls Ollama repeatedly) at the same time as
|
||||
heavy chat traffic will be slow. This is intended as an occasional
|
||||
|
||||
@@ -8,6 +8,18 @@ product discovery (Ollama), per-product image search + S3 upload,
|
||||
pricing/description enrichment, embedding generation, and the pgvector
|
||||
upsert. This module exists only to give that pipeline one clear, reusable
|
||||
entry point and a consistent result shape for callers.
|
||||
|
||||
NOT THE ELEVEN-STAGE PIPELINE, and prefer `app/services/brand_discovery.py`
|
||||
for new work. "Full pipeline" above means this module's own sequence, not the
|
||||
eleven stages in `app/core/store_catalog_pipeline.py`: there is no title
|
||||
validation, no pack-size explosion, no SKU service, no barcode, no HSN/GST and
|
||||
no validation gate here, and `image_id` carries a random uuid4 suffix, so a
|
||||
second run for the same brand cannot match the first and duplicates it.
|
||||
|
||||
This path is unchanged and still works. It also rewrites the brand's seed
|
||||
catalog via `brand_sync.export_brand_to_seed_file` (below), which the discovery
|
||||
path deliberately does not - so the two are not drop-in replacements for each
|
||||
other. See docs/BRAND_DISCOVERY.md.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
@@ -316,6 +316,37 @@ BRAND_SYNC_INTERVAL_SECONDS = int(os.getenv("BRAND_SYNC_INTERVAL_SECONDS", "300"
|
||||
# See app/services/active_brands.py.
|
||||
ACTIVE_BRANDS = os.getenv("ACTIVE_BRANDS", "")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Brand discovery (brand name -> the 11-stage pipeline)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Discovery turns a brand NAME into rows the ordinary catalog pipeline ingests.
|
||||
# See app/services/brand_discovery.py. Every value below has a working default,
|
||||
# so the feature needs no configuration to run.
|
||||
#
|
||||
# Open Food Facts is the primary source and the language model is the
|
||||
# supplement, not the reverse: OFF returns real products carrying a real GTIN,
|
||||
# while the default OLLAMA_MODEL_NAME (qwen2.5:1.5b) invents plausible ones that
|
||||
# nothing downstream can catch. Turning BRAND_DISCOVERY_USE_OFF off leaves the
|
||||
# result resting on the model alone.
|
||||
BRAND_DISCOVERY_USE_OFF = _bool("BRAND_DISCOVERY_USE_OFF", "true")
|
||||
BRAND_DISCOVERY_USE_LLM = _bool("BRAND_DISCOVERY_USE_LLM", "true")
|
||||
|
||||
# Products per discovery run. One CSV row per product; pack-size explosion
|
||||
# happens later in stage 4, so this is well inside the 2000-row per-file cap.
|
||||
BRAND_DISCOVERY_MAX_PRODUCTS = int(os.getenv("BRAND_DISCOVERY_MAX_PRODUCTS", "200"))
|
||||
|
||||
# Pack sizes kept per product when only the language model offers any. Stage 6
|
||||
# runs an image search per exploded row, so this multiplies the slowest part of
|
||||
# the run; 3 keeps a large brand inside a sane wall-clock.
|
||||
BRAND_DISCOVERY_MAX_SIZES = int(os.getenv("BRAND_DISCOVERY_MAX_SIZES", "3"))
|
||||
|
||||
# Wall-clock ceiling on the LLM half of a run, checked between prompts. Open
|
||||
# Food Facts runs first and is never subject to it, so a run that hits this
|
||||
# still returns the evidence-backed products.
|
||||
BRAND_DISCOVERY_DEADLINE_SECONDS = float(
|
||||
os.getenv("BRAND_DISCOVERY_DEADLINE_SECONDS", "300")
|
||||
)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# S3 / DigitalOcean Spaces (product image storage) - optional
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -31,7 +31,7 @@ from app.api.routers import health, brands, search, suggest, chat, catalog, syst
|
||||
from app.api.routers import stores, discounts, analytics as store_analytics, trending, recommendations, store_admin
|
||||
from app.api.routers import nutrition, nutrition_admin, upload
|
||||
from app.api.routers import auth, user_products, admin_train, mcp_info
|
||||
from app.api.routers import batch_catalog, uploads
|
||||
from app.api.routers import batch_catalog, uploads, brand_discovery
|
||||
from app.services.store_db import ensure_store_intelligence_schema
|
||||
from app.services.nutrition_db import ensure_nutrition_schema
|
||||
|
||||
@@ -307,6 +307,7 @@ app.include_router(nutrition_admin.router, prefix="/api")
|
||||
app.include_router(upload.router, prefix="/api")
|
||||
app.include_router(batch_catalog.router, prefix="/api")
|
||||
app.include_router(uploads.router, prefix="/api")
|
||||
app.include_router(brand_discovery.router, prefix="/api")
|
||||
app.include_router(mcp_info.router, prefix="/api")
|
||||
|
||||
# MCP lives outside /api on purpose: it is a protocol endpoint for AI clients,
|
||||
|
||||
1041
app/services/brand_discovery.py
Normal file
1041
app/services/brand_discovery.py
Normal file
File diff suppressed because it is too large
Load Diff
2534
data/cache/off_brand_corpus/britannia.json
vendored
Normal file
2534
data/cache/off_brand_corpus/britannia.json
vendored
Normal file
File diff suppressed because it is too large
Load Diff
@@ -1,21 +1,254 @@
|
||||
{
|
||||
"_sku_sequences": {
|
||||
"BRITAN-GOO-100": 10,
|
||||
"BRITAN-GOO-200": 12,
|
||||
"BRITAN-GOO-100": 97,
|
||||
"BRITAN-GOO-200": 153,
|
||||
"BRITAN-GOO-500": 9,
|
||||
"BRITAN-GOO-250": 1,
|
||||
"BRITAN-MAR-100": 2,
|
||||
"BRITAN-MAR-250": 2,
|
||||
"BRITAN-MAR-500": 1,
|
||||
"BRITAN-MIL-200": 2,
|
||||
"BRITAN-MIL-500": 1,
|
||||
"BRITAN-MIL-1": 1,
|
||||
"BRITAN-GOO-375": 1,
|
||||
"BRITAN-MIL-200": 58,
|
||||
"BRITAN-MIL-500": 42,
|
||||
"BRITAN-MIL-1": 28,
|
||||
"BRITAN-GOO-375": 55,
|
||||
"BRITAN-MAR-200": 1,
|
||||
"BRITAN-MAR-375": 1,
|
||||
"BRITAN-MIL-100": 1,
|
||||
"BRITAN-MIL-375": 1,
|
||||
"BRITAN-MIL-100": 58,
|
||||
"BRITAN-MIL-375": 28,
|
||||
"BRITAN-MIL-150": 1,
|
||||
"BRITAN-50-200": 1
|
||||
"BRITAN-50-200": 1,
|
||||
"BRITAN-100-450": 29,
|
||||
"BRITAN-50-76": 29,
|
||||
"BRITAN-50-100": 29,
|
||||
"BRITAN-50-250": 27,
|
||||
"BRITAN-50-500": 27,
|
||||
"BRITAN-50-110": 29,
|
||||
"BRITAN-50-1": 58,
|
||||
"BRITAN-50-455": 29,
|
||||
"BRITAN-505-40": 1,
|
||||
"BRITAN-505-80": 1,
|
||||
"BRITAN-505-150": 1,
|
||||
"BRITAN-505-200": 29,
|
||||
"BRITAN-505-500": 53,
|
||||
"BRITAN-505-1": 27,
|
||||
"BRITAN-ATT-200": 29,
|
||||
"BRITAN-ATT-400": 27,
|
||||
"BRITAN-ATT-600": 27,
|
||||
"BRITAN-BIS-295": 29,
|
||||
"BRITAN-BIS-100": 29,
|
||||
"BRITAN-BOU-50": 29,
|
||||
"BRITAN-BR-78": 29,
|
||||
"BRITAN-BRI-700": 29,
|
||||
"BRITAN-BRI-100": 142,
|
||||
"BRITAN-BRI-250": 105,
|
||||
"BRITAN-BRI-400": 27,
|
||||
"BRITAN-BRI-40": 32,
|
||||
"BRITAN-BRI-50": 29,
|
||||
"BRITAN-BRI-80": 3,
|
||||
"BRITAN-BRI-150": 3,
|
||||
"BRITAN-50-63": 29,
|
||||
"BRITAN-505-692": 29,
|
||||
"BRITAN-505-300": 29,
|
||||
"BRITAN-505-29": 17,
|
||||
"BRITAN-BOU-40": 1,
|
||||
"BRITAN-BOU-80": 1,
|
||||
"BRITAN-BOU-150": 30,
|
||||
"BRITAN-BRE-250": 29,
|
||||
"BRITAN-CAK-120": 29,
|
||||
"BRITAN-CAK-50": 29,
|
||||
"BRITAN-CAK-35": 29,
|
||||
"BRITAN-CAK-110": 29,
|
||||
"BRITAN-CAK-27": 29,
|
||||
"BRITAN-CHE-250": 29,
|
||||
"BRITAN-CHE-200": 87,
|
||||
"BRITAN-CHE-500": 81,
|
||||
"BRITAN-CHE-1": 81,
|
||||
"BRITAN-COW-1": 29,
|
||||
"BRITAN-FUD-120": 29,
|
||||
"BRITAN-GD-600": 29,
|
||||
"BRITAN-GOL-40": 1,
|
||||
"BRITAN-GOL-80": 1,
|
||||
"BRITAN-GOL-150": 1,
|
||||
"BRITAN-GOO-60": 58,
|
||||
"BRITAN-GOO-111": 29,
|
||||
"BRITAN-GOO-400": 29,
|
||||
"BRITAN-MAR-1": 29,
|
||||
"BRITAN-MIB-300": 29,
|
||||
"BRITAN-NIC-594": 29,
|
||||
"BRITAN-NIC-143": 29,
|
||||
"BRITAN-NUT-100": 129,
|
||||
"BRITAN-NUT-75": 101,
|
||||
"BRITAN-PAT-40": 1,
|
||||
"BRITAN-PAT-80": 1,
|
||||
"BRITAN-PAT-150": 1,
|
||||
"BRITAN-PAV-200": 29,
|
||||
"BRITAN-THI-150": 29,
|
||||
"BRITAN-TIM-40": 2,
|
||||
"BRITAN-TIM-80": 2,
|
||||
"BRITAN-TIM-150": 2,
|
||||
"BRITAN-TOA-250": 17,
|
||||
"BRITAN-TRA-40": 1,
|
||||
"BRITAN-TRA-80": 1,
|
||||
"BRITAN-TRA-150": 1,
|
||||
"BRITAN-TRE-60": 58,
|
||||
"BRITAN-TRE-45": 29,
|
||||
"BRITAN-TRE-47": 29,
|
||||
"BRITAN-TRE-40": 5,
|
||||
"BRITAN-TRE-80": 5,
|
||||
"BRITAN-TRE-150": 5,
|
||||
"BRITAN-TRE-51": 29,
|
||||
"BRITAN-VIT-400": 85,
|
||||
"BRITAN-WIN-180": 29,
|
||||
"BRITAN-WIN-40": 5,
|
||||
"BRITAN-WIN-80": 5,
|
||||
"BRITAN-WIN-150": 5,
|
||||
"BRITAN-BRO-450": 29,
|
||||
"BRITAN-BRT-200": 29,
|
||||
"BRITAN-BRT-500": 27,
|
||||
"BRITAN-BRT-1": 27,
|
||||
"BRITAN-CAK-64": 29,
|
||||
"BRITAN-CAK-60": 29,
|
||||
"BRITAN-CHO-20": 29,
|
||||
"BRITAN-CHO-55": 27,
|
||||
"BRITAN-CHO-150": 27,
|
||||
"BRITAN-CRA-100": 29,
|
||||
"BRITAN-DAI-1": 29,
|
||||
"BRITAN-GOB-110": 29,
|
||||
"BRITAN-GOB-50": 58,
|
||||
"BRITAN-GOO-600": 57,
|
||||
"BRITAN-GOO-120": 58,
|
||||
"BRITAN-GOO-39": 29,
|
||||
"BRITAN-HAR-40": 1,
|
||||
"BRITAN-HAR-80": 1,
|
||||
"BRITAN-HAR-150": 1,
|
||||
"BRITAN-JIM-25": 29,
|
||||
"BRITAN-JIM-685": 29,
|
||||
"BRITAN-JIM-40": 1,
|
||||
"BRITAN-JIM-80": 1,
|
||||
"BRITAN-JIM-150": 1,
|
||||
"BRITAN-JIM-460": 29,
|
||||
"BRITAN-JIM-70": 29,
|
||||
"BRITAN-LIT-13": 29,
|
||||
"BRITAN-LIT-75": 29,
|
||||
"BRITAN-MAR-117": 43,
|
||||
"BRITAN-MIL-115": 29,
|
||||
"BRITAN-MUF-100": 29,
|
||||
"BRITAN-MUF-250": 27,
|
||||
"BRITAN-MUF-400": 27,
|
||||
"BRITAN-MUL-400": 29,
|
||||
"BRITAN-NUT-120": 29,
|
||||
"BRITAN-NUT-300": 56,
|
||||
"BRITAN-NUT-50": 29,
|
||||
"BRITAN-NUT-200": 41,
|
||||
"BRITAN-NUT-150": 30,
|
||||
"BRITAN-NUT-40": 1,
|
||||
"BRITAN-NUT-80": 1,
|
||||
"BRITAN-POT-40": 1,
|
||||
"BRITAN-POT-80": 1,
|
||||
"BRITAN-POT-150": 1,
|
||||
"BRITAN-PUR-300": 29,
|
||||
"BRITAN-PUR-75": 29,
|
||||
"BRITAN-RAG-100": 29,
|
||||
"BRITAN-RAG-200": 27,
|
||||
"BRITAN-RAG-375": 27,
|
||||
"BRITAN-ROL-40": 1,
|
||||
"BRITAN-ROL-80": 1,
|
||||
"BRITAN-ROL-150": 1,
|
||||
"BRITAN-SAN-450": 29,
|
||||
"BRITAN-SUG-100": 29,
|
||||
"BRITAN-SUG-200": 27,
|
||||
"BRITAN-SUG-300": 27,
|
||||
"BRITAN-SWE-1": 29,
|
||||
"BRITAN-LAU-120": 29,
|
||||
"BRITAN-TIG-150": 30,
|
||||
"BRITAN-TIG-400": 29,
|
||||
"BRITAN-TIG-100": 57,
|
||||
"BRITAN-TIG-200": 27,
|
||||
"BRITAN-TIG-375": 27,
|
||||
"BRITAN-TIG-497": 29,
|
||||
"BRITAN-TIG-40": 1,
|
||||
"BRITAN-TIG-80": 1,
|
||||
"BRITAN-TIG-63": 29,
|
||||
"BRITAN-TIG-75": 29,
|
||||
"BRITAN-TOA-180": 28,
|
||||
"BRITAN-TOA-300": 27,
|
||||
"BRITAN-TOA-600": 27,
|
||||
"BRITAN-TOA-200": 41,
|
||||
"BRITAN-TOA-275": 29,
|
||||
"BRITAN-TRE-14": 29,
|
||||
"BRITAN-TRE-55": 29,
|
||||
"BRITAN-TRE-75": 29,
|
||||
"BRITAN-TRE-100": 167,
|
||||
"BRITAN-TRE-125": 27,
|
||||
"BRITAN-VIT-40": 1,
|
||||
"BRITAN-VIT-80": 1,
|
||||
"BRITAN-VIT-150": 1,
|
||||
"BRITAN-VIT-200": 29,
|
||||
"BRITAN-VIT-600": 27,
|
||||
"BRITAN-WIN-200": 29,
|
||||
"BRITAN-WIN-500": 157,
|
||||
"BRITAN-WIN-1": 27,
|
||||
"BRITAN-505-100": 28,
|
||||
"BRITAN-505-250": 26,
|
||||
"BRITAN-BRI-500": 78,
|
||||
"BRITAN-BOU-100": 42,
|
||||
"BRITAN-BOU-250": 26,
|
||||
"BRITAN-BOU-500": 26,
|
||||
"BRITAN-GOL-100": 28,
|
||||
"BRITAN-GOL-250": 26,
|
||||
"BRITAN-GOL-500": 26,
|
||||
"BRITAN-PAT-100": 28,
|
||||
"BRITAN-PAT-250": 26,
|
||||
"BRITAN-PAT-500": 26,
|
||||
"BRITAN-TIM-100": 56,
|
||||
"BRITAN-TIM-250": 52,
|
||||
"BRITAN-TIM-500": 52,
|
||||
"BRITAN-TRA-100": 28,
|
||||
"BRITAN-TRA-250": 26,
|
||||
"BRITAN-TRA-500": 26,
|
||||
"BRITAN-TRE-250": 130,
|
||||
"BRITAN-TRE-500": 130,
|
||||
"BRITAN-WIN-100": 140,
|
||||
"BRITAN-WIN-250": 130,
|
||||
"BRITAN-HAR-100": 28,
|
||||
"BRITAN-HAR-250": 26,
|
||||
"BRITAN-HAR-500": 26,
|
||||
"BRITAN-JIM-100": 1,
|
||||
"BRITAN-JIM-250": 1,
|
||||
"BRITAN-JIM-500": 1,
|
||||
"BRITAN-NUT-250": 26,
|
||||
"BRITAN-NUT-500": 26,
|
||||
"BRITAN-POT-100": 28,
|
||||
"BRITAN-POT-250": 26,
|
||||
"BRITAN-POT-500": 26,
|
||||
"BRITAN-ROL-100": 28,
|
||||
"BRITAN-ROL-250": 26,
|
||||
"BRITAN-ROL-500": 26,
|
||||
"BRITAN-TIG-250": 26,
|
||||
"BRITAN-TIG-500": 26,
|
||||
"BRITAN-VIT-100": 28,
|
||||
"BRITAN-VIT-250": 26,
|
||||
"BRITAN-VIT-500": 26,
|
||||
"BRITAN-JIM-92": 27,
|
||||
"BRITAN-505-715": 12,
|
||||
"BRITAN-50-50": 14,
|
||||
"BRITAN-505-38": 14,
|
||||
"BRITAN-BOU-60": 14,
|
||||
"BRITAN-BOU-120": 14,
|
||||
"BRITAN-BRO-400": 14,
|
||||
"BRITAN-GOO-68": 14,
|
||||
"BRITAN-JIM-57": 14,
|
||||
"BRITAN-JIM-138": 14,
|
||||
"BRITAN-JIM-350": 14,
|
||||
"BRITAN-MAR-39": 14,
|
||||
"BRITAN-MAR-89": 14,
|
||||
"BRITAN-MAR-300": 13,
|
||||
"BRITAN-MAR-73": 14,
|
||||
"BRITAN-MIL-335": 14,
|
||||
"BRITAN-MIL-67": 14,
|
||||
"BRITAN-MUL-450": 14,
|
||||
"BRITAN-NUT-52": 13,
|
||||
"BRITAN-NUT-1": 14,
|
||||
"BRITAN-TOA-273": 14
|
||||
}
|
||||
}
|
||||
551
tests/test_brand_discovery.py
Normal file
551
tests/test_brand_discovery.py
Normal file
@@ -0,0 +1,551 @@
|
||||
"""Tests for brand discovery - the brand-name -> 11-stage-pipeline bridge.
|
||||
|
||||
Follows the pattern in test_store_catalog_pipeline.py: monkeypatch the network
|
||||
and storage boundary *on the module object* (brand_discovery imports those names
|
||||
directly), and assert on the rows that come out rather than on a status code.
|
||||
|
||||
Nothing here reaches Open Food Facts, Ollama or a database. The two source
|
||||
functions are stubbed and every pipeline run uses `use_llm=False,
|
||||
fetch_images=False`.
|
||||
|
||||
Several of these tests pin behaviour that was WRONG in the first working
|
||||
version of this module and was found by round-tripping the real Britannia
|
||||
corpus. They are regression tests with a known failure, not speculative ones -
|
||||
each names the defect it prevents.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.api.routers.user_products import map_spreadsheet_columns, row_to_request
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services import brand_discovery as bd
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fixtures
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
"""Keep the SKU sequence counter out of the repo - see the same fixture in
|
||||
test_store_catalog_pipeline.py."""
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_network(monkeypatch):
|
||||
"""Neither source may reach the outside world by default.
|
||||
|
||||
A test that wants products stubs one of these explicitly. Without this an
|
||||
accidental real call would hit Open Food Facts from the suite.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [])
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [])
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [])
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""A fake brand table. Returns the dict of image_id -> stored row."""
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
|
||||
for row in rows:
|
||||
table[row["image_id"]] = row
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand",
|
||||
lambda brand, **kw: list(table.values()))
|
||||
monkeypatch.setattr(pipeline, "embed_texts",
|
||||
lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
|
||||
|
||||
def _off(title, *, code=None, quantity=None):
|
||||
"""One Open Food Facts corpus hit, in the shape `_from_open_facts` returns."""
|
||||
return {"title": title, "barcode": code, "size": bd._canonical_size(quantity),
|
||||
"source": "off"}
|
||||
|
||||
|
||||
def _run(products, filename="discovered.csv"):
|
||||
"""Discovered products -> CSV -> the real 11 stages."""
|
||||
return pipeline.run_pipeline(filename, bd.rows_to_csv_bytes(products),
|
||||
use_llm=False, fetch_images=False)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The contract the whole CSV bridge rests on
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_the_csv_headers_all_map_onto_catalog_fields():
|
||||
"""Every emitted header must be understood by the real column mapper.
|
||||
|
||||
This is THE contract: discovery writes a spreadsheet and the pipeline reads
|
||||
it with `map_spreadsheet_columns`. A header that does not resolve is dropped
|
||||
in silence, so the column simply never arrives and the run still reports
|
||||
success. Asserting it here means a rename on either side fails loudly.
|
||||
"""
|
||||
mapping = map_spreadsheet_columns(bd.CSV_HEADERS)
|
||||
|
||||
assert mapping.unrecognised == []
|
||||
assert mapping.ignored == []
|
||||
for header in bd.CSV_HEADERS:
|
||||
assert header in mapping.columns, f"{header!r} did not resolve to a field"
|
||||
|
||||
|
||||
def test_the_emitted_csv_parses_with_the_real_spreadsheet_reader(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts",
|
||||
lambda brand, refresh=False: [_off("Marie Gold", quantity="250 g")])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
df, mapping = pipeline.parse_spreadsheet("d.csv", bd.rows_to_csv_bytes(result.products))
|
||||
|
||||
assert len(df) == 1
|
||||
assert mapping.unrecognised == []
|
||||
request = row_to_request(df.to_dict(orient="records")[0], mapping)
|
||||
assert request.brand == "Britannia"
|
||||
assert request.product_name == "Marie Gold"
|
||||
assert request.size_variants == ["250g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: list cells were shredded by the separator
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_list_item_containing_a_separator_survives_the_round_trip():
|
||||
"""`_string_list` splits on [,;|], and the highlight generator emits commas.
|
||||
|
||||
"Available in 3 sizes: 100g, 250g, 500g" and "Baked, Not Fried" became five
|
||||
fragments instead of two highlights. The damage is invisible in the database
|
||||
- the column is populated, just wrong - so it needs a test.
|
||||
"""
|
||||
cell = bd._safe_list_cell([
|
||||
"Available in 3 sizes: 100g, 250g, 500g",
|
||||
"Baked, Not Fried",
|
||||
"Pipe | separated | too",
|
||||
])
|
||||
mapping = map_spreadsheet_columns(("highlights",))
|
||||
from app.api.routers.user_products import _string_list
|
||||
|
||||
recovered = _string_list({"highlights": cell}, mapping, "highlights")
|
||||
|
||||
assert len(recovered) == 3
|
||||
assert recovered[0] == "Available in 3 sizes: 100g / 250g / 500g"
|
||||
assert recovered[1] == "Baked / Not Fried"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: pack sizes doubled into the image_id
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.mark.parametrize("raw, expected", [
|
||||
("200 g", "200g"),
|
||||
("200g", "200g"),
|
||||
("1.5 kg", "1.5kg"),
|
||||
("75 g", "75g"),
|
||||
])
|
||||
def test_a_spaced_pack_size_is_canonicalised(raw, expected):
|
||||
"""OFF writes both "100g" and "100 g" for the same brand.
|
||||
|
||||
`build_image_id` slugifies "200 g" to "200_g", which is not a substring of a
|
||||
name containing "200g", so the size is appended anyway and the same product
|
||||
lands on two permanent rows depending on which spelling was discovered.
|
||||
"""
|
||||
assert bd._canonical_size(raw) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize("junk", ["India", "12", "200", "", None, "6"])
|
||||
def test_a_quantity_that_is_not_a_pack_size_is_rejected(junk):
|
||||
"""The Britannia corpus carries "India" and bare counts in `quantity`."""
|
||||
assert bd._canonical_size(junk) is None
|
||||
|
||||
|
||||
def test_a_size_glued_to_a_letter_in_the_title_is_separated():
|
||||
"""OFF holds "Jim Jam92 g", where there is no word boundary before the 92.
|
||||
|
||||
`off_bulk._SIZE_RE` anchors on one and so cannot see it, which left the
|
||||
unnormalised size in the product name and produced the image_id
|
||||
`britannia_jim_jam92_g_40g`.
|
||||
"""
|
||||
assert bd._canonicalise_title_size("Jim Jam92 g") == "Jim Jam 92g"
|
||||
assert bd._canonicalise_title_size("Britannia Good Day 200 g") == "Britannia Good Day 200g"
|
||||
|
||||
|
||||
def test_no_stored_image_id_carries_the_pack_size_twice(store):
|
||||
result = _run([
|
||||
bd.DiscoveredProduct(brand="Britannia", product_name="Jim Jam",
|
||||
title="Jim Jam", category="", category_hint="Biscuits & Cookies",
|
||||
description="", size_variants=["92g"]),
|
||||
])
|
||||
assert result.inserted == 1
|
||||
assert list(store) == ["britannia_jim_jam_92g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: the pack size belongs in ONE place
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_size_in_the_title_is_not_applied_twice(monkeypatch):
|
||||
"""OFF holds "Britannia Toastea 200g" with a `quantity` of "250g".
|
||||
|
||||
Taking the quantity and keeping the title produced the product name
|
||||
"Britannia Toastea 200g 250g" - two sizes, one name, a pack that does not
|
||||
exist. The title's own size wins and is removed from the name.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.product_name == "Britannia Toastea"
|
||||
assert product.size_variants == ["200g"]
|
||||
|
||||
|
||||
def test_a_case_pack_count_is_not_part_of_the_product_name(monkeypatch):
|
||||
"""OFF titles carry carton counts: "Good Day Butter Cookies (25)".
|
||||
|
||||
That is how many units ship in a box, not part of what the product is
|
||||
called, and leaving it in put the carton count into the image_id.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Good Day Butter Cookies (25)", quantity="60 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.product_name == "Good Day Butter Cookies"
|
||||
assert product.size_variants == ["60g"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: one GTIN, one pack
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_barcode_is_dropped_when_no_real_pack_size_is_known(monkeypatch):
|
||||
"""A GTIN identifies one pack, and stage 4 invents three when it has none.
|
||||
|
||||
Every column is copied into each exploded variant, so a surviving barcode
|
||||
would be stamped onto two packs that do not exist - wrong data that looks
|
||||
authoritative.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", code="8901063017702"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.size_variants == []
|
||||
assert product.barcode is None
|
||||
assert any("barcode dropped" in note for note in product.notes)
|
||||
|
||||
|
||||
def test_a_barcode_survives_when_the_pack_size_is_real(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.barcode == "8901063012516"
|
||||
assert product.size_variants == ["100g"]
|
||||
|
||||
|
||||
def test_no_gtin_is_written_to_more_than_one_row(store, monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
|
||||
_off("50 50 Gol Maal", code="8901063017702"),
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
_run(result.products)
|
||||
|
||||
barcodes = [row["barcode"] for row in store.values() if row["barcode"]]
|
||||
assert len(barcodes) == len(set(barcodes))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: the generated description poisoned the category and the tax code
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_no_category_is_written_when_no_source_stated_one(monkeypatch):
|
||||
"""Stage 3 owns category detection, and treats a supplied value as final.
|
||||
|
||||
It sets `_category_deterministic` and then rewrites the title against the
|
||||
category, so a wrong guess here does not merely mislabel a row - it renames
|
||||
the product.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", quantity="100 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.category == ""
|
||||
assert product.category_hint # still reported, for the preview
|
||||
|
||||
|
||||
def test_a_stated_category_is_carried_through(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Good Day Cashew", "category": "Biscuits & Cookies",
|
||||
"description": "Cashew cookies", "sizes": ["75g"], "providers": [],
|
||||
"source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", require_evidence=False)
|
||||
|
||||
assert result.products[0].category == "Biscuits & Cookies"
|
||||
|
||||
|
||||
def test_no_boilerplate_description_is_generated(monkeypatch):
|
||||
"""`generate_detailed_description` contains the word "taste".
|
||||
|
||||
`detect_category_from_text` matches keywords fuzzily and "taste" is one edit
|
||||
from the Oral Care keyword "paste", so every weakly-titled product was
|
||||
classified Oral Care and stage 9 stamped it with HSN 3306 - the tax code for
|
||||
dentifrices. A blank description is honest, and `_to_storage_row` already
|
||||
substitutes "<name> <size> from <brand>."
|
||||
"""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", quantity="100 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.description == ""
|
||||
assert "taste" not in (product.description or "").lower()
|
||||
|
||||
|
||||
def test_a_weakly_titled_product_is_not_classified_as_oral_care(store, monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("50 50 Gol Maal", quantity="100 g"),
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
_run(result.products)
|
||||
|
||||
stored = next(iter(store.values()))
|
||||
assert stored["category"] != "Oral Care"
|
||||
assert stored["hsn_code"] != "3306"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Regression: catalogue growth by re-run
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_a_known_product_is_reemitted_under_its_stored_name(monkeypatch):
|
||||
"""A brand prefix appearing or disappearing is the realistic drift.
|
||||
|
||||
The stored name has no brand prefix and the discovered one does; both
|
||||
normalise identically once `brand_tokens` are stripped, so the row folds and
|
||||
the image_id stays put.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
|
||||
{"product_name": "Good Day Cashew Cookies 75g"},
|
||||
])
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Britannia Good Day Cashew Cookies", quantity="75 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.matches_existing == "Good Day Cashew Cookies"
|
||||
assert product.product_name == "Good Day Cashew Cookies"
|
||||
|
||||
|
||||
def test_a_shorter_title_does_not_fold_onto_a_longer_stored_one(monkeypatch):
|
||||
"""The documented limit of the 0.85 threshold, pinned so it is a decision
|
||||
rather than a surprise.
|
||||
|
||||
"Good Day Cashew" scores 0.75 against "Good Day Cashew Cookies" and stays a
|
||||
separate product. Relaxing the threshold far enough to fold it would also
|
||||
fold "Dairy Milk Silk" onto "Dairy Milk Silk Minis", which is a different
|
||||
product - see `_same_product`.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
|
||||
{"product_name": "Good Day Cashew Cookies 75g"},
|
||||
])
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Good Day Cashew", quantity="75 g"),
|
||||
])
|
||||
|
||||
product = bd.discover_brand_products("Britannia", use_llm=False).products[0]
|
||||
|
||||
assert product.matches_existing is None
|
||||
|
||||
|
||||
def test_two_real_packs_sharing_a_name_stay_on_their_own_rows(monkeypatch):
|
||||
"""OFF holds "Jim Jam" at 25g and "Jim jam" at 92g.
|
||||
|
||||
Both normalise to the same size-free key. Re-emitting the first stored
|
||||
DISPLAY name gave the 92g pack the 25g name, which `_to_storage_row` then
|
||||
extended to "Jim Jam 25g 92g" under a brand-new image_id. The index hands
|
||||
back the base name so the size can be re-applied per pack.
|
||||
"""
|
||||
monkeypatch.setattr(bd, "get_products_by_brand", lambda brand, **kw: [
|
||||
{"product_name": "Jim Jam 25g"},
|
||||
{"product_name": "Jim jam 92g"},
|
||||
])
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Jim Jam", quantity="25 g"),
|
||||
_off("Jim jam", quantity="92 g"),
|
||||
])
|
||||
|
||||
products = bd.discover_brand_products("Britannia", use_llm=False).products
|
||||
names = {p.product_name for p in products}
|
||||
|
||||
assert names == {"Jim Jam"}
|
||||
assert sorted(s for p in products for s in p.size_variants) == ["25g", "92g"]
|
||||
assert not any("25g 92g" in p.product_name for p in products)
|
||||
|
||||
|
||||
def test_re_running_discovery_and_ingest_writes_nothing_new(store, monkeypatch):
|
||||
"""The whole feature's safety property, end to end.
|
||||
|
||||
Deterministic image_id + fill-only-blanks + `cleanup=False` make a re-run a
|
||||
no-op, but only if discovery re-emits the same names. This exercises the
|
||||
real pipeline twice with the catalog fed back in between.
|
||||
"""
|
||||
corpus = [
|
||||
_off("Britannia Toastea 200g", code="8901063342934", quantity="250 g"),
|
||||
_off("Good Day Butter Cookies (25)", quantity="60 g"),
|
||||
_off("Jim Jam92 g", code="8901063019027"),
|
||||
_off("50 50 Gol Maal", code="8901063017702"),
|
||||
]
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: corpus)
|
||||
monkeypatch.setattr(bd, "get_products_by_brand",
|
||||
lambda brand, **kw: list(store.values()))
|
||||
|
||||
first = _run(bd.discover_brand_products("Britannia", use_llm=False).products)
|
||||
after_first = dict(store)
|
||||
assert first.inserted > 0
|
||||
|
||||
second = _run(bd.discover_brand_products("Britannia", use_llm=False).products)
|
||||
|
||||
assert second.inserted == 0
|
||||
assert second.backfilled == 0
|
||||
assert second.skipped_existing == len(after_first)
|
||||
assert set(store) == set(after_first)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Sources, evidence and scoring
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_open_food_facts_results_survive_an_unreachable_ollama(monkeypatch):
|
||||
"""OFF runs first and unconditionally, so a dead LLM degrades the result
|
||||
rather than emptying it - the common case on a CPU-only box."""
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Marie Gold", code="8901063014206", quantity="250 g"),
|
||||
])
|
||||
monkeypatch.setattr(bd.ollama_service, "_generate", lambda *a, **k: "")
|
||||
monkeypatch.setattr(bd.ollama_service, "get_categories_for_brand", lambda brand: [])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=True)
|
||||
|
||||
assert len(result.products) == 1
|
||||
assert result.products[0].sources == ["off"]
|
||||
assert any("language model returned no products" in w for w in result.warnings)
|
||||
|
||||
|
||||
def test_a_product_found_by_both_sources_is_one_row_and_scores_highest(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Marie Gold", code="8901063014206", quantity="250 g"),
|
||||
])
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Marie Gold", "category": "Biscuits & Cookies",
|
||||
"description": "Tea-time biscuit", "sizes": ["250g"],
|
||||
"providers": ["Amazon"], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia")
|
||||
|
||||
assert len(result.products) == 1
|
||||
product = result.products[0]
|
||||
assert sorted(product.sources) == ["llm", "off"]
|
||||
assert product.confidence == 1.0
|
||||
assert result.counts["corroborated"] == 1
|
||||
|
||||
|
||||
def test_an_uncorroborated_llm_product_is_dropped_when_evidence_is_required(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Britannia Quantum Wafer", "category": None, "description": None,
|
||||
"sizes": [], "providers": [], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", require_evidence=True)
|
||||
|
||||
assert result.products == []
|
||||
assert result.counts["dropped_without_evidence"] == 1
|
||||
|
||||
|
||||
def test_an_uncorroborated_llm_product_is_kept_but_unticked_when_evidence_is_optional(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Britannia Quantum Wafer", "category": None, "description": None,
|
||||
"sizes": [], "providers": [], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", require_evidence=False)
|
||||
|
||||
assert len(result.products) == 1
|
||||
product = result.products[0]
|
||||
assert product.evidence is None
|
||||
assert product.confidence < 0.5
|
||||
assert product.as_preview()["selected"] is False
|
||||
|
||||
|
||||
def test_a_sub_brand_match_counts_as_registry_evidence(monkeypatch):
|
||||
""""Good Day" is a registered Britannia sub-brand, so an LLM row naming it
|
||||
is grounded without needing Open Food Facts."""
|
||||
monkeypatch.setattr(bd, "_from_llm", lambda brand, deadline, budget: [
|
||||
{"title": "Good Day Chocochip", "category": None, "description": None,
|
||||
"sizes": ["100g"], "providers": [], "source": "llm"},
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", require_evidence=True)
|
||||
|
||||
assert len(result.products) == 1
|
||||
assert result.products[0].evidence == "registry"
|
||||
|
||||
|
||||
def test_discovery_stops_at_max_products(monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off(f"Product {n}", quantity="100 g") for n in range(50)
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", max_products=10, use_llm=False)
|
||||
|
||||
assert len(result.products) == 10
|
||||
|
||||
|
||||
def test_the_brand_table_and_active_state_are_reported(monkeypatch):
|
||||
monkeypatch.setattr(bd.active_brands, "is_active_brand", lambda brand: False)
|
||||
monkeypatch.setattr(bd.active_brands, "filtering_enabled", lambda: True)
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
|
||||
assert result.table == "brand_britannia"
|
||||
assert result.parent_brand == "britannia"
|
||||
assert result.brand_active is False
|
||||
assert result.filtering_enabled is True
|
||||
|
||||
|
||||
def test_an_empty_brand_name_is_refused():
|
||||
with pytest.raises(ValueError):
|
||||
bd.discover_brand_products(" ")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Field completeness through the real stages
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_every_column_the_pipeline_can_fill_is_filled(store, monkeypatch):
|
||||
monkeypatch.setattr(bd, "_from_open_facts", lambda brand, refresh=False: [
|
||||
_off("Milk Bikis", code="8901063012516", quantity="100 g"),
|
||||
])
|
||||
|
||||
result = bd.discover_brand_products("Britannia", use_llm=False)
|
||||
_run(result.products)
|
||||
|
||||
row = next(iter(store.values()))
|
||||
for column in ("product_name", "title", "description", "category", "image_id",
|
||||
"price_range", "size_variants", "providers", "highlights",
|
||||
"nutrients", "fssai_license", "product_sku", "sku_source",
|
||||
"search_query"):
|
||||
assert row.get(column), f"{column} was left empty"
|
||||
assert row["barcode"] == "8901063012516"
|
||||
assert row["fssai_license"] == "10012022000103"
|
||||
361
tests/test_brand_discovery_api.py
Normal file
361
tests/test_brand_discovery_api.py
Normal file
@@ -0,0 +1,361 @@
|
||||
"""HTTP tests for /api/admin/brand-discovery/*.
|
||||
|
||||
Reuses the fixture set from test_batch_catalog_ingest.py verbatim - in
|
||||
particular `batch_root` and `no_background_worker`, whose absence once wrote
|
||||
batch manifests into the repository's own data directory. Read that fixture's
|
||||
docstring before removing either from a test here.
|
||||
|
||||
Discovery itself is stubbed at the module boundary; the point of this file is
|
||||
the routes, the ACTIVE_BRANDS gate and the staging handoff, not the merge logic
|
||||
(which test_brand_discovery.py covers).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.api import batch_common
|
||||
from app.core import batch_ingest
|
||||
from app.core import store_catalog_pipeline as pipeline
|
||||
from app.services import brand_discovery as bd
|
||||
|
||||
PREVIEW = "/api/admin/brand-discovery/preview"
|
||||
INGEST = "/api/admin/brand-discovery/ingest"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fixtures - see test_batch_catalog_ingest.py for the rationale behind each
|
||||
# ---------------------------------------------------------------------------
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_sku_counter(tmp_path, monkeypatch):
|
||||
from app.services import sku_service
|
||||
monkeypatch.setattr(sku_service, "_data_dir", tmp_path / "sku_sequences")
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def batch_root(tmp_path, monkeypatch):
|
||||
root = tmp_path / "batch_uploads"
|
||||
monkeypatch.setattr(batch_ingest, "BATCH_UPLOAD_DIR", root)
|
||||
return root
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def no_background_worker(monkeypatch):
|
||||
"""Stub the worker. Its absence once wrote manifests into the working tree -
|
||||
the long docstring in test_batch_catalog_ingest.py explains how."""
|
||||
from app.core import batch_worker
|
||||
|
||||
submitted: list = []
|
||||
monkeypatch.setattr(batch_worker, "submit", submitted.append)
|
||||
return submitted
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clean_job_store():
|
||||
"""The job store is a module singleton and outlives a test - see the same
|
||||
fixture in test_uploads_api.py."""
|
||||
from app.api.batch_job_store import batch_job_store
|
||||
batch_job_store._batches.clear()
|
||||
batch_job_store._cancelled.clear()
|
||||
yield
|
||||
batch_job_store._batches.clear()
|
||||
batch_job_store._cancelled.clear()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _active_brands_allow_everything(monkeypatch):
|
||||
"""Filtering off by default, so only the tests that are about the
|
||||
ACTIVE_BRANDS gate have to think about it."""
|
||||
from app.services import active_brands
|
||||
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: False)
|
||||
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: True)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def discovered(monkeypatch):
|
||||
"""A fixed two-product discovery result."""
|
||||
def fake_discover(brand, **kwargs):
|
||||
return bd.DiscoveryResult(
|
||||
brand=brand,
|
||||
parent_brand="britannia",
|
||||
table="brand_britannia",
|
||||
brand_active=True,
|
||||
filtering_enabled=False,
|
||||
products=[
|
||||
bd.DiscoveredProduct(
|
||||
brand=brand, product_name="Marie Gold", title="Marie Gold",
|
||||
category="", category_hint="Biscuits & Cookies", description="",
|
||||
size_variants=["250g"], barcode="8901063014206",
|
||||
sources=["off"], evidence="openfacts", confidence=0.9,
|
||||
),
|
||||
bd.DiscoveredProduct(
|
||||
brand=brand, product_name="Milk Bikis", title="Milk Bikis",
|
||||
category="", category_hint="Biscuits & Cookies", description="",
|
||||
size_variants=["100g"], barcode="8901063012516",
|
||||
sources=["off"], evidence="openfacts", confidence=0.9,
|
||||
),
|
||||
],
|
||||
counts={"discovered": 2},
|
||||
)
|
||||
|
||||
from app.api.routers import brand_discovery as router_module
|
||||
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products",
|
||||
fake_discover)
|
||||
return fake_discover
|
||||
|
||||
|
||||
def _payload(**overrides):
|
||||
body = {
|
||||
"brand": "Britannia",
|
||||
"products": [
|
||||
{"product_name": "Marie Gold", "size_variants": ["250g"],
|
||||
"barcode": "8901063014206", "providers": ["Amazon"],
|
||||
"highlights": ["Tea-time favourite"], "nutrients": ["Iron - Blood health"]},
|
||||
],
|
||||
}
|
||||
body.update(overrides)
|
||||
return body
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Auth
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_preview_requires_admin(client):
|
||||
assert client.post(PREVIEW, json={"brand": "Britannia"}).status_code in (401, 403)
|
||||
|
||||
|
||||
def test_ingest_requires_admin(client):
|
||||
assert client.post(INGEST, json=_payload()).status_code in (401, 403)
|
||||
|
||||
|
||||
def test_a_plain_user_cannot_discover(client, user_headers):
|
||||
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=user_headers)
|
||||
assert resp.status_code == 403
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Preview writes nothing
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_preview_returns_products_and_the_stage_list(client, admin_headers, discovered):
|
||||
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert len(body["products"]) == 2
|
||||
assert body["table"] == "brand_britannia"
|
||||
assert body["stages"] == list(pipeline.STAGE_NAMES)
|
||||
assert len(body["stages"]) == 11
|
||||
assert body["products"][0]["selected"] is True
|
||||
|
||||
|
||||
def test_preview_stages_nothing(client, admin_headers, discovered, batch_root,
|
||||
no_background_worker):
|
||||
"""The whole reason preview is a separate route."""
|
||||
client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
|
||||
|
||||
assert no_background_worker == []
|
||||
assert not batch_root.exists() or list(batch_root.iterdir()) == []
|
||||
|
||||
|
||||
def test_preview_reports_a_bad_brand_name_as_client_error(client, admin_headers):
|
||||
resp = client.post(PREVIEW, json={"brand": " "}, headers=admin_headers)
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_preview_surfaces_a_discovery_failure_rather_than_500(client, admin_headers,
|
||||
monkeypatch):
|
||||
from app.api.routers import brand_discovery as router_module
|
||||
|
||||
def boom(brand, **kwargs):
|
||||
raise RuntimeError("Open Food Facts is unreachable")
|
||||
|
||||
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", boom)
|
||||
|
||||
resp = client.post(PREVIEW, json={"brand": "Britannia"}, headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 502
|
||||
assert "unreachable" in resp.json()["detail"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The ACTIVE_BRANDS gate
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_preview_warns_when_the_brand_is_not_active(client, admin_headers, discovered,
|
||||
monkeypatch):
|
||||
from app.services import active_brands
|
||||
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
|
||||
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
|
||||
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"])
|
||||
|
||||
from app.api.routers import brand_discovery as router_module
|
||||
|
||||
def inactive(brand, **kwargs):
|
||||
result = discovered(brand, **kwargs)
|
||||
result.brand_active = False
|
||||
result.filtering_enabled = True
|
||||
return result
|
||||
|
||||
monkeypatch.setattr(router_module.brand_discovery, "discover_brand_products", inactive)
|
||||
|
||||
body = client.post(PREVIEW, json={"brand": "Britannia"},
|
||||
headers=admin_headers).json()
|
||||
|
||||
assert body["brand_active"] is False
|
||||
assert any("ACTIVE_BRANDS" in w for w in body["warnings"])
|
||||
|
||||
|
||||
def test_ingest_is_refused_for_an_inactive_brand(client, admin_headers, monkeypatch):
|
||||
"""A green run over a catalog no endpoint can read is not an acceptable
|
||||
outcome to hand back silently."""
|
||||
from app.services import active_brands
|
||||
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
|
||||
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
|
||||
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul", "Cadbury"])
|
||||
|
||||
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 409
|
||||
detail = resp.json()["detail"]
|
||||
assert "ACTIVE_BRANDS=Amul,Cadbury,Britannia" in detail
|
||||
assert "restart" in detail.lower()
|
||||
|
||||
|
||||
def test_ingest_proceeds_for_an_inactive_brand_when_acknowledged(client, admin_headers,
|
||||
monkeypatch):
|
||||
from app.services import active_brands
|
||||
monkeypatch.setattr(active_brands, "filtering_enabled", lambda: True)
|
||||
monkeypatch.setattr(active_brands, "is_active_brand", lambda brand: False)
|
||||
monkeypatch.setattr(active_brands, "active_display_names", lambda: ["Amul"])
|
||||
|
||||
resp = client.post(INGEST, json=_payload(acknowledge_inactive=True),
|
||||
headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 202
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Ingest
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_ingest_stages_one_batch_and_returns_a_pollable_id(client, admin_headers,
|
||||
no_background_worker):
|
||||
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 202
|
||||
body = resp.json()
|
||||
assert body["batch_id"]
|
||||
assert body["files_total"] == 1
|
||||
assert len(body["stage_names"]) == 11
|
||||
assert no_background_worker == [body["batch_id"]]
|
||||
|
||||
|
||||
def test_the_staged_file_is_a_real_csv_the_pipeline_can_read(client, admin_headers,
|
||||
batch_root):
|
||||
"""Not a placeholder: the bytes on the batch volume are the exact input that
|
||||
produced the rows, and they must parse with the same reader an upload uses."""
|
||||
batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"]
|
||||
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
entry = manifest.files[0]
|
||||
assert entry.filename.startswith("discovered-britannia-")
|
||||
assert entry.filename.endswith(".csv")
|
||||
|
||||
contents = (batch_ingest.batch_dir(batch_id) / entry.stored_name).read_bytes()
|
||||
df, mapping = pipeline.parse_spreadsheet(entry.filename, contents)
|
||||
|
||||
assert len(df) == 1
|
||||
assert mapping.unrecognised == []
|
||||
assert "product_name" in mapping.columns
|
||||
|
||||
|
||||
def test_the_batch_records_where_it_came_from(client, admin_headers):
|
||||
batch_id = client.post(INGEST, json=_payload(), headers=admin_headers).json()["batch_id"]
|
||||
|
||||
manifest = batch_ingest.read_manifest(batch_id)
|
||||
|
||||
assert manifest.submitted_by == "brand-discovery: Britannia"
|
||||
|
||||
|
||||
def test_ingest_runs_the_eleven_stages_end_to_end(client, admin_headers, store,
|
||||
no_background_worker, monkeypatch):
|
||||
"""Drive the real worker function over the staged file.
|
||||
|
||||
`use_llm` and `fetch_images` are turned OFF for this one, matching every
|
||||
other pipeline test in the suite: with them on, `run_batch` really calls
|
||||
Ollama once per row and really searches the open web for images. That is the
|
||||
correct production default and a terrible test - slow, and it fails when the
|
||||
machine is offline.
|
||||
"""
|
||||
body = _payload(use_llm=False, fetch_images=False)
|
||||
batch_id = client.post(INGEST, json=body, headers=admin_headers).json()["batch_id"]
|
||||
|
||||
batch_ingest.run_batch(batch_id)
|
||||
|
||||
assert len(store) == 1
|
||||
row = next(iter(store.values()))
|
||||
assert row["image_id"] == "britannia_marie_gold_250g"
|
||||
assert row["barcode"] == "8901063014206"
|
||||
assert row["product_sku"]
|
||||
assert row["highlights"] == ["Tea-time favourite"]
|
||||
|
||||
|
||||
def test_an_empty_selection_is_refused(client, admin_headers):
|
||||
resp = client.post(INGEST, json=_payload(products=[]), headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 400
|
||||
assert "nothing to ingest" in resp.json()["detail"]
|
||||
|
||||
|
||||
def test_a_blank_brand_is_refused(client, admin_headers):
|
||||
resp = client.post(INGEST, json=_payload(brand=" "), headers=admin_headers)
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_a_full_queue_is_reported_as_429(client, admin_headers, monkeypatch):
|
||||
import queue
|
||||
|
||||
def full(batch_id):
|
||||
raise queue.Full()
|
||||
|
||||
from app.core import batch_worker
|
||||
monkeypatch.setattr(batch_worker, "submit", full)
|
||||
|
||||
resp = client.post(INGEST, json=_payload(), headers=admin_headers)
|
||||
|
||||
assert resp.status_code == 429
|
||||
assert "Resume" in resp.json()["detail"]
|
||||
|
||||
|
||||
def test_ingest_defaults_turn_on_images_and_the_per_row_llm(client, admin_headers,
|
||||
monkeypatch):
|
||||
"""The two enrichment stages this feature runs with. Both are per-run
|
||||
arguments already, so neither needs a settings change."""
|
||||
seen = {}
|
||||
real = batch_common.stage_and_queue
|
||||
|
||||
def spy(valid, invalid, *, use_llm, fetch_images, submitted_by=None):
|
||||
seen["use_llm"] = use_llm
|
||||
seen["fetch_images"] = fetch_images
|
||||
return real(valid, invalid, use_llm=use_llm, fetch_images=fetch_images,
|
||||
submitted_by=submitted_by)
|
||||
|
||||
from app.api.routers import brand_discovery as router_module
|
||||
monkeypatch.setattr(router_module.batch_common, "stage_and_queue", spy)
|
||||
|
||||
client.post(INGEST, json=_payload(), headers=admin_headers)
|
||||
|
||||
assert seen == {"use_llm": True, "fetch_images": True}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
table: dict = {}
|
||||
|
||||
def fake_upsert(brand, rows, cleanup=False):
|
||||
assert cleanup is False, "cleanup=True would delete the brand's existing catalog"
|
||||
for row in rows:
|
||||
table[row["image_id"]] = dict(row)
|
||||
return len(rows)
|
||||
|
||||
monkeypatch.setattr(pipeline, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(pipeline, "get_products_by_brand", lambda b, **kw: list(table.values()))
|
||||
monkeypatch.setattr(pipeline, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
return table
|
||||
Reference in New Issue
Block a user