backend updates for bulk product uploads from user
This commit is contained in:
@@ -19,7 +19,20 @@ async def _run_job(job_id: str, brand: str, max_products: int) -> None:
|
||||
job_store.update(job_id, "running")
|
||||
try:
|
||||
summary = await ingest_brand(brand, max_products=max_products)
|
||||
job_store.update(job_id, "done", detail=f"{summary['total_products']} products ingested")
|
||||
# "Ingested" has to mean "in the database". The generation stages can all
|
||||
# succeed while the pgvector write fails, and reporting that as done is
|
||||
# how a run that stored nothing ends up looking successful in the UI.
|
||||
if summary.get("storage_error"):
|
||||
job_store.update(
|
||||
job_id,
|
||||
"failed",
|
||||
detail=(
|
||||
f"Generated {summary['total_products']} product(s) but storing them "
|
||||
f"failed, so none are in the catalog: {summary['storage_error']}"
|
||||
),
|
||||
)
|
||||
else:
|
||||
job_store.update(job_id, "done", detail=f"{summary['total_products']} products ingested")
|
||||
except Exception as e: # noqa: BLE001 - surface any failure to the UI
|
||||
logger.exception("Catalog ingestion job %s failed", job_id)
|
||||
job_store.update(job_id, "failed", detail=str(e))
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -940,11 +940,17 @@ class ProductCatalogEngine:
|
||||
if 'image_id' not in p:
|
||||
id_source = p.get('product_name') or p.get('title', 'unknown_product')
|
||||
p['image_id'] = s3_service.generate_image_id(id_source)
|
||||
upsert_brand_products(brand, enhanced_products, cleanup=True)
|
||||
logger.info("🧠 Stored embeddings to pgvector")
|
||||
stored = upsert_brand_products(brand, enhanced_products, cleanup=True)
|
||||
logger.info("🧠 Stored %s product(s) with embeddings to pgvector", stored)
|
||||
except Exception as e:
|
||||
logger.warning(f"Vector storage skipped/failed: {e}")
|
||||
|
||||
# Stage 4 is where the catalog becomes readable by the app - every
|
||||
# API read goes to pgvector, not to the dict returned here. So a
|
||||
# failure at this stage means the run produced nothing the user can
|
||||
# see, and it has to travel back to the job status rather than being
|
||||
# logged and forgotten.
|
||||
logger.error("❌ Stage 4 (pgvector storage) failed for '%s': %s", brand, e)
|
||||
catalog['storage_error'] = str(e)
|
||||
|
||||
return catalog
|
||||
|
||||
def save_catalog(self, catalog: Dict[str, Any], filename: str = None) -> str:
|
||||
|
||||
@@ -40,11 +40,17 @@ async def ingest_brand(brand: str, max_products: int = 50) -> Dict[str, Any]:
|
||||
# Placed here rather than in the API router because this function is also
|
||||
# the CLI's entry point (cli/ingest_brand.py), and a failure to write the
|
||||
# file must not turn a successful ingest into a failed job.
|
||||
try:
|
||||
from app.services.brand_sync import export_brand_to_seed_file
|
||||
export_brand_to_seed_file(brand)
|
||||
except Exception: # noqa: BLE001 - DB rows are already committed
|
||||
logger.warning("Seed-catalog export failed for %s (DB rows intact)", brand, exc_info=True)
|
||||
storage_error = catalog.get("storage_error")
|
||||
|
||||
# Nothing reached the database, so there is nothing to mirror out of it -
|
||||
# and running the export anyway would either write an empty file or leave a
|
||||
# stale one looking current.
|
||||
if not storage_error:
|
||||
try:
|
||||
from app.services.brand_sync import export_brand_to_seed_file
|
||||
export_brand_to_seed_file(brand)
|
||||
except Exception: # noqa: BLE001 - DB rows are already committed
|
||||
logger.warning("Seed-catalog export failed for %s (DB rows intact)", brand, exc_info=True)
|
||||
|
||||
summary = {
|
||||
"brand": brand,
|
||||
@@ -52,6 +58,7 @@ async def ingest_brand(brand: str, max_products: int = 50) -> Dict[str, Any]:
|
||||
"total_images": catalog.get("total_images", 0),
|
||||
"duration_seconds": round(duration, 2),
|
||||
"engine_info": catalog.get("engine_info", {}),
|
||||
"storage_error": storage_error,
|
||||
}
|
||||
logger.info("Finished ingestion for brand=%s in %.2fs: %s products",
|
||||
brand, duration, summary["total_products"])
|
||||
|
||||
@@ -22,6 +22,28 @@ from app.services.s3_service import s3_service
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class VectorStoreUnavailable(RuntimeError):
|
||||
"""The pgvector database could not be reached, so a write did not happen.
|
||||
|
||||
Exists because the silent alternative was a production bug that took a long
|
||||
time to see: upsert_brand_products() used to `return` when _connect() gave
|
||||
back None - unreachable host, wrong DB_PASSWORD, USE_PGVECTOR=false - and
|
||||
every caller read that as a successful write. The upload endpoints then
|
||||
answered "success", the seed JSON was updated, and not one row existed in
|
||||
the database. A write that cannot happen has to raise.
|
||||
"""
|
||||
|
||||
|
||||
class VectorStoreWriteFailed(RuntimeError):
|
||||
"""The INSERT ran without error but the rows are not in the table.
|
||||
|
||||
Guards against the failure modes an exception cannot catch: a statement
|
||||
silently rolled back, a trigger swallowing the row, or an ON CONFLICT
|
||||
target that quietly matched nothing. The only trustworthy proof of a write
|
||||
is reading it back.
|
||||
"""
|
||||
|
||||
|
||||
def _sanitize_name(name: str) -> str:
|
||||
"""Sanitize a brand name for use as a PostgreSQL table name suffix.
|
||||
|
||||
@@ -193,22 +215,44 @@ def ensure_brand_schema(brand: str) -> str:
|
||||
return table_name
|
||||
|
||||
|
||||
def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: bool = False) -> None:
|
||||
def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: bool = False) -> int:
|
||||
"""Insert products into brand-specific table - simplified with only essential fields
|
||||
|
||||
When `cleanup=True`, any products in the table whose image_id is NOT in the
|
||||
provided `products` list are deleted after the upsert. This ensures the
|
||||
database exactly reflects the source data. The caller is responsible for
|
||||
providing the complete set of products for the brand when using cleanup.
|
||||
|
||||
Returns the number of distinct image_ids confirmed present in the table
|
||||
afterwards - the rows are read back, so a non-raising call is proof of
|
||||
persistence rather than proof that a statement was merely sent.
|
||||
|
||||
Raises:
|
||||
VectorStoreUnavailable: the database is unreachable; nothing was written.
|
||||
VectorStoreWriteFailed: the statements ran but the rows are not there.
|
||||
ValueError: a product carries no image_id, which is the primary key
|
||||
every other product is deduplicated on.
|
||||
"""
|
||||
if not products:
|
||||
return 0
|
||||
|
||||
conn = _connect()
|
||||
if not conn:
|
||||
return
|
||||
|
||||
raise VectorStoreUnavailable(
|
||||
f"Cannot save products for '{brand}': the product database is unreachable "
|
||||
f"(USE_PGVECTOR={USE_PGVECTOR}, host={DB_HOST}:{DB_PORT}, db={DB_NAME}). "
|
||||
f"Nothing was saved. Check DB_HOST/DB_USER/DB_PASSWORD in backend/.env "
|
||||
f"and that Postgres is accepting connections."
|
||||
)
|
||||
|
||||
table_name = ensure_brand_schema(brand)
|
||||
if not table_name:
|
||||
return
|
||||
|
||||
conn.close()
|
||||
raise VectorStoreUnavailable(
|
||||
f"Cannot save products for '{brand}': the brand table could not be created "
|
||||
f"or verified in database '{DB_NAME}'. Nothing was saved."
|
||||
)
|
||||
|
||||
rows = []
|
||||
for p in products:
|
||||
# Extract only essential fields
|
||||
@@ -333,55 +377,89 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
|
||||
image_ids = [r[4] for r in rows if r[4]]
|
||||
|
||||
with conn, conn.cursor() as cur:
|
||||
cur.executemany(
|
||||
f"""
|
||||
INSERT INTO {table_name}
|
||||
(product_name, title, description, category, image_id, image_url, image_urls, price_range, size_variants, providers,
|
||||
fssai_license, product_sku, sku_source, hsn_code, final_selling_price, selling_price, barcode, barcode_type, highlights, nutrients, search_query, embedding)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
ON CONFLICT (image_id) DO UPDATE SET
|
||||
product_name = EXCLUDED.product_name,
|
||||
title = EXCLUDED.title,
|
||||
description = EXCLUDED.description,
|
||||
category = EXCLUDED.category,
|
||||
image_url = EXCLUDED.image_url,
|
||||
image_urls = EXCLUDED.image_urls,
|
||||
price_range = EXCLUDED.price_range,
|
||||
size_variants = EXCLUDED.size_variants,
|
||||
providers = EXCLUDED.providers,
|
||||
fssai_license = EXCLUDED.fssai_license,
|
||||
product_sku = EXCLUDED.product_sku,
|
||||
sku_source = EXCLUDED.sku_source,
|
||||
hsn_code = EXCLUDED.hsn_code,
|
||||
final_selling_price = EXCLUDED.final_selling_price,
|
||||
selling_price = EXCLUDED.selling_price,
|
||||
barcode = EXCLUDED.barcode,
|
||||
barcode_type = EXCLUDED.barcode_type,
|
||||
highlights = EXCLUDED.highlights,
|
||||
nutrients = EXCLUDED.nutrients,
|
||||
search_query = EXCLUDED.search_query,
|
||||
embedding = EXCLUDED.embedding,
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
""",
|
||||
rows,
|
||||
# An empty image_id is not a harmless blank: it is the conflict target, so
|
||||
# two such products would overwrite each other and the second would replace
|
||||
# the first instead of being added. Refuse the batch and name the rows.
|
||||
if len(image_ids) != len(rows):
|
||||
unnamed = [r[0] or "<no product_name>" for r in rows if not r[4]]
|
||||
conn.close()
|
||||
raise ValueError(
|
||||
f"{len(unnamed)} product(s) for '{brand}' have no image_id and cannot be "
|
||||
f"stored (a product needs a name with at least one letter or digit): "
|
||||
f"{', '.join(unnamed[:5])}"
|
||||
)
|
||||
logger.info(f"✅ Upserted {len(rows)} products into {table_name}")
|
||||
|
||||
# Remove stale products that were deleted from the source data.
|
||||
# Only runs when cleanup=True so that callers processing partial
|
||||
# product sets (e.g. multiple seed files contributing to the same
|
||||
# brand table) don't accidentally orphan each other's data.
|
||||
if cleanup and image_ids:
|
||||
cur.execute(
|
||||
f"DELETE FROM {table_name} WHERE image_id != ALL(%s::text[])",
|
||||
(image_ids,),
|
||||
expected_ids = sorted(set(image_ids))
|
||||
|
||||
try:
|
||||
with conn, conn.cursor() as cur:
|
||||
cur.executemany(
|
||||
f"""
|
||||
INSERT INTO {table_name}
|
||||
(product_name, title, description, category, image_id, image_url, image_urls, price_range, size_variants, providers,
|
||||
fssai_license, product_sku, sku_source, hsn_code, final_selling_price, selling_price, barcode, barcode_type, highlights, nutrients, search_query, embedding)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
ON CONFLICT (image_id) DO UPDATE SET
|
||||
product_name = EXCLUDED.product_name,
|
||||
title = EXCLUDED.title,
|
||||
description = EXCLUDED.description,
|
||||
category = EXCLUDED.category,
|
||||
image_url = EXCLUDED.image_url,
|
||||
image_urls = EXCLUDED.image_urls,
|
||||
price_range = EXCLUDED.price_range,
|
||||
size_variants = EXCLUDED.size_variants,
|
||||
providers = EXCLUDED.providers,
|
||||
fssai_license = EXCLUDED.fssai_license,
|
||||
product_sku = EXCLUDED.product_sku,
|
||||
sku_source = EXCLUDED.sku_source,
|
||||
hsn_code = EXCLUDED.hsn_code,
|
||||
final_selling_price = EXCLUDED.final_selling_price,
|
||||
selling_price = EXCLUDED.selling_price,
|
||||
barcode = EXCLUDED.barcode,
|
||||
barcode_type = EXCLUDED.barcode_type,
|
||||
highlights = EXCLUDED.highlights,
|
||||
nutrients = EXCLUDED.nutrients,
|
||||
search_query = EXCLUDED.search_query,
|
||||
embedding = EXCLUDED.embedding,
|
||||
updated_at = CURRENT_TIMESTAMP
|
||||
""",
|
||||
rows,
|
||||
)
|
||||
deleted = cur.rowcount
|
||||
if deleted:
|
||||
logger.info(f"🗑️ Removed {deleted} stale product(s) from {table_name}")
|
||||
|
||||
conn.close()
|
||||
# Remove stale products that were deleted from the source data.
|
||||
# Only runs when cleanup=True so that callers processing partial
|
||||
# product sets (e.g. multiple seed files contributing to the same
|
||||
# brand table) don't accidentally orphan each other's data.
|
||||
if cleanup and image_ids:
|
||||
cur.execute(
|
||||
f"DELETE FROM {table_name} WHERE image_id != ALL(%s::text[])",
|
||||
(image_ids,),
|
||||
)
|
||||
deleted = cur.rowcount
|
||||
if deleted:
|
||||
logger.info(f"🗑️ Removed {deleted} stale product(s) from {table_name}")
|
||||
|
||||
# Read the rows back. This is the line that turns "we sent an
|
||||
# INSERT" into "the data is in the table", and it is the only
|
||||
# signal the API layer is allowed to report success on.
|
||||
cur.execute(
|
||||
f"SELECT COUNT(DISTINCT image_id) FROM {table_name} WHERE image_id = ANY(%s::text[])",
|
||||
(expected_ids,),
|
||||
)
|
||||
row = cur.fetchone()
|
||||
persisted = int(row[0]) if row else 0
|
||||
|
||||
if persisted < len(expected_ids):
|
||||
raise VectorStoreWriteFailed(
|
||||
f"Wrote {len(rows)} product(s) for '{brand}' to {table_name} but only "
|
||||
f"{persisted} of {len(expected_ids)} are readable back afterwards. "
|
||||
f"The data was not saved - treat this as a failed import."
|
||||
)
|
||||
|
||||
logger.info("✅ Upserted %d product(s) into %s (%d verified in table)",
|
||||
len(rows), table_name, persisted)
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
# Every write path into the catalog funnels through here, so this is the
|
||||
# one place that has to invalidate the derived views: the brand cards'
|
||||
@@ -393,6 +471,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
|
||||
except Exception: # noqa: BLE001 - cache invalidation must never fail a write
|
||||
pass
|
||||
|
||||
return persisted
|
||||
|
||||
|
||||
def get_existing_product_image_id(brand: str, product_name: str) -> Optional[str]:
|
||||
"""Check if a product with this name exists in the brand table and return its image_id"""
|
||||
|
||||
@@ -71,6 +71,14 @@ boto3>=1.34.162
|
||||
# --- missing from this file - declared explicitly now. ---
|
||||
scikit-learn>=1.5.2
|
||||
pandas>=2.2.2
|
||||
# pandas' Excel readers are optional extras it does not install itself, and the
|
||||
# upload endpoints (/api/user/products/upload-file, /api/upload/*) accept .xlsx
|
||||
# and .xls. Undeclared, they happened to be present in some environments and
|
||||
# absent in others - so an Excel upload that worked locally failed in the
|
||||
# container with "Missing optional dependency 'openpyxl'". openpyxl reads
|
||||
# .xlsx/.xlsm; xlrd is only for the legacy .xls format.
|
||||
openpyxl>=3.1.5
|
||||
xlrd>=2.0.1
|
||||
numpy>=1.26.4
|
||||
scipy>=1.13.1
|
||||
joblib>=1.4.2
|
||||
|
||||
370
tests/test_user_products_upload.py
Normal file
370
tests/test_user_products_upload.py
Normal file
@@ -0,0 +1,370 @@
|
||||
"""
|
||||
Tests for the user product upload path.
|
||||
|
||||
The bug these exist for: uploading a spreadsheet returned a success message
|
||||
while nothing reached the database. So the assertions here are deliberately not
|
||||
"did it answer 2xx" - they are "did it answer 2xx *and* hand the rows to the
|
||||
store", and, for every failure mode, "did it refuse to call that a success".
|
||||
|
||||
Hermetic, like the rest of the suite: the store, the embedding model and the
|
||||
seed-catalog writer are all substituted, so nothing here touches a real
|
||||
database, downloads a model, or writes into data/seed_catalogs/.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
|
||||
from app.api.routers import user_products
|
||||
from app.services.vector_store import VectorStoreUnavailable
|
||||
|
||||
UPLOAD_URL = "/api/user/products/upload-file"
|
||||
|
||||
# The headers the frontend's "Download Sample CSV" button produces.
|
||||
TEMPLATE_CSV = (
|
||||
"Brand Name,Product Name / Variant,Category,Price Range,Final Price (₹),"
|
||||
"Barcode (GTIN/EAN),HSN Code,Custom Image URL,Description\n"
|
||||
"Lion Dates,Lion Dates 450g,Health Foods,₹160-220,185.00,20086040,2008,,Premium dates\n"
|
||||
"Naga,Naga Maida 2kg,Flour & Grains,₹90-110,98.00,8906012345001,1101,,Refined wheat flour\n"
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def store(monkeypatch):
|
||||
"""Capture what the endpoint hands to the database instead of writing it.
|
||||
|
||||
Also stubs the two slow collaborators. `embed_texts` would download and load
|
||||
a sentence-transformer; `get_products_by_brand` would open a real connection
|
||||
to whatever DB_HOST points at.
|
||||
"""
|
||||
calls = []
|
||||
|
||||
def fake_upsert(brand, products, cleanup=False):
|
||||
calls.append((brand, products))
|
||||
return len(products)
|
||||
|
||||
monkeypatch.setattr(user_products, "upsert_brand_products", fake_upsert)
|
||||
monkeypatch.setattr(user_products, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
|
||||
monkeypatch.setattr(user_products, "get_products_by_brand", lambda *a, **k: [])
|
||||
monkeypatch.setattr(user_products, "upsert_products_into_catalog_file", lambda brand, products: None)
|
||||
return calls
|
||||
|
||||
|
||||
def _upload(client, headers, content: str | bytes, name: str = "products.csv"):
|
||||
data = content.encode("utf-8") if isinstance(content, str) else content
|
||||
return client.post(UPLOAD_URL, files={"file": (name, data)}, headers=headers)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# The reported bug: success without persistence
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_upload_is_not_a_success_when_the_database_is_unreachable(client, user_headers, monkeypatch, store):
|
||||
"""The exact production symptom. An unreachable database must not answer 201."""
|
||||
def unavailable(brand, products, cleanup=False):
|
||||
raise VectorStoreUnavailable("the product database is unreachable (host=db:5432)")
|
||||
|
||||
monkeypatch.setattr(user_products, "upsert_brand_products", unavailable)
|
||||
|
||||
resp = _upload(client, user_headers, TEMPLATE_CSV)
|
||||
|
||||
assert resp.status_code == 503, resp.text
|
||||
detail = resp.json()["detail"]
|
||||
assert "Nothing was saved" in detail
|
||||
assert "unreachable" in detail
|
||||
|
||||
|
||||
def test_single_add_is_not_a_success_when_the_database_is_unreachable(client, user_headers, monkeypatch, store):
|
||||
def unavailable(brand, products, cleanup=False):
|
||||
raise VectorStoreUnavailable("the product database is unreachable (host=db:5432)")
|
||||
|
||||
monkeypatch.setattr(user_products, "upsert_brand_products", unavailable)
|
||||
|
||||
resp = client.post(
|
||||
"/api/user/products/add",
|
||||
json={"brand": "Lion Dates", "product_name": "Lion Dates 450g"},
|
||||
headers=user_headers,
|
||||
)
|
||||
|
||||
assert resp.status_code == 503, resp.text
|
||||
assert "not saved" in resp.json()["detail"]
|
||||
|
||||
|
||||
def test_upload_reaches_the_database_and_reports_what_it_stored(client, user_headers, store):
|
||||
resp = _upload(client, user_headers, TEMPLATE_CSV)
|
||||
|
||||
assert resp.status_code == 201, resp.text
|
||||
body = resp.json()
|
||||
assert body["status"] == "success"
|
||||
assert body["added_count"] == 2
|
||||
assert body["error_count"] == 0
|
||||
assert body["rows_total"] == 2
|
||||
|
||||
stored = {p["product_name"]: p for _, products in store for p in products}
|
||||
assert set(stored) == {"Lion Dates 450g", "Naga Maida 2kg"}
|
||||
|
||||
dates = stored["Lion Dates 450g"]
|
||||
assert dates["category"] == "Health Foods"
|
||||
assert dates["final_selling_price"] == 185.0
|
||||
assert dates["price_range"] == "₹160-220"
|
||||
# Not "20086040.0" - an identifier read as a float and stringified is a
|
||||
# different identifier.
|
||||
assert dates["barcode"] == "20086040"
|
||||
assert dates["hsn_code"] == "2008"
|
||||
assert stored["Naga Maida 2kg"]["barcode"] == "8906012345001"
|
||||
|
||||
|
||||
def test_a_real_xlsx_workbook_imports_with_its_identifiers_intact(client, user_headers, store):
|
||||
"""The reported case was an .xlsx upload, and Excel is where identifiers rot.
|
||||
|
||||
A 13-digit barcode in a spreadsheet cell is a number to Excel, so it arrives
|
||||
as a float and stringifies to "8906012345001.0"; an HSN code of 0402 loses
|
||||
its leading zero. Both are then stored - silently - as a different value
|
||||
than the one in the file.
|
||||
"""
|
||||
openpyxl = pytest.importorskip("openpyxl", reason="declared in requirements.txt for .xlsx uploads")
|
||||
|
||||
workbook = openpyxl.Workbook()
|
||||
sheet = workbook.active
|
||||
sheet.append(["Brand Name", "Product Name", "Final Price (₹)", "Barcode (GTIN/EAN)", "HSN Code"])
|
||||
sheet.append(["Lion Dates", "Lion Dates 450g", 185, 8906012345001, "0402"])
|
||||
|
||||
buffer = io.BytesIO()
|
||||
workbook.save(buffer)
|
||||
|
||||
resp = _upload(client, user_headers, buffer.getvalue(), name="products.xlsx")
|
||||
|
||||
assert resp.status_code == 201, resp.text
|
||||
assert resp.json()["added_count"] == 1
|
||||
|
||||
stored = [p for _, products in store for p in products][0]
|
||||
assert stored["barcode"] == "8906012345001"
|
||||
assert stored["hsn_code"] == "0402"
|
||||
assert stored["final_selling_price"] == 185.0
|
||||
|
||||
|
||||
def test_partial_import_is_reported_as_partial_not_success(client, user_headers, monkeypatch, store):
|
||||
"""One brand failing must neither sink the other nor be called a success."""
|
||||
def selective(brand, products, cleanup=False):
|
||||
if "naga" in brand.lower():
|
||||
raise RuntimeError("column overflow on selling_price")
|
||||
return len(products)
|
||||
|
||||
monkeypatch.setattr(user_products, "upsert_brand_products", selective)
|
||||
|
||||
resp = _upload(client, user_headers, TEMPLATE_CSV)
|
||||
|
||||
assert resp.status_code == 201, resp.text
|
||||
body = resp.json()
|
||||
assert body["status"] == "partial"
|
||||
assert body["added_count"] == 1
|
||||
assert body["error_count"] == 1
|
||||
assert body["errors"][0]["row"] == 3 # header is row 1, Naga is row 3
|
||||
assert "column overflow" in body["errors"][0]["error"]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Header handling
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_template_headers_all_map_to_their_field():
|
||||
mapping = user_products.map_spreadsheet_columns([
|
||||
"Brand Name", "Product Name / Variant", "Category", "Price Range",
|
||||
"Final Price (₹)", "Barcode (GTIN/EAN)", "HSN Code", "Custom Image URL",
|
||||
"Description",
|
||||
])
|
||||
|
||||
assert mapping.columns == {
|
||||
"brand": "Brand Name",
|
||||
"product_name": "Product Name / Variant",
|
||||
"category": "Category",
|
||||
"price_range": "Price Range",
|
||||
"final_selling_price": "Final Price (₹)",
|
||||
"barcode": "Barcode (GTIN/EAN)",
|
||||
"hsn_code": "HSN Code",
|
||||
"image_url": "Custom Image URL",
|
||||
"description": "Description",
|
||||
}
|
||||
assert mapping.ignored == []
|
||||
|
||||
|
||||
def test_a_product_sku_column_is_not_mistaken_for_the_product_name():
|
||||
"""'Product SKU' contains 'product'. Matched loosely, it used to become a
|
||||
second product_name column, and duplicate columns are what turned a row
|
||||
into a stringified Series."""
|
||||
mapping = user_products.map_spreadsheet_columns(["Brand", "Product Name", "Product SKU"])
|
||||
|
||||
assert mapping.columns["product_name"] == "Product Name"
|
||||
assert mapping.columns["product_sku"] == "Product SKU"
|
||||
|
||||
|
||||
def test_colliding_columns_are_reported_and_do_not_corrupt_rows(client, user_headers, store):
|
||||
"""Two columns for one field: keep the first, say so, keep importing."""
|
||||
csv = (
|
||||
"Brand,Product Name,Item Name,Final Price\n"
|
||||
"Lion Dates,Lion Dates 450g,Ignore This One,185\n"
|
||||
)
|
||||
|
||||
resp = _upload(client, user_headers, csv)
|
||||
|
||||
assert resp.status_code == 201, resp.text
|
||||
body = resp.json()
|
||||
assert body["added_count"] == 1
|
||||
assert body["ignored_columns"] == [
|
||||
{"column": "Item Name", "field": "product_name", "using_instead": "Product Name"}
|
||||
]
|
||||
assert any("ignored" in w for w in body["warnings"])
|
||||
|
||||
stored = [p for _, products in store for p in products]
|
||||
assert stored[0]["product_name"] == "Lion Dates 450g"
|
||||
|
||||
|
||||
def test_missing_required_columns_is_a_400_naming_the_headers(client, user_headers, store):
|
||||
resp = _upload(client, user_headers, "Foo,Bar\n1,2\n")
|
||||
|
||||
assert resp.status_code == 400
|
||||
detail = resp.json()["detail"]
|
||||
assert "brand column" in detail
|
||||
assert "Foo" in detail
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Row-level handling
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_rows_missing_a_brand_are_errors_not_silent_skips(client, user_headers, store):
|
||||
csv = (
|
||||
"Brand,Product Name\n"
|
||||
"Lion Dates,Lion Dates 450g\n"
|
||||
",Orphan Product\n"
|
||||
)
|
||||
|
||||
resp = _upload(client, user_headers, csv)
|
||||
|
||||
body = resp.json()
|
||||
assert resp.status_code == 201, resp.text
|
||||
assert body["added_count"] == 1
|
||||
assert body["error_count"] == 1
|
||||
assert body["errors"][0]["row"] == 3
|
||||
assert "brand" in body["errors"][0]["error"]
|
||||
|
||||
|
||||
def test_fully_blank_rows_are_skipped_and_counted(client, user_headers, store):
|
||||
csv = (
|
||||
"Brand,Product Name\n"
|
||||
"Lion Dates,Lion Dates 450g\n"
|
||||
",\n"
|
||||
",\n"
|
||||
)
|
||||
|
||||
resp = _upload(client, user_headers, csv)
|
||||
|
||||
body = resp.json()
|
||||
assert body["added_count"] == 1
|
||||
assert body["error_count"] == 0
|
||||
assert body["skipped_blank_rows"] == 2
|
||||
|
||||
|
||||
def test_a_file_where_every_row_fails_is_never_a_success(client, user_headers, store):
|
||||
csv = (
|
||||
"Brand,Product Name\n"
|
||||
",Orphan One\n"
|
||||
",Orphan Two\n"
|
||||
)
|
||||
|
||||
resp = _upload(client, user_headers, csv)
|
||||
|
||||
assert resp.status_code == 422, resp.text
|
||||
assert "Nothing was saved" in resp.json()["detail"]
|
||||
assert store == []
|
||||
|
||||
|
||||
def test_empty_and_headers_only_files_are_rejected(client, user_headers, store):
|
||||
assert _upload(client, user_headers, b"").status_code == 400
|
||||
assert _upload(client, user_headers, "Brand,Product Name\n").status_code == 400
|
||||
|
||||
|
||||
def test_unsupported_file_type_is_rejected(client, user_headers, store):
|
||||
resp = _upload(client, user_headers, b"%PDF-1.4", name="products.pdf")
|
||||
|
||||
assert resp.status_code == 400
|
||||
assert "Unsupported file type" in resp.json()["detail"]
|
||||
|
||||
|
||||
def test_oversized_row_count_is_rejected_before_any_write(client, user_headers, store):
|
||||
rows = "".join(f"Lion Dates,Product {i}\n" for i in range(user_products.MAX_UPLOAD_ROWS + 1))
|
||||
|
||||
resp = _upload(client, user_headers, "Brand,Product Name\n" + rows)
|
||||
|
||||
assert resp.status_code == 413
|
||||
assert store == []
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Batching: the collaborators must be called once per upload, not once per row
|
||||
# ---------------------------------------------------------------------------
|
||||
def test_one_embedding_call_and_one_write_per_brand(client, user_headers, monkeypatch, store):
|
||||
embed_calls = []
|
||||
sample_calls = []
|
||||
# Pinned rather than inherited: the assertion below is about batching, and
|
||||
# it should not silently pass because a .env happened to disable embeddings.
|
||||
monkeypatch.setattr(user_products, "USE_EMBEDDINGS", True)
|
||||
monkeypatch.setattr(user_products, "embed_texts",
|
||||
lambda texts: embed_calls.append(len(texts)) or [[0.0] * 384 for _ in texts])
|
||||
monkeypatch.setattr(user_products, "get_products_by_brand",
|
||||
lambda brand, **k: sample_calls.append(brand) or [])
|
||||
|
||||
csv = "Brand,Product Name\n" + "".join(
|
||||
f"Lion Dates,Lion Dates {i}g\n" for i in range(10)
|
||||
)
|
||||
|
||||
resp = _upload(client, user_headers, csv)
|
||||
|
||||
assert resp.json()["added_count"] == 10
|
||||
assert embed_calls == [10], "embeddings must be generated in one batched call"
|
||||
assert len(store) == 1, "one write per brand, not one per row"
|
||||
assert len(sample_calls) == 1, "one brand-sample read per brand, not one per row"
|
||||
|
||||
|
||||
def test_products_still_save_when_embeddings_are_disabled(client, user_headers, monkeypatch, store):
|
||||
"""USE_EMBEDDINGS=false must skip the model, not the row."""
|
||||
def fail_if_called(texts):
|
||||
raise AssertionError("embed_texts must not be called when USE_EMBEDDINGS is false")
|
||||
|
||||
monkeypatch.setattr(user_products, "USE_EMBEDDINGS", False)
|
||||
monkeypatch.setattr(user_products, "embed_texts", fail_if_called)
|
||||
|
||||
resp = _upload(client, user_headers, TEMPLATE_CSV)
|
||||
|
||||
assert resp.status_code == 201, resp.text
|
||||
assert resp.json()["added_count"] == 2
|
||||
assert all(p["embedding"] is None for _, products in store for p in products)
|
||||
|
||||
|
||||
def test_a_failed_embedding_does_not_lose_the_product(client, user_headers, monkeypatch, store):
|
||||
"""Semantic search is a feature of the row; it is not the row."""
|
||||
monkeypatch.setattr(user_products, "USE_EMBEDDINGS", True)
|
||||
monkeypatch.setattr(user_products, "embed_texts",
|
||||
lambda texts: (_ for _ in ()).throw(RuntimeError("model not downloaded")))
|
||||
|
||||
resp = _upload(client, user_headers, TEMPLATE_CSV)
|
||||
|
||||
assert resp.status_code == 201, resp.text
|
||||
assert resp.json()["added_count"] == 2
|
||||
|
||||
|
||||
def test_the_seed_catalog_is_only_written_for_rows_the_database_took(
|
||||
client, user_headers, monkeypatch, store
|
||||
):
|
||||
"""Dual persistence must not become divergent persistence."""
|
||||
catalog_writes = []
|
||||
monkeypatch.setattr(user_products, "upsert_products_into_catalog_file",
|
||||
lambda brand, products: catalog_writes.append((brand, len(products))))
|
||||
|
||||
def unavailable(brand, products, cleanup=False):
|
||||
raise VectorStoreUnavailable("database unreachable")
|
||||
|
||||
monkeypatch.setattr(user_products, "upsert_brand_products", unavailable)
|
||||
|
||||
resp = _upload(client, user_headers, TEMPLATE_CSV)
|
||||
|
||||
assert resp.status_code == 503
|
||||
assert catalog_writes == [], "the JSON catalog must not gain products the database refused"
|
||||
Reference in New Issue
Block a user