backend updates for bulk product uploads from user

This commit is contained in:
sriram
2026-08-17 15:23:10 +05:30
parent 5e373ea20c
commit dba8d36175
7 changed files with 1290 additions and 218 deletions

View File

@@ -19,7 +19,20 @@ async def _run_job(job_id: str, brand: str, max_products: int) -> None:
job_store.update(job_id, "running")
try:
summary = await ingest_brand(brand, max_products=max_products)
job_store.update(job_id, "done", detail=f"{summary['total_products']} products ingested")
# "Ingested" has to mean "in the database". The generation stages can all
# succeed while the pgvector write fails, and reporting that as done is
# how a run that stored nothing ends up looking successful in the UI.
if summary.get("storage_error"):
job_store.update(
job_id,
"failed",
detail=(
f"Generated {summary['total_products']} product(s) but storing them "
f"failed, so none are in the catalog: {summary['storage_error']}"
),
)
else:
job_store.update(job_id, "done", detail=f"{summary['total_products']} products ingested")
except Exception as e: # noqa: BLE001 - surface any failure to the UI
logger.exception("Catalog ingestion job %s failed", job_id)
job_store.update(job_id, "failed", detail=str(e))

File diff suppressed because it is too large Load Diff

View File

@@ -940,11 +940,17 @@ class ProductCatalogEngine:
if 'image_id' not in p:
id_source = p.get('product_name') or p.get('title', 'unknown_product')
p['image_id'] = s3_service.generate_image_id(id_source)
upsert_brand_products(brand, enhanced_products, cleanup=True)
logger.info("🧠 Stored embeddings to pgvector")
stored = upsert_brand_products(brand, enhanced_products, cleanup=True)
logger.info("🧠 Stored %s product(s) with embeddings to pgvector", stored)
except Exception as e:
logger.warning(f"Vector storage skipped/failed: {e}")
# Stage 4 is where the catalog becomes readable by the app - every
# API read goes to pgvector, not to the dict returned here. So a
# failure at this stage means the run produced nothing the user can
# see, and it has to travel back to the job status rather than being
# logged and forgotten.
logger.error("❌ Stage 4 (pgvector storage) failed for '%s': %s", brand, e)
catalog['storage_error'] = str(e)
return catalog
def save_catalog(self, catalog: Dict[str, Any], filename: str = None) -> str:

View File

@@ -40,11 +40,17 @@ async def ingest_brand(brand: str, max_products: int = 50) -> Dict[str, Any]:
# Placed here rather than in the API router because this function is also
# the CLI's entry point (cli/ingest_brand.py), and a failure to write the
# file must not turn a successful ingest into a failed job.
try:
from app.services.brand_sync import export_brand_to_seed_file
export_brand_to_seed_file(brand)
except Exception: # noqa: BLE001 - DB rows are already committed
logger.warning("Seed-catalog export failed for %s (DB rows intact)", brand, exc_info=True)
storage_error = catalog.get("storage_error")
# Nothing reached the database, so there is nothing to mirror out of it -
# and running the export anyway would either write an empty file or leave a
# stale one looking current.
if not storage_error:
try:
from app.services.brand_sync import export_brand_to_seed_file
export_brand_to_seed_file(brand)
except Exception: # noqa: BLE001 - DB rows are already committed
logger.warning("Seed-catalog export failed for %s (DB rows intact)", brand, exc_info=True)
summary = {
"brand": brand,
@@ -52,6 +58,7 @@ async def ingest_brand(brand: str, max_products: int = 50) -> Dict[str, Any]:
"total_images": catalog.get("total_images", 0),
"duration_seconds": round(duration, 2),
"engine_info": catalog.get("engine_info", {}),
"storage_error": storage_error,
}
logger.info("Finished ingestion for brand=%s in %.2fs: %s products",
brand, duration, summary["total_products"])

View File

@@ -22,6 +22,28 @@ from app.services.s3_service import s3_service
logger = logging.getLogger(__name__)
class VectorStoreUnavailable(RuntimeError):
"""The pgvector database could not be reached, so a write did not happen.
Exists because the silent alternative was a production bug that took a long
time to see: upsert_brand_products() used to `return` when _connect() gave
back None - unreachable host, wrong DB_PASSWORD, USE_PGVECTOR=false - and
every caller read that as a successful write. The upload endpoints then
answered "success", the seed JSON was updated, and not one row existed in
the database. A write that cannot happen has to raise.
"""
class VectorStoreWriteFailed(RuntimeError):
"""The INSERT ran without error but the rows are not in the table.
Guards against the failure modes an exception cannot catch: a statement
silently rolled back, a trigger swallowing the row, or an ON CONFLICT
target that quietly matched nothing. The only trustworthy proof of a write
is reading it back.
"""
def _sanitize_name(name: str) -> str:
"""Sanitize a brand name for use as a PostgreSQL table name suffix.
@@ -193,22 +215,44 @@ def ensure_brand_schema(brand: str) -> str:
return table_name
def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: bool = False) -> None:
def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: bool = False) -> int:
"""Insert products into brand-specific table - simplified with only essential fields
When `cleanup=True`, any products in the table whose image_id is NOT in the
provided `products` list are deleted after the upsert. This ensures the
database exactly reflects the source data. The caller is responsible for
providing the complete set of products for the brand when using cleanup.
Returns the number of distinct image_ids confirmed present in the table
afterwards - the rows are read back, so a non-raising call is proof of
persistence rather than proof that a statement was merely sent.
Raises:
VectorStoreUnavailable: the database is unreachable; nothing was written.
VectorStoreWriteFailed: the statements ran but the rows are not there.
ValueError: a product carries no image_id, which is the primary key
every other product is deduplicated on.
"""
if not products:
return 0
conn = _connect()
if not conn:
return
raise VectorStoreUnavailable(
f"Cannot save products for '{brand}': the product database is unreachable "
f"(USE_PGVECTOR={USE_PGVECTOR}, host={DB_HOST}:{DB_PORT}, db={DB_NAME}). "
f"Nothing was saved. Check DB_HOST/DB_USER/DB_PASSWORD in backend/.env "
f"and that Postgres is accepting connections."
)
table_name = ensure_brand_schema(brand)
if not table_name:
return
conn.close()
raise VectorStoreUnavailable(
f"Cannot save products for '{brand}': the brand table could not be created "
f"or verified in database '{DB_NAME}'. Nothing was saved."
)
rows = []
for p in products:
# Extract only essential fields
@@ -333,55 +377,89 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
image_ids = [r[4] for r in rows if r[4]]
with conn, conn.cursor() as cur:
cur.executemany(
f"""
INSERT INTO {table_name}
(product_name, title, description, category, image_id, image_url, image_urls, price_range, size_variants, providers,
fssai_license, product_sku, sku_source, hsn_code, final_selling_price, selling_price, barcode, barcode_type, highlights, nutrients, search_query, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (image_id) DO UPDATE SET
product_name = EXCLUDED.product_name,
title = EXCLUDED.title,
description = EXCLUDED.description,
category = EXCLUDED.category,
image_url = EXCLUDED.image_url,
image_urls = EXCLUDED.image_urls,
price_range = EXCLUDED.price_range,
size_variants = EXCLUDED.size_variants,
providers = EXCLUDED.providers,
fssai_license = EXCLUDED.fssai_license,
product_sku = EXCLUDED.product_sku,
sku_source = EXCLUDED.sku_source,
hsn_code = EXCLUDED.hsn_code,
final_selling_price = EXCLUDED.final_selling_price,
selling_price = EXCLUDED.selling_price,
barcode = EXCLUDED.barcode,
barcode_type = EXCLUDED.barcode_type,
highlights = EXCLUDED.highlights,
nutrients = EXCLUDED.nutrients,
search_query = EXCLUDED.search_query,
embedding = EXCLUDED.embedding,
updated_at = CURRENT_TIMESTAMP
""",
rows,
# An empty image_id is not a harmless blank: it is the conflict target, so
# two such products would overwrite each other and the second would replace
# the first instead of being added. Refuse the batch and name the rows.
if len(image_ids) != len(rows):
unnamed = [r[0] or "<no product_name>" for r in rows if not r[4]]
conn.close()
raise ValueError(
f"{len(unnamed)} product(s) for '{brand}' have no image_id and cannot be "
f"stored (a product needs a name with at least one letter or digit): "
f"{', '.join(unnamed[:5])}"
)
logger.info(f"✅ Upserted {len(rows)} products into {table_name}")
# Remove stale products that were deleted from the source data.
# Only runs when cleanup=True so that callers processing partial
# product sets (e.g. multiple seed files contributing to the same
# brand table) don't accidentally orphan each other's data.
if cleanup and image_ids:
cur.execute(
f"DELETE FROM {table_name} WHERE image_id != ALL(%s::text[])",
(image_ids,),
expected_ids = sorted(set(image_ids))
try:
with conn, conn.cursor() as cur:
cur.executemany(
f"""
INSERT INTO {table_name}
(product_name, title, description, category, image_id, image_url, image_urls, price_range, size_variants, providers,
fssai_license, product_sku, sku_source, hsn_code, final_selling_price, selling_price, barcode, barcode_type, highlights, nutrients, search_query, embedding)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (image_id) DO UPDATE SET
product_name = EXCLUDED.product_name,
title = EXCLUDED.title,
description = EXCLUDED.description,
category = EXCLUDED.category,
image_url = EXCLUDED.image_url,
image_urls = EXCLUDED.image_urls,
price_range = EXCLUDED.price_range,
size_variants = EXCLUDED.size_variants,
providers = EXCLUDED.providers,
fssai_license = EXCLUDED.fssai_license,
product_sku = EXCLUDED.product_sku,
sku_source = EXCLUDED.sku_source,
hsn_code = EXCLUDED.hsn_code,
final_selling_price = EXCLUDED.final_selling_price,
selling_price = EXCLUDED.selling_price,
barcode = EXCLUDED.barcode,
barcode_type = EXCLUDED.barcode_type,
highlights = EXCLUDED.highlights,
nutrients = EXCLUDED.nutrients,
search_query = EXCLUDED.search_query,
embedding = EXCLUDED.embedding,
updated_at = CURRENT_TIMESTAMP
""",
rows,
)
deleted = cur.rowcount
if deleted:
logger.info(f"🗑️ Removed {deleted} stale product(s) from {table_name}")
conn.close()
# Remove stale products that were deleted from the source data.
# Only runs when cleanup=True so that callers processing partial
# product sets (e.g. multiple seed files contributing to the same
# brand table) don't accidentally orphan each other's data.
if cleanup and image_ids:
cur.execute(
f"DELETE FROM {table_name} WHERE image_id != ALL(%s::text[])",
(image_ids,),
)
deleted = cur.rowcount
if deleted:
logger.info(f"🗑️ Removed {deleted} stale product(s) from {table_name}")
# Read the rows back. This is the line that turns "we sent an
# INSERT" into "the data is in the table", and it is the only
# signal the API layer is allowed to report success on.
cur.execute(
f"SELECT COUNT(DISTINCT image_id) FROM {table_name} WHERE image_id = ANY(%s::text[])",
(expected_ids,),
)
row = cur.fetchone()
persisted = int(row[0]) if row else 0
if persisted < len(expected_ids):
raise VectorStoreWriteFailed(
f"Wrote {len(rows)} product(s) for '{brand}' to {table_name} but only "
f"{persisted} of {len(expected_ids)} are readable back afterwards. "
f"The data was not saved - treat this as a failed import."
)
logger.info("✅ Upserted %d product(s) into %s (%d verified in table)",
len(rows), table_name, persisted)
finally:
conn.close()
# Every write path into the catalog funnels through here, so this is the
# one place that has to invalidate the derived views: the brand cards'
@@ -393,6 +471,8 @@ def upsert_brand_products(brand: str, products: List[Dict[str, Any]], cleanup: b
except Exception: # noqa: BLE001 - cache invalidation must never fail a write
pass
return persisted
def get_existing_product_image_id(brand: str, product_name: str) -> Optional[str]:
"""Check if a product with this name exists in the brand table and return its image_id"""

View File

@@ -71,6 +71,14 @@ boto3>=1.34.162
# --- missing from this file - declared explicitly now. ---
scikit-learn>=1.5.2
pandas>=2.2.2
# pandas' Excel readers are optional extras it does not install itself, and the
# upload endpoints (/api/user/products/upload-file, /api/upload/*) accept .xlsx
# and .xls. Undeclared, they happened to be present in some environments and
# absent in others - so an Excel upload that worked locally failed in the
# container with "Missing optional dependency 'openpyxl'". openpyxl reads
# .xlsx/.xlsm; xlrd is only for the legacy .xls format.
openpyxl>=3.1.5
xlrd>=2.0.1
numpy>=1.26.4
scipy>=1.13.1
joblib>=1.4.2

View File

@@ -0,0 +1,370 @@
"""
Tests for the user product upload path.
The bug these exist for: uploading a spreadsheet returned a success message
while nothing reached the database. So the assertions here are deliberately not
"did it answer 2xx" - they are "did it answer 2xx *and* hand the rows to the
store", and, for every failure mode, "did it refuse to call that a success".
Hermetic, like the rest of the suite: the store, the embedding model and the
seed-catalog writer are all substituted, so nothing here touches a real
database, downloads a model, or writes into data/seed_catalogs/.
"""
from __future__ import annotations
import io
import pytest
from app.api.routers import user_products
from app.services.vector_store import VectorStoreUnavailable
UPLOAD_URL = "/api/user/products/upload-file"
# The headers the frontend's "Download Sample CSV" button produces.
TEMPLATE_CSV = (
"Brand Name,Product Name / Variant,Category,Price Range,Final Price (₹),"
"Barcode (GTIN/EAN),HSN Code,Custom Image URL,Description\n"
"Lion Dates,Lion Dates 450g,Health Foods,₹160-220,185.00,20086040,2008,,Premium dates\n"
"Naga,Naga Maida 2kg,Flour & Grains,₹90-110,98.00,8906012345001,1101,,Refined wheat flour\n"
)
@pytest.fixture
def store(monkeypatch):
"""Capture what the endpoint hands to the database instead of writing it.
Also stubs the two slow collaborators. `embed_texts` would download and load
a sentence-transformer; `get_products_by_brand` would open a real connection
to whatever DB_HOST points at.
"""
calls = []
def fake_upsert(brand, products, cleanup=False):
calls.append((brand, products))
return len(products)
monkeypatch.setattr(user_products, "upsert_brand_products", fake_upsert)
monkeypatch.setattr(user_products, "embed_texts", lambda texts: [[0.0] * 384 for _ in texts])
monkeypatch.setattr(user_products, "get_products_by_brand", lambda *a, **k: [])
monkeypatch.setattr(user_products, "upsert_products_into_catalog_file", lambda brand, products: None)
return calls
def _upload(client, headers, content: str | bytes, name: str = "products.csv"):
data = content.encode("utf-8") if isinstance(content, str) else content
return client.post(UPLOAD_URL, files={"file": (name, data)}, headers=headers)
# ---------------------------------------------------------------------------
# The reported bug: success without persistence
# ---------------------------------------------------------------------------
def test_upload_is_not_a_success_when_the_database_is_unreachable(client, user_headers, monkeypatch, store):
"""The exact production symptom. An unreachable database must not answer 201."""
def unavailable(brand, products, cleanup=False):
raise VectorStoreUnavailable("the product database is unreachable (host=db:5432)")
monkeypatch.setattr(user_products, "upsert_brand_products", unavailable)
resp = _upload(client, user_headers, TEMPLATE_CSV)
assert resp.status_code == 503, resp.text
detail = resp.json()["detail"]
assert "Nothing was saved" in detail
assert "unreachable" in detail
def test_single_add_is_not_a_success_when_the_database_is_unreachable(client, user_headers, monkeypatch, store):
def unavailable(brand, products, cleanup=False):
raise VectorStoreUnavailable("the product database is unreachable (host=db:5432)")
monkeypatch.setattr(user_products, "upsert_brand_products", unavailable)
resp = client.post(
"/api/user/products/add",
json={"brand": "Lion Dates", "product_name": "Lion Dates 450g"},
headers=user_headers,
)
assert resp.status_code == 503, resp.text
assert "not saved" in resp.json()["detail"]
def test_upload_reaches_the_database_and_reports_what_it_stored(client, user_headers, store):
resp = _upload(client, user_headers, TEMPLATE_CSV)
assert resp.status_code == 201, resp.text
body = resp.json()
assert body["status"] == "success"
assert body["added_count"] == 2
assert body["error_count"] == 0
assert body["rows_total"] == 2
stored = {p["product_name"]: p for _, products in store for p in products}
assert set(stored) == {"Lion Dates 450g", "Naga Maida 2kg"}
dates = stored["Lion Dates 450g"]
assert dates["category"] == "Health Foods"
assert dates["final_selling_price"] == 185.0
assert dates["price_range"] == "₹160-220"
# Not "20086040.0" - an identifier read as a float and stringified is a
# different identifier.
assert dates["barcode"] == "20086040"
assert dates["hsn_code"] == "2008"
assert stored["Naga Maida 2kg"]["barcode"] == "8906012345001"
def test_a_real_xlsx_workbook_imports_with_its_identifiers_intact(client, user_headers, store):
"""The reported case was an .xlsx upload, and Excel is where identifiers rot.
A 13-digit barcode in a spreadsheet cell is a number to Excel, so it arrives
as a float and stringifies to "8906012345001.0"; an HSN code of 0402 loses
its leading zero. Both are then stored - silently - as a different value
than the one in the file.
"""
openpyxl = pytest.importorskip("openpyxl", reason="declared in requirements.txt for .xlsx uploads")
workbook = openpyxl.Workbook()
sheet = workbook.active
sheet.append(["Brand Name", "Product Name", "Final Price (₹)", "Barcode (GTIN/EAN)", "HSN Code"])
sheet.append(["Lion Dates", "Lion Dates 450g", 185, 8906012345001, "0402"])
buffer = io.BytesIO()
workbook.save(buffer)
resp = _upload(client, user_headers, buffer.getvalue(), name="products.xlsx")
assert resp.status_code == 201, resp.text
assert resp.json()["added_count"] == 1
stored = [p for _, products in store for p in products][0]
assert stored["barcode"] == "8906012345001"
assert stored["hsn_code"] == "0402"
assert stored["final_selling_price"] == 185.0
def test_partial_import_is_reported_as_partial_not_success(client, user_headers, monkeypatch, store):
"""One brand failing must neither sink the other nor be called a success."""
def selective(brand, products, cleanup=False):
if "naga" in brand.lower():
raise RuntimeError("column overflow on selling_price")
return len(products)
monkeypatch.setattr(user_products, "upsert_brand_products", selective)
resp = _upload(client, user_headers, TEMPLATE_CSV)
assert resp.status_code == 201, resp.text
body = resp.json()
assert body["status"] == "partial"
assert body["added_count"] == 1
assert body["error_count"] == 1
assert body["errors"][0]["row"] == 3 # header is row 1, Naga is row 3
assert "column overflow" in body["errors"][0]["error"]
# ---------------------------------------------------------------------------
# Header handling
# ---------------------------------------------------------------------------
def test_template_headers_all_map_to_their_field():
mapping = user_products.map_spreadsheet_columns([
"Brand Name", "Product Name / Variant", "Category", "Price Range",
"Final Price (₹)", "Barcode (GTIN/EAN)", "HSN Code", "Custom Image URL",
"Description",
])
assert mapping.columns == {
"brand": "Brand Name",
"product_name": "Product Name / Variant",
"category": "Category",
"price_range": "Price Range",
"final_selling_price": "Final Price (₹)",
"barcode": "Barcode (GTIN/EAN)",
"hsn_code": "HSN Code",
"image_url": "Custom Image URL",
"description": "Description",
}
assert mapping.ignored == []
def test_a_product_sku_column_is_not_mistaken_for_the_product_name():
"""'Product SKU' contains 'product'. Matched loosely, it used to become a
second product_name column, and duplicate columns are what turned a row
into a stringified Series."""
mapping = user_products.map_spreadsheet_columns(["Brand", "Product Name", "Product SKU"])
assert mapping.columns["product_name"] == "Product Name"
assert mapping.columns["product_sku"] == "Product SKU"
def test_colliding_columns_are_reported_and_do_not_corrupt_rows(client, user_headers, store):
"""Two columns for one field: keep the first, say so, keep importing."""
csv = (
"Brand,Product Name,Item Name,Final Price\n"
"Lion Dates,Lion Dates 450g,Ignore This One,185\n"
)
resp = _upload(client, user_headers, csv)
assert resp.status_code == 201, resp.text
body = resp.json()
assert body["added_count"] == 1
assert body["ignored_columns"] == [
{"column": "Item Name", "field": "product_name", "using_instead": "Product Name"}
]
assert any("ignored" in w for w in body["warnings"])
stored = [p for _, products in store for p in products]
assert stored[0]["product_name"] == "Lion Dates 450g"
def test_missing_required_columns_is_a_400_naming_the_headers(client, user_headers, store):
resp = _upload(client, user_headers, "Foo,Bar\n1,2\n")
assert resp.status_code == 400
detail = resp.json()["detail"]
assert "brand column" in detail
assert "Foo" in detail
# ---------------------------------------------------------------------------
# Row-level handling
# ---------------------------------------------------------------------------
def test_rows_missing_a_brand_are_errors_not_silent_skips(client, user_headers, store):
csv = (
"Brand,Product Name\n"
"Lion Dates,Lion Dates 450g\n"
",Orphan Product\n"
)
resp = _upload(client, user_headers, csv)
body = resp.json()
assert resp.status_code == 201, resp.text
assert body["added_count"] == 1
assert body["error_count"] == 1
assert body["errors"][0]["row"] == 3
assert "brand" in body["errors"][0]["error"]
def test_fully_blank_rows_are_skipped_and_counted(client, user_headers, store):
csv = (
"Brand,Product Name\n"
"Lion Dates,Lion Dates 450g\n"
",\n"
",\n"
)
resp = _upload(client, user_headers, csv)
body = resp.json()
assert body["added_count"] == 1
assert body["error_count"] == 0
assert body["skipped_blank_rows"] == 2
def test_a_file_where_every_row_fails_is_never_a_success(client, user_headers, store):
csv = (
"Brand,Product Name\n"
",Orphan One\n"
",Orphan Two\n"
)
resp = _upload(client, user_headers, csv)
assert resp.status_code == 422, resp.text
assert "Nothing was saved" in resp.json()["detail"]
assert store == []
def test_empty_and_headers_only_files_are_rejected(client, user_headers, store):
assert _upload(client, user_headers, b"").status_code == 400
assert _upload(client, user_headers, "Brand,Product Name\n").status_code == 400
def test_unsupported_file_type_is_rejected(client, user_headers, store):
resp = _upload(client, user_headers, b"%PDF-1.4", name="products.pdf")
assert resp.status_code == 400
assert "Unsupported file type" in resp.json()["detail"]
def test_oversized_row_count_is_rejected_before_any_write(client, user_headers, store):
rows = "".join(f"Lion Dates,Product {i}\n" for i in range(user_products.MAX_UPLOAD_ROWS + 1))
resp = _upload(client, user_headers, "Brand,Product Name\n" + rows)
assert resp.status_code == 413
assert store == []
# ---------------------------------------------------------------------------
# Batching: the collaborators must be called once per upload, not once per row
# ---------------------------------------------------------------------------
def test_one_embedding_call_and_one_write_per_brand(client, user_headers, monkeypatch, store):
embed_calls = []
sample_calls = []
# Pinned rather than inherited: the assertion below is about batching, and
# it should not silently pass because a .env happened to disable embeddings.
monkeypatch.setattr(user_products, "USE_EMBEDDINGS", True)
monkeypatch.setattr(user_products, "embed_texts",
lambda texts: embed_calls.append(len(texts)) or [[0.0] * 384 for _ in texts])
monkeypatch.setattr(user_products, "get_products_by_brand",
lambda brand, **k: sample_calls.append(brand) or [])
csv = "Brand,Product Name\n" + "".join(
f"Lion Dates,Lion Dates {i}g\n" for i in range(10)
)
resp = _upload(client, user_headers, csv)
assert resp.json()["added_count"] == 10
assert embed_calls == [10], "embeddings must be generated in one batched call"
assert len(store) == 1, "one write per brand, not one per row"
assert len(sample_calls) == 1, "one brand-sample read per brand, not one per row"
def test_products_still_save_when_embeddings_are_disabled(client, user_headers, monkeypatch, store):
"""USE_EMBEDDINGS=false must skip the model, not the row."""
def fail_if_called(texts):
raise AssertionError("embed_texts must not be called when USE_EMBEDDINGS is false")
monkeypatch.setattr(user_products, "USE_EMBEDDINGS", False)
monkeypatch.setattr(user_products, "embed_texts", fail_if_called)
resp = _upload(client, user_headers, TEMPLATE_CSV)
assert resp.status_code == 201, resp.text
assert resp.json()["added_count"] == 2
assert all(p["embedding"] is None for _, products in store for p in products)
def test_a_failed_embedding_does_not_lose_the_product(client, user_headers, monkeypatch, store):
"""Semantic search is a feature of the row; it is not the row."""
monkeypatch.setattr(user_products, "USE_EMBEDDINGS", True)
monkeypatch.setattr(user_products, "embed_texts",
lambda texts: (_ for _ in ()).throw(RuntimeError("model not downloaded")))
resp = _upload(client, user_headers, TEMPLATE_CSV)
assert resp.status_code == 201, resp.text
assert resp.json()["added_count"] == 2
def test_the_seed_catalog_is_only_written_for_rows_the_database_took(
client, user_headers, monkeypatch, store
):
"""Dual persistence must not become divergent persistence."""
catalog_writes = []
monkeypatch.setattr(user_products, "upsert_products_into_catalog_file",
lambda brand, products: catalog_writes.append((brand, len(products))))
def unavailable(brand, products, cleanup=False):
raise VectorStoreUnavailable("database unreachable")
monkeypatch.setattr(user_products, "upsert_brand_products", unavailable)
resp = _upload(client, user_headers, TEMPLATE_CSV)
assert resp.status_code == 503
assert catalog_writes == [], "the JSON catalog must not gain products the database refused"