Brand valid image generation

This commit is contained in:
sriram
2026-09-11 15:47:56 +05:30
parent e1a5962f82
commit ce4fa70dee
31 changed files with 5586 additions and 1495 deletions

View File

@@ -17,14 +17,20 @@ now the only admin way in. What made it the survivor is the unit of work: five
files are one batch with one id, so the question a colleague actually asks -
"did the drop land?" - has one answer rather than five.
WHY THE NETWORK STAGES DEFAULT OFF HERE
---------------------------------------
Turning both on for a single file is a considered trade. At twenty files it is
thousands of outbound requests and, for the image stage, a Playwright subprocess
that can burn three minutes on its own - on a single-vCPU container that is also
serving the API. So a batch opts IN to those stages; it does not opt out.
`USE_OLLAMA` is false in production anyway, which makes `use_llm` a no-op there
and the honest default obvious.
WHY IMAGES DEFAULT OFF HERE AND THE LLM DEFAULTS ON
---------------------------------------------------
Image search for a single file is a considered trade. At twenty files it is
thousands of outbound requests and a Playwright subprocess that can burn three
minutes on its own - on a single-vCPU container that is also serving the API.
So a batch opts IN to that stage; it does not opt out.
`use_llm` is different and defaults ON: it gates only the description written
in stage 2 for rows the sheet left blank, which is the one field a shopper
reads and a store almost never supplies. It is cheap to leave on because
`ollama_service._ensure_client` caches its reachability probe per batch and
`store_catalog_pipeline.LlmBreaker` stops calling after three consecutive
misses, so with `USE_OLLAMA` false (production today) the cost is one warning
per file and the rows fall back to a factual template.
Note that these defaults bind THIS router only. The open upload endpoint runs
itself and takes its two flags from `UPLOAD_AUTORUN_FETCH_IMAGES` and
@@ -145,7 +151,7 @@ async def preview_catalog_batch(files: List[UploadFile] = File(...)) -> dict:
dependencies=[Depends(require_admin)])
async def ingest_catalog_batch(
files: List[UploadFile] = File(...),
use_llm: bool = False,
use_llm: bool = True,
fetch_images: bool = False,
) -> BatchOut:
"""Stage the files, queue the batch, and return an id to poll.
@@ -325,7 +331,8 @@ class InboxStartRequest(InboxSelection):
# Chosen HERE, not by the sender - see the note on the POST handler in
# uploads.py. These commit the host to outbound work, so the decision
# belongs to the person who can see what the machine is already doing.
use_llm: bool = False
# The LLM is on by default for the reason in the module docstring.
use_llm: bool = True
fetch_images: bool = False
# Who runs it. "inprocess" is this container's worker thread and is the
# default, so an existing client that never sends the field is unaffected.

View File

@@ -124,11 +124,14 @@ def _normalize_header(raw: Any) -> str:
_EXACT_HEADERS: Dict[str, str] = {
# brand
# brand. "productbrand" is what a camel-cased ProductBrand header becomes
# after _normalize_header, and is listed so the mapping does not depend on
# the keyword rules' ordering.
"brand": "brand", "brand name": "brand", "brands": "brand",
"company": "brand", "manufacturer": "brand", "company name": "brand",
"product brand": "brand", "productbrand": "brand",
# product name
"product": "product_name", "product name": "product_name",
"product": "product_name", "product name": "product_name", "productname": "product_name",
"product name variant": "product_name", "product variant": "product_name",
"variant": "product_name", "name": "product_name", "item": "product_name",
"item name": "product_name", "product title": "product_name",
@@ -138,18 +141,27 @@ _EXACT_HEADERS: Dict[str, str] = {
# category
"category": "category", "categories": "category", "cat": "category",
"product category": "category", "segment": "category",
# price
# price. Two columns, two meanings: `selling_price` is what the shop
# charges - a retail / sale / selling price, or a bare "Price" - and
# `final_selling_price` is the tax-inclusive ceiling (MRP, "final price").
# A sheet that carries only one of them still fills both, because
# vector_store.upsert_brand_products copies selling -> final when final is
# blank. Before this split every price header landed in final_selling_price
# and selling_price was NULL for every uploaded row.
"price range": "price_range", "range": "price_range", "mrp range": "price_range",
"price": "final_selling_price", "final price": "final_selling_price",
"final price rs": "final_selling_price", "final selling price": "final_selling_price",
"selling price": "final_selling_price", "mrp": "final_selling_price",
"rate": "final_selling_price", "amount": "final_selling_price",
"cost": "final_selling_price", "unit price": "final_selling_price",
"selling price": "selling_price", "sellingprice": "selling_price",
"retail price": "selling_price", "retailprice": "selling_price",
"sale price": "selling_price", "sp": "selling_price",
"price": "selling_price", "unit price": "selling_price", "rate": "selling_price",
"final price": "final_selling_price", "final price rs": "final_selling_price",
"final selling price": "final_selling_price", "mrp": "final_selling_price",
"amount": "final_selling_price", "cost": "final_selling_price",
# identifiers
"barcode": "barcode", "barcode gtin ean": "barcode", "bar code": "barcode",
"gtin": "barcode", "ean": "barcode", "upc": "barcode", "ean13": "barcode",
"hsn": "hsn_code", "hsn code": "hsn_code", "hsn sac": "hsn_code",
"sku": "product_sku", "product sku": "product_sku", "sku code": "product_sku",
"sku": "product_sku", "product sku": "product_sku", "productsku": "product_sku",
"sku code": "product_sku",
"fssai": "fssai_license", "fssai license": "fssai_license",
"fssai license number": "fssai_license", "fssai number": "fssai_license",
# text / media
@@ -173,7 +185,13 @@ _KEYWORD_RULES: Tuple[Tuple[str, Tuple[str, ...]], ...] = (
("barcode", ("barcode", "bar code", "gtin", "ean", "upc")),
("product_sku", ("sku",)),
("price_range", ("price range", "range")),
("final_selling_price", ("final price", "selling price", "price", "mrp", "rate", "cost")),
# Order is load-bearing: "final selling price" contains "selling price",
# and every price header contains "price", so the final-price forms must
# be claimed before the generic ones. "sp" is deliberately exact-only - as
# a substring it is inside "spice", "display" and "sponge".
("final_selling_price", ("final price", "final selling price", "mrp")),
("selling_price", ("selling price", "retail price", "sale price", "price", "rate")),
("final_selling_price", ("cost",)),
("image_url", ("image", "photo", "picture", "url", "link")),
("description", ("description", "desc", "detail")),
("category", ("category", "segment")),
@@ -346,6 +364,7 @@ def row_to_request(row: Dict[str, Any], mapping: _ColumnMapping) -> AddProductRe
product_sku=_text(row, mapping, "product_sku"),
hsn_code=_text(row, mapping, "hsn_code"),
final_selling_price=_number(row, mapping, "final_selling_price"),
selling_price=_number(row, mapping, "selling_price"),
barcode=_text(row, mapping, "barcode"),
image_url=_text(row, mapping, "image_url"),
)
@@ -528,8 +547,9 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
price_range = req.price_range
if not price_range:
if req.final_selling_price:
price_range = f"₹{req.final_selling_price}"
sheet_price = req.final_selling_price or req.selling_price
if sheet_price:
price_range = f"₹{sheet_price}"
elif sample_existing.get("price_range"):
price_range = sample_existing.get("price_range")
else: