Catalog feature updates on column fields

This commit is contained in:
sriram
2026-09-08 15:18:29 +05:30
parent 2749bee1a3
commit 10b24c6348
60 changed files with 9224 additions and 31 deletions

View File

@@ -53,6 +53,11 @@ from app.services.vector_store import (
get_products_by_brand,
)
from app.services.brand_sync import upsert_products_into_catalog_file
from app.services.enrichment.catalog_consensus import (
consensus_rows,
consensus_value,
fssai_for_brand,
)
from app.services.embeddings_service import embed_texts
from app.services.s3_service import s3_service
@@ -409,12 +414,20 @@ class _PersistOutcome:
unavailable: Optional[str] = None
# How many of a brand's rows to read when working out what they agree on.
# Bounded because this is a nicety, not the write: the largest brand table here
# holds ~250 rows, so this reads all of them for every real brand while still
# refusing to degenerate into the full `SELECT *` that once ran per uploaded
# row. One query per brand, not per row - that part is load-bearing.
_CONSENSUS_SAMPLE_LIMIT = 300
def _brand_sample(brand_parent: str) -> Dict[str, Any]:
"""The most recently updated product for a brand, used to inherit defaults.
`limit=1` matters: this used to be a full `SELECT *` of the brand table,
executed once per uploaded row. A 200-row file against a brand with a few
thousand products meant 200 full table reads before a single insert.
Kept for the fields where one arbitrary sibling is a defensible default
(category, price band, size). For `fssai_license` and `providers` it is
NOT defensible - see `_brand_defaults`.
"""
try:
existing = get_products_by_brand(brand_parent, limit=1)
@@ -424,8 +437,51 @@ def _brand_sample(brand_parent: str) -> Dict[str, Any]:
return existing[0] if existing else {}
def _brand_defaults(brand_parent: str) -> Dict[str, Any]:
"""One arbitrary sibling row, plus what the brand's rows actually AGREE on.
The distinction matters for exactly two fields.
`fssai_license` identifies the food business legally answerable for the
product. Inheriting it from one arbitrary sibling is already weak; the code
this replaces was worse - it fell back to the bare constant
"10012042000244", which is Lion Dates' real registered licence, and stamped
it onto any brand with no sample row. `scripts/merge_haldiram.py` exists
because that reached production.
`providers` is a claim about where a product can be bought. The replaced
default asserted all six of Amazon/Flipkart/BigBasket/Jiomart/Blinkit/Zepto
for every product nobody had checked.
Consensus over the brand's own rows is the honest version of both, and it
declines to answer when the rows disagree.
"""
# Two reads on purpose, and the split matters on a memory-capped host.
#
# `_brand_sample` is SELECT * limit 1 - one row, embedding and all, because
# the fields it seeds (category, price band, size) need the whole row.
#
# The consensus read is 300 rows, so it takes only the two columns it
# actually inspects. Measured: SELECT * over 244 rows costs 3.0 MB, of
# which 4.7 KB per row is an embedding string nothing here reads. Two named
# columns is roughly 50 KB for the same rows.
rows = consensus_rows(brand_parent, ["fssai_license", "providers"],
limit=_CONSENSUS_SAMPLE_LIMIT)
fssai, fssai_source = fssai_for_brand(brand_parent, rows)
providers, _why = consensus_value("providers", rows)
return {
"sample": _brand_sample(brand_parent),
"fssai_license": fssai,
"fssai_source": fssai_source,
"providers": providers,
}
def _build_product_dict(req: AddProductRequest, brand_parent: str,
sample_existing: Dict[str, Any]) -> Dict[str, Any]:
sample_existing: Dict[str, Any],
defaults: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
"""Fill in everything the catalog needs that the user did not supply.
Pure apart from the optional S3 image lookup - no database access and no
@@ -438,8 +494,22 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
raise ValueError(f"product name {product_name!r} has no letters or digits to identify it by")
image_id = f"{brand_slug}_{product_slug}"
defaults = defaults or {}
category = req.category or sample_existing.get("category") or "Health Foods"
fssai_license = req.fssai_license or sample_existing.get("fssai_license") or "10012042000244"
# NO CONSTANT FALLBACK HERE, EVER.
#
# This line used to end `or "10012042000244"` - Lion Dates' real registered
# FSSAI licence - so any brand without a sample row was silently attributed
# to a food business that had never heard of the product. An FSSAI number
# is who is legally answerable for what is in the packet; inventing one is
# not a cosmetic default.
#
# The order now is: what the uploader supplied, then the curated brand
# registry, then what the brand's own rows unambiguously agree on, then
# NOTHING. A blank licence is a gap someone can fill; a confidently wrong
# one is a liability nobody knows to look for.
fssai_license = req.fssai_license or defaults.get("fssai_license") or None
description = req.description or (
f"Introducing {product_name} from the trusted {brand_parent} brand. "
@@ -465,7 +535,11 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
else:
price_range = "₹100-250"
providers = req.providers or list(sample_existing.get("providers") or ["Amazon", "Flipkart", "BigBasket", "Jiomart", "Blinkit", "Zepto"])
# Likewise no invented marketplace list. Claiming a product is stocked by
# Amazon, Flipkart, BigBasket, Jiomart, Blinkit AND Zepto because nobody
# checked is a false availability claim on every row it touches. Consensus
# across the brand's own rows, or empty.
providers = req.providers or list(defaults.get("providers") or [])
highlights = req.highlights or list(sample_existing.get("highlights") or ["100% Quality Assurance", "Authentic Brand Product"])
nutrients = req.nutrients or list(sample_existing.get("nutrients") or ["Energy - High", "Protein - Good Source"])
@@ -582,11 +656,13 @@ def _persist_products(items: List[Tuple[Optional[int], AddProductRequest]]) -> _
# row) and one embedding call for the whole upload.
built: "OrderedDict[str, List[Tuple[Optional[int], AddProductRequest, Dict[str, Any]]]]" = OrderedDict()
for brand_parent, rows in groups.items():
sample = _brand_sample(brand_parent)
defaults = _brand_defaults(brand_parent)
sample = defaults["sample"]
prepared = []
for row_number, req in rows:
try:
prepared.append((row_number, req, _build_product_dict(req, brand_parent, sample)))
prepared.append((row_number, req,
_build_product_dict(req, brand_parent, sample, defaults)))
except Exception as exc: # noqa: BLE001 - one bad row, not the file
outcome.failures.append({
"row": row_number,