Catalog feature updates on column fields
This commit is contained in:
@@ -53,6 +53,11 @@ from app.services.vector_store import (
|
||||
get_products_by_brand,
|
||||
)
|
||||
from app.services.brand_sync import upsert_products_into_catalog_file
|
||||
from app.services.enrichment.catalog_consensus import (
|
||||
consensus_rows,
|
||||
consensus_value,
|
||||
fssai_for_brand,
|
||||
)
|
||||
from app.services.embeddings_service import embed_texts
|
||||
from app.services.s3_service import s3_service
|
||||
|
||||
@@ -409,12 +414,20 @@ class _PersistOutcome:
|
||||
unavailable: Optional[str] = None
|
||||
|
||||
|
||||
# How many of a brand's rows to read when working out what they agree on.
|
||||
# Bounded because this is a nicety, not the write: the largest brand table here
|
||||
# holds ~250 rows, so this reads all of them for every real brand while still
|
||||
# refusing to degenerate into the full `SELECT *` that once ran per uploaded
|
||||
# row. One query per brand, not per row - that part is load-bearing.
|
||||
_CONSENSUS_SAMPLE_LIMIT = 300
|
||||
|
||||
|
||||
def _brand_sample(brand_parent: str) -> Dict[str, Any]:
|
||||
"""The most recently updated product for a brand, used to inherit defaults.
|
||||
|
||||
`limit=1` matters: this used to be a full `SELECT *` of the brand table,
|
||||
executed once per uploaded row. A 200-row file against a brand with a few
|
||||
thousand products meant 200 full table reads before a single insert.
|
||||
Kept for the fields where one arbitrary sibling is a defensible default
|
||||
(category, price band, size). For `fssai_license` and `providers` it is
|
||||
NOT defensible - see `_brand_defaults`.
|
||||
"""
|
||||
try:
|
||||
existing = get_products_by_brand(brand_parent, limit=1)
|
||||
@@ -424,8 +437,51 @@ def _brand_sample(brand_parent: str) -> Dict[str, Any]:
|
||||
return existing[0] if existing else {}
|
||||
|
||||
|
||||
def _brand_defaults(brand_parent: str) -> Dict[str, Any]:
|
||||
"""One arbitrary sibling row, plus what the brand's rows actually AGREE on.
|
||||
|
||||
The distinction matters for exactly two fields.
|
||||
|
||||
`fssai_license` identifies the food business legally answerable for the
|
||||
product. Inheriting it from one arbitrary sibling is already weak; the code
|
||||
this replaces was worse - it fell back to the bare constant
|
||||
"10012042000244", which is Lion Dates' real registered licence, and stamped
|
||||
it onto any brand with no sample row. `scripts/merge_haldiram.py` exists
|
||||
because that reached production.
|
||||
|
||||
`providers` is a claim about where a product can be bought. The replaced
|
||||
default asserted all six of Amazon/Flipkart/BigBasket/Jiomart/Blinkit/Zepto
|
||||
for every product nobody had checked.
|
||||
|
||||
Consensus over the brand's own rows is the honest version of both, and it
|
||||
declines to answer when the rows disagree.
|
||||
"""
|
||||
# Two reads on purpose, and the split matters on a memory-capped host.
|
||||
#
|
||||
# `_brand_sample` is SELECT * limit 1 - one row, embedding and all, because
|
||||
# the fields it seeds (category, price band, size) need the whole row.
|
||||
#
|
||||
# The consensus read is 300 rows, so it takes only the two columns it
|
||||
# actually inspects. Measured: SELECT * over 244 rows costs 3.0 MB, of
|
||||
# which 4.7 KB per row is an embedding string nothing here reads. Two named
|
||||
# columns is roughly 50 KB for the same rows.
|
||||
rows = consensus_rows(brand_parent, ["fssai_license", "providers"],
|
||||
limit=_CONSENSUS_SAMPLE_LIMIT)
|
||||
|
||||
fssai, fssai_source = fssai_for_brand(brand_parent, rows)
|
||||
providers, _why = consensus_value("providers", rows)
|
||||
|
||||
return {
|
||||
"sample": _brand_sample(brand_parent),
|
||||
"fssai_license": fssai,
|
||||
"fssai_source": fssai_source,
|
||||
"providers": providers,
|
||||
}
|
||||
|
||||
|
||||
def _build_product_dict(req: AddProductRequest, brand_parent: str,
|
||||
sample_existing: Dict[str, Any]) -> Dict[str, Any]:
|
||||
sample_existing: Dict[str, Any],
|
||||
defaults: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
|
||||
"""Fill in everything the catalog needs that the user did not supply.
|
||||
|
||||
Pure apart from the optional S3 image lookup - no database access and no
|
||||
@@ -438,8 +494,22 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
|
||||
raise ValueError(f"product name {product_name!r} has no letters or digits to identify it by")
|
||||
image_id = f"{brand_slug}_{product_slug}"
|
||||
|
||||
defaults = defaults or {}
|
||||
category = req.category or sample_existing.get("category") or "Health Foods"
|
||||
fssai_license = req.fssai_license or sample_existing.get("fssai_license") or "10012042000244"
|
||||
|
||||
# NO CONSTANT FALLBACK HERE, EVER.
|
||||
#
|
||||
# This line used to end `or "10012042000244"` - Lion Dates' real registered
|
||||
# FSSAI licence - so any brand without a sample row was silently attributed
|
||||
# to a food business that had never heard of the product. An FSSAI number
|
||||
# is who is legally answerable for what is in the packet; inventing one is
|
||||
# not a cosmetic default.
|
||||
#
|
||||
# The order now is: what the uploader supplied, then the curated brand
|
||||
# registry, then what the brand's own rows unambiguously agree on, then
|
||||
# NOTHING. A blank licence is a gap someone can fill; a confidently wrong
|
||||
# one is a liability nobody knows to look for.
|
||||
fssai_license = req.fssai_license or defaults.get("fssai_license") or None
|
||||
|
||||
description = req.description or (
|
||||
f"Introducing {product_name} from the trusted {brand_parent} brand. "
|
||||
@@ -465,7 +535,11 @@ def _build_product_dict(req: AddProductRequest, brand_parent: str,
|
||||
else:
|
||||
price_range = "₹100-250"
|
||||
|
||||
providers = req.providers or list(sample_existing.get("providers") or ["Amazon", "Flipkart", "BigBasket", "Jiomart", "Blinkit", "Zepto"])
|
||||
# Likewise no invented marketplace list. Claiming a product is stocked by
|
||||
# Amazon, Flipkart, BigBasket, Jiomart, Blinkit AND Zepto because nobody
|
||||
# checked is a false availability claim on every row it touches. Consensus
|
||||
# across the brand's own rows, or empty.
|
||||
providers = req.providers or list(defaults.get("providers") or [])
|
||||
highlights = req.highlights or list(sample_existing.get("highlights") or ["100% Quality Assurance", "Authentic Brand Product"])
|
||||
nutrients = req.nutrients or list(sample_existing.get("nutrients") or ["Energy - High", "Protein - Good Source"])
|
||||
|
||||
@@ -582,11 +656,13 @@ def _persist_products(items: List[Tuple[Optional[int], AddProductRequest]]) -> _
|
||||
# row) and one embedding call for the whole upload.
|
||||
built: "OrderedDict[str, List[Tuple[Optional[int], AddProductRequest, Dict[str, Any]]]]" = OrderedDict()
|
||||
for brand_parent, rows in groups.items():
|
||||
sample = _brand_sample(brand_parent)
|
||||
defaults = _brand_defaults(brand_parent)
|
||||
sample = defaults["sample"]
|
||||
prepared = []
|
||||
for row_number, req in rows:
|
||||
try:
|
||||
prepared.append((row_number, req, _build_product_dict(req, brand_parent, sample)))
|
||||
prepared.append((row_number, req,
|
||||
_build_product_dict(req, brand_parent, sample, defaults)))
|
||||
except Exception as exc: # noqa: BLE001 - one bad row, not the file
|
||||
outcome.failures.append({
|
||||
"row": row_number,
|
||||
|
||||
Reference in New Issue
Block a user