Add Dagster orchestration and reduce active brands in backend
This commit is contained in:
@@ -87,7 +87,13 @@ class SeedResponse(BaseModel):
|
||||
class TrainRequest(BaseModel):
|
||||
models: Optional[List[str]] = Field(
|
||||
default=None,
|
||||
description="Subset of models to (re)train: discount, trending, popularity, forecast, store_performance, purchase_propensity. Omit to train all.",
|
||||
description=(
|
||||
"Subset of models to (re)train. Any of: discount, trending, popularity, "
|
||||
"forecast, store_performance, purchase_propensity. Omit to train the "
|
||||
"models that are actually served (discount, trending, popularity) - the "
|
||||
"other three fit fine but nothing reads their output back, so ask for "
|
||||
"them by name."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -1,66 +0,0 @@
|
||||
{
|
||||
"brand": "lion dates",
|
||||
"search_query": "Lion Dates products catalog",
|
||||
"generation_timestamp": "C:\\Brand_Catalog_LLM\\RAG_Model_Nutrition_Intelligence\\RAG_Model_Full_Implement\\backend\\app\\api\\routers\\user_products.py",
|
||||
"total_products": 1,
|
||||
"total_images": 10,
|
||||
"products": [
|
||||
{
|
||||
"image_id": "lion_dates_lion_dates_450g",
|
||||
"product_name": "Lion Dates 450g",
|
||||
"title": "Lion Dates 450g",
|
||||
"brand": "Lion Dates",
|
||||
"brand_name": "Lion Dates",
|
||||
"category": "Health Foods",
|
||||
"description": "Introducing Lion Dates 450g from the trusted Lion Dates brand. A premium quality product offering superior taste, authentic ingredients, and reliable value. Backed by Lion Dates's reputation for quality and consistency.",
|
||||
"price_range": "₹160-220",
|
||||
"size_variants": [
|
||||
"450g"
|
||||
],
|
||||
"providers": [
|
||||
"Amazon",
|
||||
"Flipkart",
|
||||
"BigBasket"
|
||||
],
|
||||
"highlights": [
|
||||
"lion dates Brand - Trusted Quality",
|
||||
"Food - Spreads Category",
|
||||
"Affordable at ₹9-11",
|
||||
"Available in 10g",
|
||||
"Premium Quality",
|
||||
"Available on 3 platforms"
|
||||
],
|
||||
"nutrients": [
|
||||
"Vitamin E - Antioxidant protection",
|
||||
"Omega-3 - Heart health",
|
||||
"Dietary Fiber - Digestive health",
|
||||
"Protein - Muscle building",
|
||||
"Magnesium - Muscle function",
|
||||
"Carbohydrates - Quick energy",
|
||||
"Healthy Fats - Heart health"
|
||||
],
|
||||
"fssai_license": "10012042000244",
|
||||
"product_sku": "LION-LION_D-001",
|
||||
"sku_source": "User Upload",
|
||||
"hsn_code": "2008",
|
||||
"final_selling_price": 185.0,
|
||||
"selling_price": 185.0,
|
||||
"barcode": "20086040",
|
||||
"barcode_type": "GTIN-13",
|
||||
"image_url": "https://liondates.com/cdn/shop/files/1.Datesinhoney_productfocus.png?v=1773383963&width=1445",
|
||||
"image_urls": [
|
||||
"https://liondates.com/cdn/shop/files/1.Datesinhoney_productfocus.png?v=1773383963&width=1445",
|
||||
"https://liondates.com/cdn/shop/files/dateshoney_1.jpg?v=1739704587&width=1080",
|
||||
"https://liondates.com/cdn/shop/files/Lion-Fig-in-Honey-Lion-Dates-95545533.jpg?v=1716380700&width=720",
|
||||
"https://liondates.com/cdn/shop/files/Lion-Fig-in-Honey-Lion-Dates-95545649.jpg?v=1739705462&width=1080",
|
||||
"http://liondates.com/cdn/shop/files/Lion-Mixed-Nuts-in-Honey-Lion-Dates-95787173.jpg?v=1716438485",
|
||||
"http://liondates.com/cdn/shop/files/Lion-Amla-in-Honey-Lion-Dates-95589650.jpg?v=1716381007",
|
||||
"https://liondates.com/cdn/shop/files/2.Datesinhoney_benefits.png?v=1773383963&width=390",
|
||||
"https://liondates.com/cdn/shop/files/Arabian_dates_500g_front.png?v=1739617157&width=1838",
|
||||
"https://liondates.com/cdn/shop/files/Sukkari_dates_front.png?v=1739615376&width=2048",
|
||||
"https://5.imimg.com/data5/SELLER/Default/2023/6/312791417/NH/RZ/CD/180805796/lion-honey-dates-250x250.webp"
|
||||
],
|
||||
"search_query": "Lion Dates Lion Dates 450g Health Foods Introducing Lion Dates 450g from the trusted Lion Dates brand. A premium quality product offering superior taste, authentic ingredients, and reliable value. Backed by Lion Dates's reputation for quality and consistency. ₹160-220"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -152,6 +152,23 @@ DATABASE_URL = os.getenv(
|
||||
# 0 disables the loop, leaving the startup run and POST /api/system/brand-sync.
|
||||
BRAND_SYNC_INTERVAL_SECONDS = int(os.getenv("BRAND_SYNC_INTERVAL_SECONDS", "300"))
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Active brands (development working set)
|
||||
# ---------------------------------------------------------------------------
|
||||
# Comma-separated brand names that the application and every pipeline operate
|
||||
# on. BLANK OR UNSET MEANS EVERY BRAND IS ACTIVE - that is the backwards
|
||||
# compatible default and the way to switch this feature off again.
|
||||
#
|
||||
# Nothing is deleted when this is set: the other brand_* tables and their
|
||||
# embeddings stay in the database untouched, they simply stop being discovered.
|
||||
# Going from 3 brands to 5, 10 or all of them is an edit to this one line.
|
||||
#
|
||||
# Names are resolved through resolve_parent_brand + _sanitize_name, the same
|
||||
# two steps that pick a product's storage table, so "Tata" here activates
|
||||
# brand_hindustan_unilever exactly as ingesting Tata products would.
|
||||
# See app/services/active_brands.py.
|
||||
ACTIVE_BRANDS = os.getenv("ACTIVE_BRANDS", "")
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# S3 / DigitalOcean Spaces (product image storage) - optional
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
24
app/intelligence/artifacts/archive/README.md
Normal file
24
app/intelligence/artifacts/archive/README.md
Normal file
@@ -0,0 +1,24 @@
|
||||
# Archived model artifacts
|
||||
|
||||
These three models still have **all of their training code, CLI flags and API
|
||||
options intact**. Only the pre-trained `.joblib` bundles were moved out of the
|
||||
loaded directory, because nothing in the application ever reads them back.
|
||||
|
||||
| Artifact | Why archived |
|
||||
|---|---|
|
||||
| `demand_forecast_model.joblib` | `train_forecast` fits it and writes the `demand_forecast` table, but `store_db.get_latest_demand_forecast()` has **zero callers** - no endpoint, service or MCP tool consumes the forecast. |
|
||||
| `store_performance_model.joblib` | Referenced only from `ml_training_service.train_store_performance`. No inference consumer. |
|
||||
| `purchase_propensity_model.joblib` | Referenced only from `ml_training_service.train_purchase_propensity`. No inference consumer. |
|
||||
|
||||
They are excluded from `train_all()`'s **default** set
|
||||
(`ml_training_service.PRODUCTION_MODELS`), not from the codebase. To rebuild one:
|
||||
|
||||
```
|
||||
python scripts/train_ml_models.py --models store_performance
|
||||
```
|
||||
|
||||
or `POST /api/admin/store-intelligence/train {"models": ["store_performance"]}`.
|
||||
|
||||
Training writes to `MODEL_ARTIFACTS_DIR` (the parent directory), so a retrain
|
||||
promotes the model back to production automatically - wire up an endpoint that
|
||||
reads it first, or it will simply sit there unread again.
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
130
app/services/active_brands.py
Normal file
130
app/services/active_brands.py
Normal file
@@ -0,0 +1,130 @@
|
||||
"""The single source of truth for which brands the application works on.
|
||||
|
||||
WHY THIS EXISTS
|
||||
---------------
|
||||
The dev dataset grew to 31 seed catalogs and 16 `brand_*` tables. Every
|
||||
"all brands" query fans out across every table, boot parses ~19MB of seed
|
||||
JSON, and 14 of those seed files have no table yet - so the next auto-seed
|
||||
would silently create 14 more. That is far more than an 8GB dev machine
|
||||
needs to exercise the features.
|
||||
|
||||
Rather than delete data, ONE setting decides which brands participate:
|
||||
|
||||
ACTIVE_BRANDS=Amul,Cadbury,Hindustan Unilever
|
||||
|
||||
and every read path, pipeline, RAG query, MCP tool and Dagster asset
|
||||
inherits it, because they all funnel through `vector_store`'s two brand
|
||||
discovery functions (`list_available_brands` and `_list_brand_table_suffixes`)
|
||||
plus `brand_sync.load_seed_catalogs`. Going back to 5, 10 or 25 brands is a
|
||||
one-line `.env` change, not a code edit.
|
||||
|
||||
THE EMPTY-MEANS-ALL CONTRACT
|
||||
----------------------------
|
||||
An unset or blank `ACTIVE_BRANDS` disables filtering entirely, so this module
|
||||
is a no-op on any deployment that does not opt in. That is deliberate:
|
||||
`.env.production` leaves it unset, so production keeps serving every brand
|
||||
while local development runs on three. It also means the feature can be
|
||||
switched off wholesale if it ever gets in the way.
|
||||
|
||||
NAMES ARE RESOLVED, NOT MATCHED
|
||||
-------------------------------
|
||||
Configured names go through `resolve_parent_brand` -> `_sanitize_name`, the
|
||||
same two steps that decide which table a product is stored in. So
|
||||
`ACTIVE_BRANDS=Tata` activates `brand_hindustan_unilever` (the "hul tata tea"
|
||||
alias claims it), exactly like an ingest of that brand would. Comparing raw
|
||||
strings here would have let the config and the storage layer disagree.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import FrozenSet, Iterable, List, Optional
|
||||
|
||||
from app.infrastructure.settings import ACTIVE_BRANDS as _RAW_ACTIVE_BRANDS
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
|
||||
# Parsed once. `_CACHE` holds `(suffixes, display_names)`; `None` in slot 0
|
||||
# means "no filtering configured", which is different from "an empty set of
|
||||
# active brands" - the latter would hide the entire catalog.
|
||||
_CACHE: Optional[tuple] = None
|
||||
|
||||
|
||||
def _sanitize(name: str) -> str:
|
||||
"""Brand name -> table suffix.
|
||||
|
||||
Imported lazily from vector_store because vector_store imports THIS module
|
||||
at the top level; a top-level import here would be a cycle. query_intent
|
||||
already breaks the identical cycle the same way.
|
||||
"""
|
||||
from app.services.vector_store import _sanitize_name
|
||||
|
||||
return _sanitize_name(name)
|
||||
|
||||
|
||||
def _parse(raw: str) -> tuple:
|
||||
names = [part.strip() for part in (raw or "").split(",")]
|
||||
names = [n for n in names if n]
|
||||
if not names:
|
||||
return (None, [])
|
||||
|
||||
suffixes = []
|
||||
display = []
|
||||
for name in names:
|
||||
suffix = _sanitize(resolve_parent_brand(name))
|
||||
if not suffix or suffix in suffixes:
|
||||
continue
|
||||
suffixes.append(suffix)
|
||||
display.append(name)
|
||||
return (frozenset(suffixes), display)
|
||||
|
||||
|
||||
def _load() -> tuple:
|
||||
global _CACHE
|
||||
if _CACHE is None:
|
||||
_CACHE = _parse(_RAW_ACTIVE_BRANDS)
|
||||
return _CACHE
|
||||
|
||||
|
||||
def invalidate() -> None:
|
||||
"""Drop the parsed config. For tests that monkeypatch the raw setting."""
|
||||
global _CACHE
|
||||
_CACHE = None
|
||||
|
||||
|
||||
def filtering_enabled() -> bool:
|
||||
"""True when ACTIVE_BRANDS is set to a non-empty list."""
|
||||
return _load()[0] is not None
|
||||
|
||||
|
||||
def active_brand_suffixes() -> Optional[FrozenSet[str]]:
|
||||
"""Table suffixes of the active brands, or None when filtering is off."""
|
||||
return _load()[0]
|
||||
|
||||
|
||||
def is_active_suffix(suffix: str) -> bool:
|
||||
"""Whether a `brand_<suffix>` table participates. True for all when off."""
|
||||
active = _load()[0]
|
||||
return True if active is None else (suffix or "").lower() in active
|
||||
|
||||
|
||||
def filter_suffixes(suffixes: Iterable[str]) -> List[str]:
|
||||
"""Keep only the active suffixes, preserving the caller's order."""
|
||||
active = _load()[0]
|
||||
if active is None:
|
||||
return list(suffixes)
|
||||
return [s for s in suffixes if (s or "").lower() in active]
|
||||
|
||||
|
||||
def is_active_brand(brand: str) -> bool:
|
||||
"""Whether a brand NAME (in any alias form) resolves to an active table."""
|
||||
active = _load()[0]
|
||||
if active is None:
|
||||
return True
|
||||
return _sanitize(resolve_parent_brand(brand)) in active
|
||||
|
||||
|
||||
def active_display_names() -> List[str]:
|
||||
"""The configured names, verbatim, for logs and Dagster partition keys.
|
||||
|
||||
Empty when filtering is off - callers that need the real brand list in
|
||||
that case should ask `vector_store.list_available_brands()` instead.
|
||||
"""
|
||||
return list(_load()[1])
|
||||
@@ -29,6 +29,11 @@ from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from app.services.brand_registry import resolve_parent_brand
|
||||
from app.services.active_brands import (
|
||||
active_brand_suffixes,
|
||||
is_active_brand,
|
||||
is_active_suffix,
|
||||
)
|
||||
from app.services.vector_store import (
|
||||
_connect,
|
||||
_list_brand_table_suffixes,
|
||||
@@ -45,6 +50,28 @@ logger = logging.getLogger(__name__)
|
||||
# app/services/brand_sync.py -> parents[2] is backend/
|
||||
SEED_DIR = Path(__file__).resolve().parents[2] / "data" / "seed_catalogs"
|
||||
|
||||
# Catalogs for brands outside ACTIVE_BRANDS live in this subdirectory. It is a
|
||||
# plain subfolder rather than a separate tree so the two stay side by side, and
|
||||
# it is NOT matched by SEED_DIR.glob("*.json") - which is what keeps the boot
|
||||
# auto-seed from parsing ~19MB and creating a table for every archived brand.
|
||||
ARCHIVE_DIR_NAME = "archive"
|
||||
|
||||
|
||||
def seed_catalog_paths(seed_dir: Path = SEED_DIR) -> List[Path]:
|
||||
"""Every seed catalog, active directory first, then the archive.
|
||||
|
||||
The archive is searched too, deliberately: re-activating a brand must be a
|
||||
one-line ACTIVE_BRANDS change, so where a file physically sits is
|
||||
presentation, not policy. Filtering by brand happens in load_seed_catalogs.
|
||||
"""
|
||||
if not seed_dir.exists():
|
||||
return []
|
||||
paths = sorted(seed_dir.glob("*.json"))
|
||||
archive = seed_dir / ARCHIVE_DIR_NAME
|
||||
if archive.is_dir():
|
||||
paths.extend(sorted(archive.glob("*.json")))
|
||||
return paths
|
||||
|
||||
# The column list written by upsert_brand_products, minus `embedding` (handled
|
||||
# separately because it is huge and optional). Keeping these in sync is what
|
||||
# makes an exported file round-trip back through the seeder without losing
|
||||
@@ -95,7 +122,7 @@ def index_seed_files(seed_dir: Path = SEED_DIR) -> Dict[str, List[Path]]:
|
||||
index: Dict[str, List[Path]] = defaultdict(list)
|
||||
if not seed_dir.exists():
|
||||
return {}
|
||||
for path in sorted(seed_dir.glob("*.json")):
|
||||
for path in seed_catalog_paths(seed_dir):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
@@ -267,25 +294,69 @@ def load_seed_catalogs(seed_dir: Path = SEED_DIR,
|
||||
logger.error("Seed directory not found: %s", seed_dir)
|
||||
return {}
|
||||
|
||||
files = sorted(seed_dir.glob("*.json"))
|
||||
files = seed_catalog_paths(seed_dir)
|
||||
if only:
|
||||
wanted = [w.lower() for w in only]
|
||||
files = [f for f in files if any(w in f.name.lower() for w in wanted)]
|
||||
|
||||
# `only` is an explicit request for named files, so it wins over the
|
||||
# ACTIVE_BRANDS filter - that is how an archived brand can still be
|
||||
# re-ingested deliberately without first editing the config.
|
||||
respect_active = not only
|
||||
|
||||
brand_products: Dict[str, List[Dict[str, Any]]] = defaultdict(list)
|
||||
skipped_inactive = 0
|
||||
for path in files:
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
logger.warning("Skipping %s - no brand/products found", path.name)
|
||||
continue
|
||||
resolved = resolve_parent_brand(data["brand"])
|
||||
if respect_active and not is_active_brand(resolved):
|
||||
skipped_inactive += 1
|
||||
continue
|
||||
brand_products[resolved].extend(data["products"])
|
||||
logger.info("Read %d products from %s -> resolved brand '%s'",
|
||||
len(data["products"]), path.name, resolved)
|
||||
|
||||
if skipped_inactive:
|
||||
logger.info("Skipped %d seed catalog(s) outside ACTIVE_BRANDS", skipped_inactive)
|
||||
|
||||
return dict(brand_products)
|
||||
|
||||
|
||||
def load_brand_products(brand: str) -> List[Dict[str, Any]]:
|
||||
"""Every seed product belonging to `brand`, resolved properly.
|
||||
|
||||
Use this instead of `load_seed_catalogs(only=[brand])` when you have a
|
||||
BRAND NAME. `only=` is a case-insensitive substring match on the FILE NAME,
|
||||
which is the right thing for `seed_sample_data.py --only amul` but the
|
||||
wrong thing for a brand:
|
||||
|
||||
* "Hindustan Unilever" contains a space; the file is
|
||||
`brand_catalog_hindustan_unilever.json`. The substring never matches
|
||||
and you silently get zero products - no error, just an empty result.
|
||||
* Even with the slug, a filename match misses the other files that feed
|
||||
the same table: `brand_catalog_tata.json` holds 121 Hindustan Unilever
|
||||
products because the "hul tata tea" alias claims it.
|
||||
|
||||
Going through `index_seed_files()` fixes both: it is keyed by the resolved
|
||||
table suffix and its value is the full list of files feeding that table.
|
||||
Archived catalogs are included, so an explicitly requested brand is found
|
||||
whether or not it is currently active.
|
||||
"""
|
||||
index = index_seed_files()
|
||||
products: List[Dict[str, Any]] = []
|
||||
for path in index.get(brand_slug(brand), []):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
products.extend(data["products"])
|
||||
logger.info("Read %d products from %s for brand '%s'",
|
||||
len(data["products"]), path.name, brand)
|
||||
return products
|
||||
|
||||
|
||||
def seed_brands(brand_products: Dict[str, List[Dict[str, Any]]], cleanup: bool = True) -> int:
|
||||
"""Upsert grouped products into their brand tables. Returns the row count."""
|
||||
total = 0
|
||||
@@ -339,7 +410,7 @@ def _detect_collisions(seed_dir: Path = SEED_DIR) -> List[Dict[str, Any]]:
|
||||
by_slug: Dict[str, set] = defaultdict(set)
|
||||
if not seed_dir.exists():
|
||||
return []
|
||||
for path in sorted(seed_dir.glob("*.json")):
|
||||
for path in seed_catalog_paths(seed_dir):
|
||||
data = _read_catalog(path)
|
||||
if data is None:
|
||||
continue
|
||||
@@ -354,6 +425,9 @@ def _detect_collisions(seed_dir: Path = SEED_DIR) -> List[Dict[str, Any]]:
|
||||
def reconcile_brand_catalogs(dry_run: bool = False) -> Dict[str, Any]:
|
||||
"""Repair the table <-> seed-file correspondence in both directions.
|
||||
|
||||
Scoped to ACTIVE_BRANDS when that is set: an archived brand is neither
|
||||
exported nor seeded, in either direction.
|
||||
|
||||
Idempotent and non-destructive:
|
||||
|
||||
* A populated table with no seed file gets one exported.
|
||||
@@ -373,6 +447,22 @@ def reconcile_brand_catalogs(dry_run: bool = False) -> Dict[str, Any]:
|
||||
file_index = index_seed_files()
|
||||
collisions = _detect_collisions()
|
||||
|
||||
# BOTH SIDES MUST BE NARROWED TO THE ACTIVE BRANDS, OR NEITHER.
|
||||
#
|
||||
# _db_brand_counts() goes through _list_brand_table_suffixes(), so under
|
||||
# ACTIVE_BRANDS it only sees the active tables. index_seed_files() reads the
|
||||
# archive directory too, deliberately, so that re-activating a brand needs
|
||||
# only a config change.
|
||||
#
|
||||
# Left mismatched, every archived brand looks like "a seed file whose table
|
||||
# is empty" and lands in `to_seed` - so the boot reconcile would re-seed all
|
||||
# 27 archived catalogs on the next restart, recreate their tables, and
|
||||
# silently undo the archiving. Verified: a dry run reported exactly that.
|
||||
if active_brand_suffixes() is not None:
|
||||
file_index = {
|
||||
slug: paths for slug, paths in file_index.items() if is_active_suffix(slug)
|
||||
}
|
||||
|
||||
to_export = sorted(
|
||||
suffix for suffix, count in db_counts.items()
|
||||
if count > 0 and suffix not in file_index
|
||||
|
||||
@@ -24,8 +24,25 @@ from app.services.discount_service import predict_discounts_for_store
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Every model this module knows how to train. Still the full set accepted by
|
||||
# scripts/train_ml_models.py and POST /api/admin/store-intelligence/train.
|
||||
ALL_MODELS = ["discount", "trending", "popularity", "forecast", "store_performance", "purchase_propensity"]
|
||||
|
||||
# What train_all() trains when the caller does not name a subset.
|
||||
#
|
||||
# `forecast`, `store_performance` and `purchase_propensity` are omitted
|
||||
# deliberately: all three fit cleanly, but nothing reads their output back.
|
||||
# store_db.get_latest_demand_forecast() has zero callers, and neither of the
|
||||
# other two has an inference consumer anywhere in the app or the MCP tools.
|
||||
# Training them by default spent roughly half of every retrain, and ~5MB of
|
||||
# artifacts, producing bundles no request would ever load.
|
||||
#
|
||||
# Their training code, CLI flags and API options are untouched - ask for one by
|
||||
# name (`--models store_performance`) and it trains and is written to
|
||||
# MODEL_ARTIFACTS_DIR exactly as before. Move a model back into this list once
|
||||
# something actually serves it.
|
||||
PRODUCTION_MODELS = ["discount", "trending", "popularity"]
|
||||
|
||||
|
||||
def _load_common_frames():
|
||||
order_items = store_db.get_order_items_df()
|
||||
@@ -144,7 +161,11 @@ def train_purchase_propensity(orders: pd.DataFrame) -> Dict:
|
||||
|
||||
|
||||
def train_all(models: Optional[List[str]] = None) -> Dict[str, Dict]:
|
||||
models = models or ALL_MODELS
|
||||
"""Train `models`, defaulting to the models that are actually served.
|
||||
|
||||
Pass an explicit list (including any of ALL_MODELS) to train more.
|
||||
"""
|
||||
models = models or PRODUCTION_MODELS
|
||||
order_items, orders, store_products = _load_common_frames()
|
||||
results: Dict[str, Dict] = {}
|
||||
|
||||
|
||||
@@ -205,8 +205,16 @@ def _brand_index() -> tuple[list[str], Dict[str, str]]:
|
||||
if _MENTION_CACHE["map"] is not None and now - _MENTION_CACHE["at"] < BRAND_MENTION_TTL_SECONDS:
|
||||
return _MENTION_CACHE["known"], _MENTION_CACHE["map"]
|
||||
|
||||
known = list(KNOWN_BRANDS)
|
||||
mapping = dict(BRAND_SEARCH_MAP)
|
||||
# KNOWN_BRANDS/BRAND_SEARCH_MAP are a hand-tuned static table of 16 brands.
|
||||
# Under ACTIVE_BRANDS they must be narrowed too, not just added to: without
|
||||
# this a query for an archived brand ("Nestle") still parses as a brand
|
||||
# mention, gets routed to brand_catalog mode, and returns zero rows - the
|
||||
# same silent wrong-result failure mode as fuzzy category detection. Dropped
|
||||
# here, it is treated as ordinary search text instead.
|
||||
from app.services.active_brands import is_active_brand
|
||||
|
||||
known = [b for b in KNOWN_BRANDS if is_active_brand(b)]
|
||||
mapping = {k: v for k, v in BRAND_SEARCH_MAP.items() if is_active_brand(v)}
|
||||
try:
|
||||
from app.services.vector_store import list_available_brands
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ from app.infrastructure.settings import (
|
||||
DB_CONNECT_TIMEOUT_SECONDS,
|
||||
)
|
||||
from app.services.brand_registry import BRAND_ALIASES, resolve_parent_brand
|
||||
from app.services.active_brands import filter_suffixes, is_active_suffix
|
||||
|
||||
|
||||
from app.services.s3_service import s3_service
|
||||
@@ -562,6 +563,12 @@ def list_available_brands() -> List[str]:
|
||||
# Extract brand name from table name (e.g., brand_britannia -> britannia)
|
||||
table_name = table[0]
|
||||
suffix = table_name[len("brand_"):].lower() if table_name.startswith("brand_") else table_name.lower()
|
||||
# ACTIVE_BRANDS narrows the whole app here: this list feeds
|
||||
# /api/brands, the MCP list_brands tool, rag_service,
|
||||
# query_intent's brand index, nutrition enrichment, store_db
|
||||
# and get_products_all_brands. The table itself is untouched.
|
||||
if not is_active_suffix(suffix):
|
||||
continue
|
||||
# Prefer the original display name from the brand map if available
|
||||
brands.append(display_name_for_suffix(suffix, brand_map))
|
||||
except Exception as e:
|
||||
@@ -1220,12 +1227,23 @@ def _table_exists(cur, table_name: str) -> bool:
|
||||
return bool(row and row[0])
|
||||
|
||||
|
||||
def _list_brand_table_suffixes(cur) -> List[str]:
|
||||
"""Return brand-table suffixes (e.g. 'parle' from 'brand_parle') for every brand table."""
|
||||
def _list_brand_table_suffixes(cur, *, include_inactive: bool = False) -> List[str]:
|
||||
"""Return brand-table suffixes (e.g. 'parle' from 'brand_parle') for every brand table.
|
||||
|
||||
Narrowed to ACTIVE_BRANDS by default. This is the shared chokepoint for
|
||||
get_brand_overview, semantic_search, text_search, lexical_search,
|
||||
list_all_categories and brand_sync._db_brand_counts, which is why the
|
||||
active-brand setting reaches all of them without touching any of them.
|
||||
|
||||
`include_inactive=True` is the escape hatch for callers that address an
|
||||
archived brand deliberately - an explicit re-ingest or a backfill - so
|
||||
narrowing the catalog never means losing the ability to maintain it.
|
||||
"""
|
||||
cur.execute(
|
||||
"""
|
||||
SELECT table_name FROM information_schema.tables
|
||||
WHERE table_schema = 'public' AND table_name LIKE 'brand_%'
|
||||
"""
|
||||
)
|
||||
return [row[0][len("brand_"):] for row in cur.fetchall()]
|
||||
suffixes = [row[0][len("brand_"):] for row in cur.fetchall()]
|
||||
return suffixes if include_inactive else filter_suffixes(suffixes)
|
||||
|
||||
Reference in New Issue
Block a user